From 97fdd89f6a0b332f1b0602bd852666b3201703d1 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Sun, 4 Oct 2026 01:29:57 +0200 Subject: [PATCH 01/31] sync: "Keep mine" no longer deletes files; hold syncs that would delete a large part of a save "Keep mine" (and "Keep both", which ends the same way) recorded every local file as shared with the peer before the peer had pulled any of it. When that pull did not finish - a large save, a request that timed out, a device that went away - the next sync read the files the peer was missing as deletions made over there, and deleted them on the device whose version had just been kept. On a save of ~240k files this removed tens of thousands; the folders then differed again, the conflict came back, and the same answer did it again. - markResolvedLocal records as shared only what both sides verifiably hold (the intersection of the two manifests). A file only this device has is new to the peer, not deleted by it. If the peer cannot be asked, nothing is recorded as shared. Same for extra save locations (ResolveRootConflict). - A sync that would delete more than 100 files and more than 10% of a save, on either side, is held as a conflict instead of applied. A deletion confirmed on purpose (the emptied-folder flow) is not held. - The conflict carries uncapped counts of what differs (diffFiles stops at 100), and the dialog says what each answer does: "Keep mine"/"Keep both" delete nothing; "Keep theirs" deletes N files only this device has, shown in red when that is many. Older builds without the counts fall back to counting the capped list. TestKeepLocal_PeerThatNeverPulled_DeletesNothingHere fails on v2.4.1 (300 files -> 100); either change alone makes it pass. Also adds tests that an interrupted pull or a peer holding part of a save never turns into deletions. Co-Authored-By: Claude Opus 5.5 --- .../src/components/ConflictModal.svelte | 50 ++++++++- internal/p2p/syncengine/conflict.go | 33 +++++- internal/p2p/syncengine/engine.go | 36 ++++++- .../p2p/syncengine/keeplocal_safety_test.go | 34 ++++++ internal/p2p/syncengine/massdelete.go | 28 +++++ internal/p2p/syncengine/massdelete_test.go | 69 ++++++++++++ internal/p2p/syncengine/multiroot_conflict.go | 12 ++- .../p2p/syncengine/partial_safety_test.go | 101 ++++++++++++++++++ 8 files changed, 349 insertions(+), 14 deletions(-) create mode 100644 internal/p2p/syncengine/keeplocal_safety_test.go create mode 100644 internal/p2p/syncengine/massdelete.go create mode 100644 internal/p2p/syncengine/massdelete_test.go create mode 100644 internal/p2p/syncengine/partial_safety_test.go diff --git a/cmd/opensave-app/frontend/src/components/ConflictModal.svelte b/cmd/opensave-app/frontend/src/components/ConflictModal.svelte index 4a9311ff..b16e4151 100644 --- a/cmd/opensave-app/frontend/src/components/ConflictModal.svelte +++ b/cmd/opensave-app/frontend/src/components/ConflictModal.svelte @@ -103,9 +103,16 @@ $: diffFiles = conflict?.diffFiles ?? []; $: diffTotal = conflict?.diffTotal ?? 0; $: diffCapped = diffTotal > diffFiles.length; - $: changedCount = diffFiles.filter((d) => d.status === 'changed').length; - $: onlyLocalCount = diffFiles.filter((d) => d.status === 'only-local').length; - $: onlyRemoteCount = diffFiles.filter((d) => d.status === 'only-remote').length; + // The uncapped counts when the device sends them; counted from the capped + // list otherwise (an older build), which is exact below the cap. + const countOr = (n, status) => (typeof n === 'number' ? n : diffFiles.filter((d) => d.status === status).length); + $: changedCount = countOr(conflict?.changedTotal, 'changed'); + $: onlyLocalCount = countOr(conflict?.onlyLocalTotal, 'only-local'); + $: onlyRemoteCount = countOr(conflict?.onlyRemoteTotal, 'only-remote'); + // "Keep theirs" deletes every file only this device has. Said before the + // click, in numbers, and loudly once it is a lot. + $: theirsDeletes = onlyLocalCount; + $: theirsDeletesMany = theirsDeletes >= 100; // Bytes that differ, per side: a changed file counts its own size on each // side, a file present on only one side counts only there. @@ -217,12 +224,28 @@
last change {fmtMs(remoteMs)}
- {#if diffCapped} + {#if diffCapped && typeof conflict.onlyLocalTotal !== 'number'}

Counts above cover the first {diffFiles.length} of {diffTotal} differing files.

{/if} + + {#if conflict.diffTotal > 0} - + @@ -362,6 +386,22 @@ color: var(--text-faint); margin: -8px 0 12px; } + .consequences { + margin: 0 0 14px; + padding-left: 18px; + font-size: 0.8rem; + color: var(--text-dim); + } + .consequences li { + margin: 2px 0; + } + .consequences li.danger, + .btn.danger { + color: var(--danger, #e5484d); + } + .btn.danger { + border-color: currentColor; + } .v-time { font-size: 0.74rem; color: var(--text-faint); diff --git a/internal/p2p/syncengine/conflict.go b/internal/p2p/syncengine/conflict.go index 66293b5a..1cef4b5b 100644 --- a/internal/p2p/syncengine/conflict.go +++ b/internal/p2p/syncengine/conflict.go @@ -41,7 +41,7 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution // see both sides still differing from the stale base and re-conflict // immediately — the "keep looping" bug. e.Log("info", fmt.Sprintf("conflict on %q resolved: keep LOCAL — our version becomes the shared state", gameID)) - if err := e.markResolvedLocal(gameID, peer); err != nil { + if err := e.markResolvedLocal(ctx, gameID, peer); err != nil { return "", err } e.clearConflict(gameID) @@ -103,7 +103,7 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution // From here it is "keep mine": this device's version becomes the // shared one, and the peer is asked to take it — which, if it has // changes of its own, asks the person there in turn. - if err := e.markResolvedLocal(gameID, peer); err != nil { + if err := e.markResolvedLocal(ctx, gameID, peer); err != nil { return "", err } e.clearConflict(gameID) @@ -160,7 +160,20 @@ func (e *Engine) markResolvedConverged(gameID string, peer Peer) { // unambiguously the newest, so the sync direction is a push. The peer // then receives our version (its own engine snapshots first, or re-raises // its own conflict if it has newer unsynced work of its own). -func (e *Engine) markResolvedLocal(gameID string, peer Peer) error { +// +// The lineage, though, must say only what both sides verifiably hold. It used +// to record every local file as shared (persistLineage(local, local)) before +// the peer had pulled any of it. When that pull did not finish — a large save, +// a request that timed out, a device that went away — the next sync found +// those files "shared before and missing on the peer now", read that as the +// peer having deleted them, and deleted them HERE: on the device whose version +// the person had just chosen to keep. A save of a quarter-million files lost +// tens of thousands that way, and the folders differing again re-raised the +// conflict, inviting the same answer. Now the record is the intersection of +// the two saves as they stand; a file only this device has is new to the +// peer, not deleted by it, and goes over on the next sync. If the peer cannot +// be asked, nothing is recorded as shared, so nothing can be read as deleted. +func (e *Engine) markResolvedLocal(ctx context.Context, gameID string, peer Peer) error { game, err := e.Store.GetGame(gameID) if err != nil { return err @@ -170,7 +183,19 @@ func (e *Engine) markResolvedLocal(gameID string, peer Peer) error { if err != nil { return err } - e.persistLineage(gameID, peer.ID, local, local) + var files, dirs []string + if remote, err := e.peerStateForResolution(ctx, gameID, peer); err == nil { + files, dirs = IntersectLineage(local, remote.Manifest) + } else { + e.Log("info", fmt.Sprintf("could not read %s's save while resolving %q (%v) — nothing is recorded as shared, so nothing it lacks is taken as deleted", peer.Name, gameID, err)) + } + if rules := e.rulesFor(gameID); !rules.Empty() { + files = filterPathList(files, rules) + dirs = filterPathList(dirs, rules) + } + if err := e.Store.SetSyncState(gameID, peer.ID, files, dirs); err != nil { + e.Log("warn", fmt.Sprintf("persist sync lineage failed: %v", err)) + } _ = e.Store.SetAgreedHash(gameID, peer.ID, local.ManifestHash()) _ = e.Store.SetLastManifestHash(gameID, e.contentHashOf(gameID, local, game.SavePath)) e.recordSynced(gameID, peer.ID) diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index e3b02b00..f60d67fe 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -27,6 +27,13 @@ type Conflict struct { RemoteStats SideStats `json:"remoteStats"` DiffFiles []DiffFile `json:"diffFiles"` // capped; DiffTotal is the real count DiffTotal int `json:"diffTotal"` + // The uncapped counts by kind. DiffFiles stops at 100, and counting from + // it told someone choosing "keep theirs" on a save of a quarter-million + // files that it would remove a few dozen, when it removed tens of + // thousands: every OnlyLocal file goes. + OnlyLocalTotal int `json:"onlyLocalTotal"` + OnlyRemoteTotal int `json:"onlyRemoteTotal"` + ChangedTotal int `json:"changedTotal"` } // SideStats summarises one side's save state for the conflict UI. @@ -673,6 +680,22 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re } handedOverDeletion := e.handOverEmptying(gameID, peer, localManifest.Files, &decision) + // A sync about to delete a large part of a save — here or on the peer — + // stops and asks instead. emptiedUnconfirmed above covers a save emptied + // completely; this covers one gutted partly, which is what a partial copy + // on one device turns into once it is read as the other's deletions. Two + // waves of that took tens of thousands of files out of a save before + // anyone was asked. A deletion someone confirmed on purpose + // (handedOverDeletion, DeletionConfirmed) is not second-guessed. + if !handedOverDeletion && !remoteData.DeletionConfirmed { + if n, total := massDeletion(localManifest, remoteData.Manifest, decision); n > 0 { + e.Log("warn", fmt.Sprintf("syncing %q with %q would delete %d of %d files — holding it for a decision instead", + game.Name, peer.Name, n, total)) + e.registerConflict(gameID, peer, localManifest, remoteData) + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, nil + } + } + // Nothing below is about files arriving, only about local files leaving. // A pull that brings files this device never held destroys nothing. atRisk := filesAtRisk(localManifest, decision) @@ -1260,6 +1283,10 @@ func (e *Engine) registerConflict(gameID string, peer Peer, localManifest delta. // Capture comparison data while we hold both manifests, so the UI can // show which side is further along and exactly what differs. diffs := diffManifests(localManifest, remoteData.Manifest) + counts := map[string]int{} + for _, d := range diffs { + counts[d.Status]++ + } const maxDiffFiles = 100 total := len(diffs) if len(diffs) > maxDiffFiles { @@ -1269,10 +1296,11 @@ func (e *Engine) registerConflict(gameID string, peer Peer, localManifest delta. e.mu.Lock() e.activeConflicts[gameID] = &Conflict{ Peer: peer, LocalSnap: localSnap, RemoteSnap: remoteSnap, - LocalStats: manifestStats(localManifest), - RemoteStats: manifestStats(remoteData.Manifest), - DiffFiles: diffs, - DiffTotal: total, + LocalStats: manifestStats(localManifest), + RemoteStats: manifestStats(remoteData.Manifest), + DiffFiles: diffs, + DiffTotal: total, + OnlyLocalTotal: counts["only-local"], OnlyRemoteTotal: counts["only-remote"], ChangedTotal: counts["changed"], } e.mu.Unlock() diff --git a/internal/p2p/syncengine/keeplocal_safety_test.go b/internal/p2p/syncengine/keeplocal_safety_test.go new file mode 100644 index 00000000..064298f5 --- /dev/null +++ b/internal/p2p/syncengine/keeplocal_safety_test.go @@ -0,0 +1,34 @@ +package syncengine + +import ( + "context" + "fmt" + "testing" +) + +// "Keep mine" against a peer holding part of the save, where the peer never +// gets to pull (a pull that times out, a device that goes away). The next +// sync must not read the peer's missing files as deletions and remove them +// from the device whose version the person chose to keep. +func TestKeepLocal_PeerThatNeverPulled_DeletesNothingHere(t *testing.T) { + env := setupEngine(t) + for i := 0; i < 300; i++ { + write(t, env.localDir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } + for i := 0; i < 100; i++ { + write(t, env.remoteDir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } + + // The person answers the conflict with "Keep mine". + if err := env.engine.markResolvedLocal(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + // The peer's pull never happens; its folder stays as it was. + + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + if n := countFiles(t, env.localDir); n != 300 { + t.Fatalf("after \"keep mine\" this device went from 300 to %d files — the peer's partial copy was read as deletions", n) + } +} diff --git a/internal/p2p/syncengine/massdelete.go b/internal/p2p/syncengine/massdelete.go new file mode 100644 index 00000000..aea0cc26 --- /dev/null +++ b/internal/p2p/syncengine/massdelete.go @@ -0,0 +1,28 @@ +package syncengine + +import "github.com/opensave/opensave/internal/delta" + +// A sync is held for a decision when it would delete more than both of these: +// enough files that it is not a game tidying its slots, and enough of the save +// that it is not one corner of a big one. Ordinary play deletes a handful of +// files; a device holding part of a save being read as the other's deletions +// deletes thousands. +const ( + massDeleteMinFiles = 100 + massDeleteFraction = 0.10 +) + +// massDeletion reports how many files the decision would delete, on either +// side, and how many the save holds, when that crosses the thresholds above; +// zero otherwise. +func massDeletion(local, remote delta.Manifest, d Decision) (deleting, total int) { + n := len(d.FilesToDeleteLocally) + len(d.FilesToDeleteOnPeer) + total = len(local.Files) + if len(remote.Files) > total { + total = len(remote.Files) + } + if n <= massDeleteMinFiles || total == 0 || float64(n) <= massDeleteFraction*float64(total) { + return 0, total + } + return n, total +} diff --git a/internal/p2p/syncengine/massdelete_test.go b/internal/p2p/syncengine/massdelete_test.go new file mode 100644 index 00000000..a18b34c7 --- /dev/null +++ b/internal/p2p/syncengine/massdelete_test.go @@ -0,0 +1,69 @@ +package syncengine + +import ( + "context" + "fmt" + "os" + "path/filepath" + "testing" +) + +// Both sides converge on n files, then the peer loses some of them. +func convergedThenPeerLoses(t *testing.T, n, lost int) *engineEnv { + t.Helper() + env := setupEngine(t) + writeSmallFiles(t, env.localDir, n) + writeSmallFiles(t, env.remoteDir, n) + if res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil || res.Status != "in_sync" { + t.Fatalf("setup sync = %+v, %v", res, err) + } + for i := 0; i < lost; i++ { + if err := os.Remove(filepath.Join(env.remoteDir, "world", fmt.Sprintf("map_%04d.bin", i))); err != nil { + t.Fatal(err) + } + } + return env +} + +func TestMassDeletion_IsHeldForADecision(t *testing.T) { + env := convergedThenPeerLoses(t, 1000, 500) + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatal(err) + } + if res.Status != "conflict" { + t.Fatalf("result = %+v, want the sync held as a conflict", res) + } + if n := countFiles(t, env.localDir); n != 1000 { + t.Fatalf("this side has %d files; the 500 deletions were applied instead of held", n) + } + if _, ok := env.engine.ActiveConflicts()["game1"]; !ok { + t.Error("no conflict registered to ask about") + } + + // "Keep mine": the peer's missing files go back to it, nothing is deleted here. + if _, err := env.engine.ResolveConflict(context.Background(), "game1", env.peer.ID, "keep-local"); err != nil { + t.Fatal(err) + } + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + if n := countFiles(t, env.localDir); n != 1000 { + t.Fatalf("after keep-mine this side has %d files, want 1000", n) + } +} + +// Ordinary play deletes a few files; that still just syncs. +func TestMassDeletion_SmallDeletionsStillSync(t *testing.T) { + env := convergedThenPeerLoses(t, 1000, 20) + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatal(err) + } + if res.Status == "conflict" { + t.Fatalf("20 of 1000 deletions were held: %+v", res) + } + if n := countFiles(t, env.localDir); n != 980 { + t.Fatalf("this side has %d files, want the peer's 20 deletions applied (980)", n) + } +} diff --git a/internal/p2p/syncengine/multiroot_conflict.go b/internal/p2p/syncengine/multiroot_conflict.go index 1fea1ba6..401d69de 100644 --- a/internal/p2p/syncengine/multiroot_conflict.go +++ b/internal/p2p/syncengine/multiroot_conflict.go @@ -140,7 +140,17 @@ func (e *Engine) ResolveRootConflict(ctx context.Context, gameID, peerID, root, if err := e.Store.SetAgreedHashForRoot(gameID, peerID, root, local.RootHash(delta.PrimaryRoot)); err != nil { return err } - files, dirs := IntersectLineage(local, local) + // Shared only what the peer verifiably holds too — see + // markResolvedLocal for what recording the whole local copy as shared + // did once the peer's pull did not finish. + var files, dirs []string + if remoteData, err := e.Transport.FetchManifest(ctx, peer, gameID, ManifestQuery{ + Name: game.Name, SavePath: game.SavePath, AppID: game.AppID, CoverURL: game.CoverURL, + }); err == nil { + if remoteRoot, has := remoteData.Manifest.Extra[root]; has { + files, dirs = IntersectLineage(local, delta.Manifest{Files: remoteRoot.Files, Dirs: remoteRoot.Dirs}) + } + } _ = e.Store.SetSyncStateForRoot(gameID, peerID, root, files, dirs) e.clearRootConflict(gameID, root) e.Transport.TriggerPeerPull(peer, gameID) diff --git a/internal/p2p/syncengine/partial_safety_test.go b/internal/p2p/syncengine/partial_safety_test.go new file mode 100644 index 00000000..fa85845d --- /dev/null +++ b/internal/p2p/syncengine/partial_safety_test.go @@ -0,0 +1,101 @@ +package syncengine + +import ( + "context" + "fmt" + "os" + "path/filepath" + "sync/atomic" + "testing" +) + +// A transfer that stops part-way must never turn into deletions. +// +// Seen in 2.3.1: a device that had received part of a 240k-file save offered +// its partial copy back, the missing files were read as "deleted over there", +// and 23k files were deleted on the device that still had them. 2.4 records +// as shared only what both manifests showed (persistLineage), which should +// rule that out; these hold it there. + +func countFiles(t *testing.T, dir string) int { + t.Helper() + n := 0 + _ = filepath.Walk(dir, func(p string, info os.FileInfo, err error) error { + if err == nil && !info.IsDir() { + n++ + } + return nil + }) + return n +} + +func writeSmallFiles(t *testing.T, dir string, n int) { + t.Helper() + for i := 0; i < n; i++ { + write(t, dir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } +} + +func interruptedThenResumed(t *testing.T, env *engineEnv, total, stopAfter int) { + t.Helper() + writeSmallFiles(t, env.remoteDir, total) + + ctx, cancel := context.WithCancel(context.Background()) + var fetched atomic.Int64 + env.transport.onFetchBlocks = func() { + if fetched.Add(1) == int64(stopAfter) { + cancel() + } + } + if _, err := env.engine.SyncWithPeer(ctx, "game1", env.peer); err == nil { + t.Fatal("setup: the first sync was meant to be interrupted") + } + env.transport.onFetchBlocks = nil + cancel() + + got := countFiles(t, env.localDir) + if got == 0 || got >= total { + t.Fatalf("setup: after the interruption this side has %d of %d files; want a partial copy", got, total) + } + + // Resume. Nothing may be deleted anywhere, and everything must arrive. + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("resumed sync: %v", err) + } + if len(env.transport.deletedOnPeer) > 0 { + t.Fatalf("the partial copy was propagated as %d deletions on the peer, e.g. %s", + len(env.transport.deletedOnPeer), env.transport.deletedOnPeer[0]) + } + if n := countFiles(t, env.remoteDir); n != total { + t.Fatalf("peer has %d files, want all %d", n, total) + } + if n := countFiles(t, env.localDir); n != total { + t.Fatalf("this side has %d files after resuming, want %d", n, total) + } +} + +func TestPartialPull_InterruptedThenResumed(t *testing.T) { + env := setupEngine(t) + interruptedThenResumed(t, env, 200, 60) +} + +// The other side of the same situation: this device has everything, the peer +// holds only part of it and never synced. The part it lacks is new to it, not +// deleted by it. +func TestPartialPeer_NeverReadAsDeletions(t *testing.T) { + env := setupEngine(t) + writeSmallFiles(t, env.localDir, 300) + for i := 0; i < 80; i++ { + write(t, env.remoteDir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } + + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + if n := countFiles(t, env.localDir); n != 300 { + t.Fatalf("this side went from 300 to %d files — the peer's partial copy was read as deletions", n) + } + if len(env.transport.deletedOnPeer) > 0 { + t.Fatalf("deleted on peer: %v", env.transport.deletedOnPeer[:1]) + } +} From fe9ce64394609a0b1c87b1ec19430d1098349464 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:00:24 +0200 Subject: [PATCH 02/31] sync: batch small files, skip unchanged manifests, probe presence on its own Performance - Batched pulls (protocol revision 2): small files are fetched up to 256 per request (4MB), 4 requests in flight, over a new /api/p2p/files route. Each file is still written through the PatchWriter (verified, fsynced, renamed). One request per file cost ~200ms on a real LAN; on loopback 5000 files went from 42s to 12s. Peers below revision 2 and relay connections keep the per-file path. - Unchanged games: when this device still holds the agreed base, the manifest request carries ifHash and a peer that still holds it too answers "unchanged" instead of sending the manifest. Older peers ignore the parameter; their full answer is used, not fetched twice. - Manifests are gzipped for clients that accept it (every Go client does, so older versions read them too): ~5.6x smaller with real hashes. - Manifests and batches get a 10-minute client timeout instead of 30s; a large manifest over a VPN never arrived inside 30s, so those games never synced. - An in-sync pass also mirrors the peer's latest snapshot, which only a pull did before; games identical everywhere from the start showed history on one device only. - A batch is bounded by what the responder counts (each requested block at the file's full block size, at most the file's real size), so a batch of many small files with a few larger ones is never refused as too large; a batch refused anyway is fetched again as two halves. Robustness - Presence: paired LAN peers are probed every 10s on their own ticker. Probes used to ride along with the minute-long reconcile, which waits for a whole sync-all pass; peers reached over a VPN (no UDP broadcast) showed online/offline minutes late. Probe rounds are serialized. - Games whose save folder is not on this device are skipped by sync-all and dropped from the retry queue instead of failing every 20s. - "syncing X with Y" is logged only when there is something to do; most passes produced only "already in sync" lines, which rotated everything else out of the log within hours. - The interrupted-pull safety test also runs over batched pulls. Co-Authored-By: Claude Opus 5.5 --- e2e/batch_bench_test.go | 59 ++++ internal/p2p/engine.go | 76 +++++ internal/p2p/manifest_gzip_test.go | 63 ++++ internal/p2p/missing_skip_test.go | 28 ++ internal/p2p/presence_test.go | 82 +++++ internal/p2p/routes.go | 139 +++++++- internal/p2p/syncengine/batch.go | 296 ++++++++++++++++++ internal/p2p/syncengine/batch_test.go | 237 ++++++++++++++ internal/p2p/syncengine/engine.go | 63 +++- internal/p2p/syncengine/engine_test.go | 53 ++++ .../p2p/syncengine/partial_safety_test.go | 14 +- internal/p2p/syncengine/quick.go | 73 +++++ internal/p2p/syncengine/quick_test.go | 117 +++++++ internal/p2p/syncengine/transport.go | 43 +++ internal/p2p/transport_lan.go | 65 +++- internal/p2p/transport_wan.go | 14 + 16 files changed, 1410 insertions(+), 12 deletions(-) create mode 100644 e2e/batch_bench_test.go create mode 100644 internal/p2p/manifest_gzip_test.go create mode 100644 internal/p2p/missing_skip_test.go create mode 100644 internal/p2p/presence_test.go create mode 100644 internal/p2p/syncengine/batch.go create mode 100644 internal/p2p/syncengine/batch_test.go create mode 100644 internal/p2p/syncengine/quick.go create mode 100644 internal/p2p/syncengine/quick_test.go diff --git a/e2e/batch_bench_test.go b/e2e/batch_bench_test.go new file mode 100644 index 00000000..0299c4e8 --- /dev/null +++ b/e2e/batch_bench_test.go @@ -0,0 +1,59 @@ +package e2e + +import ( + "fmt" + "net/http" + "os" + "path/filepath" + "strconv" + "strings" + "testing" + "time" + + "github.com/opensave/opensave/internal/p2p" + "github.com/opensave/opensave/internal/p2p/syncengine" + "github.com/opensave/opensave/testutil" +) + +// TestBench_ManySmallFiles times pulling a save of many small files between +// two real daemons, batched and file by file. Opt-in, because it is a +// measurement rather than a check: OPENSAVE_BENCH_FILES=5000 go test ./e2e -run TestBench -v +func TestBench_ManySmallFiles(t *testing.T) { + n, _ := strconv.Atoi(os.Getenv("OPENSAVE_BENCH_FILES")) + if n <= 0 { + t.Skip("set OPENSAVE_BENCH_FILES to run") + } + for _, mode := range []struct { + name string + proto int + }{{"batched", syncengine.ProtoBatchFiles}, {"file-by-file", syncengine.ProtoMultiRoot}} { + t.Run(mode.name, func(t *testing.T) { + restore := p2p.SetServedProto(mode.proto) + defer p2p.SetServedProto(restore) + + a := testutil.NewTestDaemon(t, "Bench-A") + b := testutil.NewTestDaemon(t, "Bench-B") + a.PairWith(b) + chunk := strings.Repeat("x", 3500) // a Project Zomboid map chunk is ~3.5KB + for i := 0; i < n; i++ { + a.WriteSave(fmt.Sprintf("map/map_%d_%d.bin", i/100, i%100), chunk+strconv.Itoa(i)) + } + restoreSyncOnTrack := suppressSyncOnTrack(a, b) + gameID := a.TrackGame("Bench") + b.API(http.MethodPost, "/api/games", map[string]string{"name": "Bench", "savePath": b.SaveDir}, nil) + restoreSyncOnTrack() + + start := time.Now() + b.API(http.MethodPost, "/api/games/"+gameID+"/sync", nil, nil) + count := func() int { + m, _ := filepath.Glob(filepath.Join(b.SaveDir, "map", "*.bin")) + return len(m) + } + if !testutil.WaitFor(60*time.Minute, func() bool { return count() >= n }) { + t.Fatalf("only %d of %d files arrived", count(), n) + } + el := time.Since(start) + t.Logf("%s: %d files in %s = %.2f ms/file", mode.name, n, el.Round(time.Millisecond), float64(el.Microseconds())/1000/float64(n)) + }) + } +} diff --git a/internal/p2p/engine.go b/internal/p2p/engine.go index b1777e2e..f99603bf 100644 --- a/internal/p2p/engine.go +++ b/internal/p2p/engine.go @@ -9,11 +9,15 @@ import ( "encoding/json" "errors" "fmt" + "io/fs" "net/http" + "os" + "path/filepath" "strings" "sync" "time" + "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/e2ee" "github.com/opensave/opensave/internal/p2p/discovery" "github.com/opensave/opensave/internal/p2p/pairing" @@ -119,6 +123,8 @@ type Engine struct { // that a device has gone away — see offlineStrikes. pingMu sync.Mutex pingMisses map[string]int + // probeMu serializes PingPairedPeers rounds. + probeMu sync.Mutex // Lineage refreshes already running after a peer-applied deletion, keyed // by game+peer. Deletions arrive one file at a time, and each one leaves @@ -382,6 +388,11 @@ func (e *Engine) clearPingMisses(peerID string) { // PingPairedPeers probes every paired LAN peer and updates their // online/offline status. func (e *Engine) PingPairedPeers(ctx context.Context) { + // One probe round at a time: the presence loop and a sync starting can + // both ask, and two overlapping rounds would count one outage as two + // misses toward offlineStrikes. + e.probeMu.Lock() + defer e.probeMu.Unlock() peers, err := e.Store.ListPeers() if err != nil { return @@ -542,6 +553,8 @@ func (e *Engine) StartResyncLoop() { stop := e.stopRetry e.pendingMu.Unlock() + e.startPresenceLoop(stop) + // Tracked on the engine's lifecycle so a shutdown lands between ticks // rather than half-way through a transfer. e.GoSync(func(ctx context.Context) { @@ -571,6 +584,40 @@ func (e *Engine) StartResyncLoop() { }) } +// presenceInterval is how often paired LAN peers are probed just to know +// whether they are there. +// +// Probing used to happen only as a side effect: before the 60-second +// reconcile, before a retry, before a manual sync. The reconcile runs on the +// same goroutine as the sync-everything pass it starts, so while a big +// library synced the next probe waited minutes, and with offlineStrikes +// that was minutes more before a device that had gone showed as gone. Peers +// seen by UDP broadcast hid this; one reached over a VPN such as Tailscale, +// which does not carry broadcasts, depends on probes alone. +// +// 10s makes "back" show within ~10s and "gone" within ~30s. A probe is one +// small GET per peer. A var only so a test can shorten it. +var presenceInterval = 10 * time.Second + +// startPresenceLoop probes paired peers on its own ticker, independent of +// whatever sync work is running. It stops with the resync loop. +func (e *Engine) startPresenceLoop(stop chan struct{}) { + e.GoSync(func(ctx context.Context) { + ticker := time.NewTicker(presenceInterval) + defer ticker.Stop() + for { + select { + case <-stop: + return + case <-ctx.Done(): + return + case <-ticker.C: + e.PingPairedPeers(ctx) + } + } + }) +} + // reconcileAllGames refreshes peer status and re-syncs every game — the // periodic backstop against missed sync triggers. func (e *Engine) reconcileAllGames(ctx context.Context) { @@ -602,6 +649,13 @@ func (e *Engine) retryPendingResyncs(ctx context.Context) { if ctx.Err() != nil { return // shutting down; the failsafe picks these up next start } + // A save folder that is not here cannot be synced, and retrying it + // every tick only logs the same failure. Dropped from the queue; the + // periodic reconcile picks it up once the folder is back. + if g, err := e.Store.GetGame(id); err == nil && saveFolderAbsent(g.SavePath) { + e.ClearPendingResync(id) + continue + } e.Log("info", fmt.Sprintf("retrying interrupted sync for %s", id)) results, err := e.Sync.SyncGame(ctx, id, online) if err == nil { @@ -616,6 +670,22 @@ func (e *Engine) retryPendingResyncs(ctx context.Context) { } } +// saveFolderAbsent reports whether a game's save folder is not on this device +// — the same test as daemon.SaveFolderMissing, which this package cannot +// import. For a single-file save it is the folder the file goes in that +// counts: the file itself may not have been written yet. +func saveFolderAbsent(savePath string) bool { + if savePath == "" { + return false + } + folder := savePath + if isFile, err := delta.ResolveLocalSaveFilePath(savePath); err == nil && isFile { + folder = filepath.Dir(savePath) + } + _, err := os.Stat(folder) + return errors.Is(err, fs.ErrNotExist) +} + // SyncAllGames syncs every tracked game (used when a peer comes online). func (e *Engine) SyncAllGames(ctx context.Context) { if e.Pause.Paused() { @@ -633,6 +703,12 @@ func (e *Engine) SyncAllGames(ctx context.Context) { if !g.AutoSync { continue } + // Not on this device (a drive it doesn't have, a folder that is + // gone): nothing to sync, and it is shown as such on screen. Trying + // anyway logged an error for it against every peer, every pass. + if saveFolderAbsent(g.SavePath) { + continue + } results, err := e.Sync.SyncGame(ctx, g.ID, online) if errors.Is(err, syncengine.ErrHeld) { continue // said once, when it was held; asked about on screen diff --git a/internal/p2p/manifest_gzip_test.go b/internal/p2p/manifest_gzip_test.go new file mode 100644 index 00000000..cca409cf --- /dev/null +++ b/internal/p2p/manifest_gzip_test.go @@ -0,0 +1,63 @@ +package p2p + +import ( + "crypto/sha256" + "encoding/hex" + "fmt" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/p2p/syncengine" +) + +// A large manifest goes over the wire gzipped, and an ordinary Go client — +// which is what every OpenSave version uses — reads it without knowing. +func TestManifestIsCompressedAndReadable(t *testing.T) { + m := delta.Manifest{Files: map[string]delta.FileEntry{}} + for i := 0; i < 5000; i++ { + sum := sha256.Sum256([]byte(fmt.Sprint(i))) // random-looking, like real hashes + h := hex.EncodeToString(sum[:]) + m.Files[fmt.Sprintf("Saves/world/map_%d_%d.bin", i/100, i%100)] = delta.FileEntry{ + Size: 3500, Hash: h, BlockSize: 65536, Blocks: []delta.Block{{Index: 0, Hash: h, Length: 3500}}, + } + } + resp := syncengine.ManifestResponse{Manifest: m, ActiveBranch: "main"} + + var wire atomic.Int64 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + cw := &countingResponse{ResponseWriter: w} + jsonOKCompressed(cw, r, resp) + wire.Store(cw.n) + })) + defer srv.Close() + + req, _ := http.NewRequest(http.MethodGet, srv.URL, nil) + var got syncengine.ManifestResponse + if err := doJSONWith(lanBulkClient, req, &got); err != nil { + t.Fatal(err) + } + if len(got.Manifest.Files) != len(m.Files) || got.Manifest.ManifestHash() != m.ManifestHash() { + t.Fatalf("manifest came back with %d files and a different hash", len(got.Manifest.Files)) + } + + plain := httptest.NewRecorder() + jsonOK(plain, resp) + if wire.Load()*2 > int64(plain.Body.Len()) { + t.Errorf("compressed %d bytes vs %d plain — expected at least 2x smaller", wire.Load(), plain.Body.Len()) + } + t.Logf("manifest of %d files: %d bytes plain, %d on the wire", len(m.Files), plain.Body.Len(), wire.Load()) +} + +type countingResponse struct { + http.ResponseWriter + n int64 +} + +func (c *countingResponse) Write(b []byte) (int, error) { + n, err := c.ResponseWriter.Write(b) + c.n += int64(n) + return n, err +} diff --git a/internal/p2p/missing_skip_test.go b/internal/p2p/missing_skip_test.go new file mode 100644 index 00000000..b5de9828 --- /dev/null +++ b/internal/p2p/missing_skip_test.go @@ -0,0 +1,28 @@ +package p2p + +import ( + "os" + "path/filepath" + "testing" +) + +func TestSaveFolderAbsent(t *testing.T) { + dir := t.TempDir() + if saveFolderAbsent(dir) { + t.Error("an existing folder was reported absent") + } + if !saveFolderAbsent(filepath.Join(dir, "gone")) { + t.Error("a missing folder was not reported absent") + } + // A single-file save not written yet: its folder is what counts. + file := filepath.Join(dir, "save.sav") + if err := os.WriteFile(file, []byte("x"), 0o666); err != nil { + t.Fatal(err) + } + if saveFolderAbsent(file) { + t.Error("an existing single-file save was reported absent") + } + if saveFolderAbsent("") { + t.Error("an empty path was reported absent") + } +} diff --git a/internal/p2p/presence_test.go b/internal/p2p/presence_test.go new file mode 100644 index 00000000..de0397de --- /dev/null +++ b/internal/p2p/presence_test.go @@ -0,0 +1,82 @@ +package p2p + +import ( + "net" + "net/http" + "net/http/httptest" + "path/filepath" + "strconv" + "sync/atomic" + "testing" + "time" + + "github.com/opensave/opensave/internal/snapshot" + "github.com/opensave/opensave/internal/store" +) + +func waitUntil(d time.Duration, cond func() bool) bool { + deadline := time.Now().Add(d) + for time.Now().Before(deadline) { + if cond() { + return true + } + time.Sleep(20 * time.Millisecond) + } + return cond() +} + +// A peer's online state follows it on its own ticker, whatever else the +// engine is doing — not only when a sync or the minute-long reconcile happens +// to probe. A peer reached over a VPN gets no UDP broadcasts, so without this +// its indicator lagged by minutes. +func TestPresenceLoop_FollowsPeer(t *testing.T) { + old := presenceInterval + presenceInterval = 50 * time.Millisecond + t.Cleanup(func() { presenceInterval = old }) + + var up atomic.Bool + up.Store(true) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if !up.Load() { + w.WriteHeader(http.StatusServiceUnavailable) + return + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"status":"ok","appVersion":"2.4.0"}`)) + })) + t.Cleanup(srv.Close) + host, portStr, _ := net.SplitHostPort(srv.Listener.Addr().String()) + port, _ := strconv.Atoi(portStr) + + s, err := store.Open(filepath.Join(t.TempDir(), "opensave.db")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { s.Close() }) + if err := s.EnsureDefaultSettings(t.TempDir(), t.TempDir()); err != nil { + t.Fatal(err) + } + if err := s.UpsertPeer(store.Peer{ID: "node_vpn", Name: "Over VPN", Address: host, Port: port, Status: "offline"}); err != nil { + t.Fatal(err) + } + + e := New(s, snapshot.New(s), func(string, string) {}) + stop := make(chan struct{}) + e.startPresenceLoop(stop) + t.Cleanup(func() { close(stop); e.cancel(); e.bgSyncWG.Wait() }) + + status := func() string { + p, err := s.GetPeer("node_vpn") + if err != nil { + return "" + } + return p.Status + } + if !waitUntil(3*time.Second, func() bool { return status() == "online" }) { + t.Fatalf("peer answering pings still %q", status()) + } + up.Store(false) + if !waitUntil(3*time.Second, func() bool { return status() == "offline" }) { + t.Fatalf("peer no longer answering still %q", status()) + } +} diff --git a/internal/p2p/routes.go b/internal/p2p/routes.go index 67d11a5c..2d0ac70b 100644 --- a/internal/p2p/routes.go +++ b/internal/p2p/routes.go @@ -1,6 +1,7 @@ package p2p import ( + "compress/gzip" "context" "encoding/json" "errors" @@ -11,6 +12,7 @@ import ( "os" "path/filepath" "strconv" + "strings" "sync/atomic" "time" @@ -46,6 +48,7 @@ func (e *Engine) RegisterRoutes(r chi.Router) { r.Use(e.refuseWhilePaused) r.Get("/api/p2p/manifest/{gameId}", e.handleManifest) r.Post("/api/p2p/blocks/{gameId}", e.handleBlocks) + r.Post("/api/p2p/files/{gameId}", e.handleFileBatch) r.Post("/api/p2p/delete-file/{gameId}", e.handleDeleteFile) r.Get("/api/sync/trigger/{gameId}", e.handleSyncTrigger) }) @@ -724,7 +727,19 @@ func (e *Engine) handleManifest(w http.ResponseWriter, r *http.Request) { if latest, err := e.Snapshots.LatestSnapshot(gameID, ""); err == nil { resp.LatestSnapshot = &syncengine.SnapshotInfo{ID: latest.ID, Timestamp: latest.Timestamp, Comment: latest.Comment} } - jsonOK(w, resp) + // The asker already holds the state both devices last agreed on and says + // so; if this device still holds it too, "unchanged" is the whole answer. + // For a save of a quarter-million files the full manifest is tens of MB + // of JSON, and the minute-long reconcile used to fetch it from every peer + // every minute to learn that nothing had changed. Single-location games + // only: extra locations carry their own bases (see RootHash). + if ifHash := r.URL.Query().Get("ifHash"); ifHash != "" && len(manifest.Extra) == 0 && + manifest.ManifestHash() == ifHash { + resp.Manifest = delta.Manifest{} + resp.Unchanged = true + resp.ManifestHash = ifHash + } + jsonOKCompressed(w, r, resp) } // holdForServing holds the game's save still for a manifest to be served from @@ -787,6 +802,100 @@ func (e *Engine) handleBlocks(w http.ResponseWriter, r *http.Request) { jsonOK(w, map[string]any{"blocks": encodeBlocks(blocks, wantsGzip(body.Encodings))}) } +// Bounds on one batch request, so a peer cannot make this device read and +// hold an unbounded amount in a single response. The requester stays well +// under them (syncengine.batchMaxFiles / batchMaxBytes). +const ( + fileBatchMaxFiles = 1024 + fileBatchMaxBytes = 16 << 20 +) + +// handleFileBatch serves the blocks of many files of one save location in a +// single response: handleBlocks for a list. One round trip per file was what +// made a save of many small files take hours; see syncengine.ProtoBatchFiles. +// +// A file that cannot be read is reported in its own entry rather than failing +// the batch, so the requester can say which one it was. +func (e *Engine) handleFileBatch(w http.ResponseWriter, r *http.Request) { + gameID := chi.URLParam(r, "gameId") + var body struct { + Root string `json:"root"` + Files []syncengine.FileBlocksRequest `json:"files"` + Encodings []string `json:"encodings"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil || len(body.Files) == 0 { + jsonError(w, http.StatusBadRequest, "files are required") + return + } + if len(body.Files) > fileBatchMaxFiles { + jsonError(w, http.StatusBadRequest, fmt.Sprintf("at most %d files per batch", fileBatchMaxFiles)) + return + } + game, err := e.trackedGameForPeer(gameID) + if err != nil { + jsonError(w, http.StatusNotFound, "Game not found.") + return + } + base, ok := e.resolveServeRoot(gameID, game, body.Root) + if !ok { + jsonError(w, http.StatusNotFound, "This device has no save location named "+strconv.Quote(body.Root)+" for that game.") + return + } + singleFile, _ := delta.ResolveLocalSaveFilePath(base) + + // What the answer will hold: each file's requested blocks at its block + // size, but never more than the file itself. Counting every block at full + // size took 256 files of 4 KB for 16 MB — the whole limit — and refused + // any batch of small files with a few larger ones among them, which failed + // whole pulls of saves made of many small files. + var requested int64 + for _, f := range body.Files { + bs := f.BlockSize + if bs <= 0 { + bs = 64 * 1024 + } + n := int64(len(f.BlockIndices)) * int64(bs) + if f.RelPath != "" && delta.IsSafePath(base, f.RelPath) { + p := delta.LocalNameFor(base, f.RelPath) + if singleFile { + p = base + } + if info, err := os.Stat(p); err == nil && info.Size() < n { + n = info.Size() + } + } + requested += n + } + if requested > fileBatchMaxBytes { + jsonError(w, http.StatusBadRequest, "batch too large") + return + } + gz := wantsGzip(body.Encodings) + + out := make([]syncengine.FileBlocks, 0, len(body.Files)) + for _, f := range body.Files { + entry := syncengine.FileBlocks{RelPath: f.RelPath} + if f.RelPath == "" || !delta.IsSafePath(base, f.RelPath) { + entry.Error = "invalid path" + out = append(out, entry) + continue + } + // Resolved, not joined — see handleBlocks. + fullPath := delta.LocalNameFor(base, f.RelPath) + if singleFile { + fullPath = base + } + blocks, err := delta.ReadBlocks(fullPath, f.BlockIndices, f.BlockSize) + if err != nil { + entry.Error = "read blocks failed: " + err.Error() + } else { + entry.Blocks = encodeBlocks(blocks, gz) + } + out = append(out, entry) + } + jsonOK(w, map[string]any{"files": out}) +} + func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") var body struct { @@ -979,6 +1088,32 @@ func progressEventFromMap(data map[string]any) syncengine.ProgressEvent { return ev } +// jsonOKCompressed is jsonOK, gzipped when the asker accepts it. +// +// For manifests. A save of a few hundred thousand files makes a manifest of +// tens of MB of JSON, which over a VPN did not arrive inside a peer's request +// timeout — so the peer never learned what this device holds, every sync of +// that game failed, and the two devices never converged. Manifest JSON (paths +// and hex hashes) compresses several times over. Go's HTTP client asks for +// gzip and unpacks it on its own unless told not to, so every version of +// OpenSave reads these, including ones that predate this. +func jsonOKCompressed(w http.ResponseWriter, r *http.Request, v any) { + if !strings.Contains(r.Header.Get("Accept-Encoding"), "gzip") { + jsonOK(w, v) + return + } + w.Header().Set("Content-Type", "application/json") + w.Header().Set("Content-Encoding", "gzip") + w.Header().Add("Vary", "Accept-Encoding") + w.WriteHeader(http.StatusOK) + zw, err := gzip.NewWriterLevel(w, gzip.BestSpeed) + if err != nil { + return + } + _ = json.NewEncoder(zw).Encode(v) + _ = zw.Close() +} + func jsonOK(w http.ResponseWriter, v any) { w.Header().Set("Content-Type", "application/json") w.WriteHeader(http.StatusOK) @@ -1019,7 +1154,7 @@ func (e *Engine) resolveServeRoot(gameID string, game store.Game, root string) ( // argument — the read still happens concurrently. var servedProto atomic.Int64 -func init() { servedProto.Store(syncengine.ProtoMultiRoot) } +func init() { servedProto.Store(syncengine.ProtoBatchFiles) } // ServedProto reports the protocol revision advertised to peers. func ServedProto() int { return int(servedProto.Load()) } diff --git a/internal/p2p/syncengine/batch.go b/internal/p2p/syncengine/batch.go new file mode 100644 index 00000000..7a15d94e --- /dev/null +++ b/internal/p2p/syncengine/batch.go @@ -0,0 +1,296 @@ +package syncengine + +import ( + "context" + "errors" + "fmt" + "os" + "strings" + "sync" + "time" + + "github.com/opensave/opensave/internal/delta" +) + +// Pulling many small files, many to a request. +// +// A pull used to cost one HTTP round trip per file, one after the other. +// Measured on a Windows LAN, that came to ~200ms a file, of which writing the +// file (temp, fsync, rename) was ~2ms — so a save made of a quarter-million +// small files took half a day to move 50MB. Batching removes the round trips; +// the writing is the same as before, file by file, and just as safe: each +// file still goes through a PatchWriter (temp file, content verified against +// the manifest's hash, fsync, rename) and gets the peer's mtime. + +const ( + // batchFileMaxBytes is the most a file may need fetched to go into a + // batch. Anything bigger keeps the per-file path, which already fetches + // its blocks in parallel and has nothing to gain here. + batchFileMaxBytes = 256 << 10 + // batchMaxFiles and batchMaxBytes bound one request, well inside what the + // responder accepts (p2p.fileBatchMaxFiles / fileBatchMaxBytes). + batchMaxFiles = 256 + batchMaxBytes = 4 << 20 + // batchMaxClaimed bounds a request as the responder measures it: every + // block requested at the file's full block size, whatever the file's real + // size. 256 files of 4 KB claim 16 MB that way, which is the responder's + // whole limit (p2p.fileBatchMaxBytes), so any file of two blocks in such + // a batch had it refused as "batch too large" — and the whole pull with + // it, over and over, on saves of many small files. Kept under the limit + // with room to spare. + batchMaxClaimed = 12 << 20 +) + +// claimedBytes is what a file costs a batch as the responder counts it. +func claimedBytes(f delta.FileEntry, indices []int) int64 { + bs := f.BlockSize + if bs <= 0 { + bs = 64 * 1024 + } + return int64(len(indices)) * int64(bs) +} + +// batchWorkers is how many batch requests are in flight at once: enough to +// keep the link and the responder's disk busy while this side writes. A var +// only so a test can make the order deterministic. +var batchWorkers = 4 + +type batchJob struct { + relPath string + localPath string + remote delta.FileEntry + indices []int +} + +// changedBytes is how much of a file a pull has to fetch. +func changedBytes(f delta.FileEntry, indices []int) int64 { + var n int64 + for _, idx := range indices { + if idx >= 0 && idx < len(f.Blocks) { + n += int64(f.Blocks[idx].Length) + } + } + return n +} + +// groupBatches splits jobs into requests of at most batchMaxFiles files, +// batchMaxBytes of blocks, and batchMaxClaimed as the responder counts them. +func groupBatches(jobs []batchJob) [][]batchJob { + var out [][]batchJob + var cur []batchJob + var curBytes, curClaimed int64 + for _, j := range jobs { + b := changedBytes(j.remote, j.indices) + c := claimedBytes(j.remote, j.indices) + if len(cur) > 0 && (len(cur) >= batchMaxFiles || curBytes+b > batchMaxBytes || curClaimed+c > batchMaxClaimed) { + out = append(out, cur) + cur, curBytes, curClaimed = nil, 0, 0 + } + cur = append(cur, j) + curBytes += b + curClaimed += c + } + if len(cur) > 0 { + out = append(out, cur) + } + return out +} + +// pullBatched fetches and writes jobs, many files per request. It returns the +// paths it wrote, including on error, since those files are on disk. +func (e *Engine) pullBatched(ctx context.Context, batcher BatchFetcher, peer Peer, gameID, root string, + jobs []batchJob, throttle *throttler, tracker *progressTracker, onProgress func(force bool)) ([]string, error) { + + ctx, cancel := context.WithCancel(ctx) + defer cancel() + + var ( + mu sync.Mutex + written []string + firstErr error + ) + fail := func(err error) { + mu.Lock() + if firstErr == nil { + firstErr = err + cancel() + } + mu.Unlock() + } + + batches := groupBatches(jobs) + workers := batchWorkers + if workers > len(batches) { + workers = len(batches) + } + work := make(chan []batchJob) + var wg sync.WaitGroup + for i := 0; i < workers; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for batch := range work { + done, err := e.pullOneBatch(ctx, batcher, peer, gameID, root, batch, throttle, tracker, onProgress) + mu.Lock() + written = append(written, done...) + mu.Unlock() + if err != nil { + fail(err) + return + } + // One line per batch instead of one per file: a quarter-million + // "file updated" lines rotated every other message out of the log. + if len(done) > 0 { + e.Log("info", fmt.Sprintf("%d files updated (%s … %s)", len(done), done[0], done[len(done)-1])) + } + onProgress(true) + } + }() + } + for _, b := range batches { + select { + case work <- b: + case <-ctx.Done(): + } + } + close(work) + wg.Wait() + + mu.Lock() + defer mu.Unlock() + if firstErr != nil { + return written, firstErr + } + return written, ctx.Err() +} + +// pullOneBatch fetches one batch and writes its files. +func (e *Engine) pullOneBatch(ctx context.Context, batcher BatchFetcher, peer Peer, gameID, root string, + batch []batchJob, throttle *throttler, tracker *progressTracker, onProgress func(force bool)) ([]string, error) { + + req := make([]FileBlocksRequest, len(batch)) + for i, j := range batch { + req[i] = FileBlocksRequest{RelPath: j.relPath, BlockIndices: j.indices, BlockSize: j.remote.BlockSize} + } + + var resp []FileBlocks + var err error + const maxAttempts = 3 + for attempt := 1; attempt <= maxAttempts; attempt++ { + resp, err = batcher.FetchFileBatch(ctx, peer, gameID, root, req) + if err == nil || errors.Is(err, ErrBatchUnsupported) { + break + } + // Refused as too large — a responder that counts or limits batches + // differently: the same request will be refused again, so it goes + // as two halves instead. Asking again unchanged failed the whole pull. + if strings.Contains(err.Error(), "batch too large") && len(batch) > 1 { + half := len(batch) / 2 + first, err := e.pullOneBatch(ctx, batcher, peer, gameID, root, batch[:half], throttle, tracker, onProgress) + if err != nil { + return first, err + } + second, err := e.pullOneBatch(ctx, batcher, peer, gameID, root, batch[half:], throttle, tracker, onProgress) + return append(first, second...), err + } + e.Log("warn", fmt.Sprintf("batch fetch attempt %d/%d of %d files failed: %v", attempt, maxAttempts, len(batch), err)) + if attempt < maxAttempts { + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-time.After(time.Duration(attempt) * time.Second): + } + } + } + if errors.Is(err, ErrBatchUnsupported) { + return e.pullBatchOneByOne(ctx, peer, gameID, root, batch, throttle, tracker, onProgress) + } + if err != nil { + return nil, fmt.Errorf("fetch batch of %d files: %w", len(batch), err) + } + + byPath := make(map[string]FileBlocks, len(resp)) + for _, f := range resp { + byPath[f.RelPath] = f + } + + var written []string + for _, j := range batch { + if ctx.Err() != nil { + return written, ctx.Err() + } + f, ok := byPath[j.relPath] + if !ok { + return written, fmt.Errorf("fetch blocks for %s: %s did not send it", j.relPath, peer.Name) + } + if f.Error != "" { + return written, fmt.Errorf("fetch blocks for %s: %s", j.relPath, f.Error) + } + if err := writeBatchedFile(j, f.Blocks); err != nil { + return written, err + } + written = append(written, j.relPath) + n := changedBytes(j.remote, j.indices) + tracker.add(n) + onProgress(false) + throttle.wait(ctx, n) + } + return written, nil +} + +// writeBatchedFile writes one file from its fetched blocks, exactly as +// pullFile does: unchanged blocks from the copy on disk, the rest from the +// peer, verified and renamed into place by the PatchWriter. +func writeBatchedFile(j batchJob, blocks []BlockData) error { + writer, err := delta.NewPatchWriter(j.localPath, j.remote) + if err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + committed := false + defer func() { + if !committed { + writer.Abort() + } + }() + incoming := make(map[int]bool, len(j.indices)) + for _, idx := range j.indices { + incoming[idx] = true + } + if err := writer.SeedUnchanged(j.localPath, incoming); err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + for _, b := range blocks { + if err := writer.WriteBlock(b.Index, b.Data); err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + } + if err := writer.Commit(); err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + committed = true + if j.remote.MtimeMs > 0 { + mtime := time.UnixMilli(int64(j.remote.MtimeMs)) + _ = os.Chtimes(j.localPath, mtime, mtime) + } + return nil +} + +// pullBatchOneByOne is the fallback when the transport cannot batch for this +// peer after all: the same files, through the per-file path. +func (e *Engine) pullBatchOneByOne(ctx context.Context, peer Peer, gameID, root string, + batch []batchJob, throttle *throttler, tracker *progressTracker, onProgress func(force bool)) ([]string, error) { + + var written []string + for _, j := range batch { + if err := e.pullFile(ctx, peer, FileRef{GameID: gameID, Root: root, RelPath: j.relPath}, j.localPath, + j.remote, j.indices, throttle, tracker, onProgress); err != nil { + return written, err + } + if j.remote.MtimeMs > 0 { + mtime := time.UnixMilli(int64(j.remote.MtimeMs)) + _ = os.Chtimes(j.localPath, mtime, mtime) + } + written = append(written, j.relPath) + } + return written, nil +} diff --git a/internal/p2p/syncengine/batch_test.go b/internal/p2p/syncengine/batch_test.go new file mode 100644 index 00000000..c468cf6c --- /dev/null +++ b/internal/p2p/syncengine/batch_test.go @@ -0,0 +1,237 @@ +package syncengine + +import ( + "bytes" + "context" + "fmt" + "os" + "path/filepath" + "strings" + "sync" + "testing" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/snapshot" +) + +// batchTransport is fakeTransport plus batched fetches, answering manifests +// with a chosen protocol revision. +type batchTransport struct { + *fakeTransport + proto int + + mu sync.Mutex + batchCalls int + blockCalls int + failPath string // answered with a per-file error + // maxFiles, when set, refuses bigger batches the way a responder that + // counts differently does. + maxFiles int +} + +func (b *batchTransport) FetchManifest(ctx context.Context, peer Peer, gameID string, q ManifestQuery) (ManifestResponse, error) { + resp, err := b.fakeTransport.FetchManifest(ctx, peer, gameID, q) + resp.Proto = b.proto + return resp, err +} + +func (b *batchTransport) FetchBlocks(ctx context.Context, peer Peer, ref FileRef, idx []int, bs int) ([]BlockData, error) { + b.mu.Lock() + b.blockCalls++ + b.mu.Unlock() + return b.fakeTransport.FetchBlocks(ctx, peer, ref, idx, bs) +} + +func (b *batchTransport) FetchFileBatch(ctx context.Context, peer Peer, gameID, root string, files []FileBlocksRequest) ([]FileBlocks, error) { + b.mu.Lock() + b.batchCalls++ + b.mu.Unlock() + if len(files) > batchMaxFiles { + return nil, fmt.Errorf("batch of %d files is over the limit", len(files)) + } + if b.maxFiles > 0 && len(files) > b.maxFiles { + return nil, fmt.Errorf(`peer returned 400: {"error":"batch too large"}`) + } + out := make([]FileBlocks, 0, len(files)) + for _, f := range files { + if f.RelPath == b.failPath { + out = append(out, FileBlocks{RelPath: f.RelPath, Error: "read blocks failed: locked"}) + continue + } + blocks, err := b.fakeTransport.FetchBlocks(ctx, peer, FileRef{GameID: gameID, Root: root, RelPath: f.RelPath}, f.BlockIndices, f.BlockSize) + if err != nil { + out = append(out, FileBlocks{RelPath: f.RelPath, Error: err.Error()}) + continue + } + out = append(out, FileBlocks{RelPath: f.RelPath, Blocks: blocks}) + } + return out, nil +} + +func setupBatchEngine(t *testing.T, proto int) (*engineEnv, *batchTransport) { + t.Helper() + env := setupEngine(t) + bt := &batchTransport{fakeTransport: env.transport, proto: proto} + env.engine = New(env.store, snapshot.New(env.store), bt) + return env, bt +} + +// A save of many small files, plus one big one, in two folders. +func writeManyFiles(t *testing.T, dir string, n int) { + t.Helper() + for i := 0; i < n; i++ { + write(t, dir, fmt.Sprintf("map/chunk_%04d.bin", i), strings.Repeat(fmt.Sprintf("chunk %d;", i), 20)) + } + big := bytes.Repeat([]byte("large save data "), (512<<10)/16) // 512KB: per-file path + if err := os.WriteFile(filepath.Join(dir, "big.sav"), big, 0o666); err != nil { + t.Fatal(err) + } +} + +func assertSameTree(t *testing.T, want, got string) { + t.Helper() + wm, err := delta.BuildManifest(want) + if err != nil { + t.Fatal(err) + } + gm, err := delta.BuildManifest(got) + if err != nil { + t.Fatal(err) + } + if wm.ManifestHash() != gm.ManifestHash() { + t.Fatalf("trees differ: %d files remote, %d local", len(wm.Files), len(gm.Files)) + } + for p, f := range wm.Files { + if gm.Files[p].MtimeMs != f.MtimeMs { + t.Errorf("%s: mtime %d, want the peer's %d", p, gm.Files[p].MtimeMs, f.MtimeMs) + break + } + } +} + +func TestPull_BatchesSmallFiles(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoBatchFiles) + writeManyFiles(t, env.remoteDir, 600) + + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + if res.Status != "updated" || res.Direction != "pull" { + t.Fatalf("result = %+v, want updated/pull", res) + } + assertSameTree(t, env.remoteDir, env.localDir) + + // 600 one-block files: each claims a 64 KB block as the responder counts, + // so 192 fit a request (batchMaxClaimed) -> 4 requests; the big file alone + // goes block by block. + if bt.batchCalls != 4 { + t.Errorf("batch requests = %d, want 4", bt.batchCalls) + } + if bt.blockCalls != 1 { + t.Errorf("per-file block requests = %d, want 1 (the big file)", bt.blockCalls) + } + if len(env.transport.syncEvents) == 0 || env.transport.syncEvents[len(env.transport.syncEvents)-1] != "sync-complete" { + t.Errorf("sync did not complete: %v", env.transport.syncEvents) + } +} + +// A peer that predates batching is asked file by file, as before. +func TestPull_OldPeerIsNotBatched(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoMultiRoot) + writeManyFiles(t, env.remoteDir, 50) + + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + assertSameTree(t, env.remoteDir, env.localDir) + if bt.batchCalls != 0 { + t.Errorf("batch requests to an old peer = %d, want 0", bt.batchCalls) + } + if bt.blockCalls != 51 { + t.Errorf("per-file requests = %d, want 51", bt.blockCalls) + } +} + +// One file the peer cannot read fails the pull and names the file, like the +// per-file path does; nothing half-written is left in its place. +func TestPull_BatchFileErrorNamesTheFile(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoBatchFiles) + writeManyFiles(t, env.remoteDir, 10) + bt.failPath = "map/chunk_0003.bin" + + _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err == nil || !strings.Contains(err.Error(), "map/chunk_0003.bin") { + t.Fatalf("err = %v, want one naming map/chunk_0003.bin", err) + } + if _, statErr := os.Stat(filepath.Join(env.localDir, "map", "chunk_0003.bin")); statErr == nil { + t.Error("the file the peer could not serve was written anyway") + } + leftovers, _ := filepath.Glob(filepath.Join(env.localDir, "map", "*"+delta.TmpSuffix)) + if len(leftovers) > 0 { + t.Errorf("temp files left behind: %v", leftovers) + } +} + +func TestGroupBatches_RespectsLimits(t *testing.T) { + var jobs []batchJob + for i := 0; i < 600; i++ { + jobs = append(jobs, batchJob{relPath: fmt.Sprint(i), + remote: delta.FileEntry{Blocks: []delta.Block{{Index: 0, Length: 32 << 10}}}, + indices: []int{0}}) + } + for _, b := range groupBatches(jobs) { + var n int64 + for _, j := range b { + n += changedBytes(j.remote, j.indices) + } + if len(b) > batchMaxFiles || n > batchMaxBytes { + t.Errorf("batch of %d files / %d bytes is over the limit", len(b), n) + } + } +} + +// Many small files, some of several blocks: what the responder counts — +// every block at full block size — must stay under its limit too. 256 files +// of one 64 KB block already claim 16 MB, and each file of more blocks tipped +// a batch over it: "batch too large", and the whole pull failed. +func TestGroupBatches_StaysUnderWhatTheResponderCounts(t *testing.T) { + var jobs []batchJob + for i := 0; i < 2000; i++ { + blocks := []delta.Block{{Index: 0, Length: 4 << 10}} + indices := []int{0} + if i%10 == 0 { + blocks = []delta.Block{{Index: 0, Length: 64 << 10}, {Index: 1, Length: 64 << 10}, {Index: 2, Length: 10 << 10}} + indices = []int{0, 1, 2} + } + jobs = append(jobs, batchJob{relPath: fmt.Sprint(i), + remote: delta.FileEntry{BlockSize: 64 << 10, Blocks: blocks}, indices: indices}) + } + for _, b := range groupBatches(jobs) { + var claimed int64 + for _, j := range b { + claimed += claimedBytes(j.remote, j.indices) + } + if claimed > 16<<20 { + t.Errorf("a batch of %d files claims %d bytes, over the responder's limit", len(b), claimed) + } + } +} + +// A responder that refuses a batch as too large gets it again in halves, +// rather than the pull failing. +func TestPull_ABatchRefusedAsTooLargeIsSplit(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoBatchFiles) + bt.maxFiles = 10 + for i := 0; i < 40; i++ { + write(t, env.remoteDir, fmt.Sprintf("chunks/c%02d.bin", i), fmt.Sprint("chunk ", i)) + } + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("sync failed though the batches could be split: %v", err) + } + for i := 0; i < 40; i++ { + if _, err := os.Stat(filepath.Join(env.localDir, "chunks", fmt.Sprintf("c%02d.bin", i))); err != nil { + t.Fatalf("c%02d.bin did not arrive", i) + } + } +} diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index f60d67fe..ac58f604 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -404,15 +404,34 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re if err != nil { return Result{}, err } - e.Log("info", fmt.Sprintf("syncing %q with %q (%s)", game.Name, peer.Name, - map[bool]string{true: "WAN relay", false: "direct LAN"}[peer.Wan()])) - - // 1. Fetch remote manifest + branch info. isFile, _ := delta.ResolveLocalSaveFilePath(game.SavePath) - remoteData, err := e.Transport.FetchManifest(ctx, peer, gameID, ManifestQuery{ + query := ManifestQuery{ Name: game.Name, SavePath: game.SavePath, IsFile: isFile, AppID: game.AppID, CoverURL: game.CoverURL, - }) + } + + // 0. Nothing changed on either side since they last agreed: say so and + // stop, without moving either manifest. Quietly, too — this is the + // answer for nearly every game on every periodic pass. + res, done, prefetched := e.quickInSync(ctx, gameID, game, peer, query) + if done { + return res, nil + } + + // Said once there is something to do (after the in-sync check below), + // not here: a line per game per peer per periodic pass, most of them + // followed by "already in sync", rotated everything else out of the log + // within hours. + syncingLine := fmt.Sprintf("syncing %q with %q (%s)", game.Name, peer.Name, + map[bool]string{true: "WAN relay", false: "direct LAN"}[peer.Wan()]) + + // 1. Fetch remote manifest + branch info. + var remoteData ManifestResponse + if prefetched != nil { + remoteData = *prefetched + } else { + remoteData, err = e.Transport.FetchManifest(ctx, peer, gameID, query) + } if err != nil { // The peer simply isn't tracking this game (they untracked it, or // never had it). That's a stable state, not a transient network @@ -651,7 +670,6 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re decision := ComputeWithDeletions(localManifest, remoteData.Manifest, lineageFiles, lineageDirs, agreedHash, deleted) if !decision.HasChanges() { - e.Log("success", fmt.Sprintf("%q already in sync with %q", game.Name, peer.Name)) e.persistLineage(gameID, peer.ID, localManifest, remoteData.Manifest) // Both sides verifiably identical: this is a convergence point. _ = e.Store.SetAgreedHash(gameID, peer.ID, localManifest.ManifestHash()) @@ -670,8 +688,23 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re // are read through the gate — so the branch hold goes first. releaseAligned() e.syncExtraRoots(ctx, gameID, game, peer, remoteData) + // Share the peer's latest history entry here too. A pull does this; + // a sync that found both sides already identical used to skip it, + // so a game whose saves were identical everywhere from the start + // (Steam Cloud, a copied folder) showed a snapshot on the device + // that tracked it and none on any other. The local files ARE the + // peer's at this point, so the mirror describes the same content. + // Once per peer snapshot: recordMirrorSnapshot skips an id it has. + // + // Last, after the other locations have synced: it zips every location, + // and taking that time before them delayed their sync for nothing. + if remoteData.LatestSnapshot != nil && len(localManifest.Files) > 0 { + e.recordMirrorSnapshot(gameID, game, peer, *remoteData.LatestSnapshot, + fmt.Sprintf("Synced from peer: %s (%s)", peer.Name, remoteData.LatestSnapshot.Comment)) + } return Result{Status: "in_sync", Direction: "none"}, nil } + e.Log("info", syncingLine) if emptiedUnconfirmed(remoteData.Manifest.Files, decision, remoteData.DeletionConfirmed) { e.Log("info", fmt.Sprintf("%q holds none of %q's save files now, and has not confirmed deleting them — keeping this device's copies", @@ -1589,6 +1622,11 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s // can record the lineage from confirmed fact rather than rediscovering it // with a manifest round trip. See the sync-complete event below. var pulled []string + // Small files are collected here and fetched many to a request once the + // per-file checks below have passed; see pullBatched. + batcher, canBatch := e.Transport.(BatchFetcher) + canBatch = canBatch && !peer.Wan() && remoteData.Proto >= ProtoBatchFiles + var batched []batchJob for _, relPath := range filesToPull { if !delta.IsSafePath(root.Path, relPath) { return fmt.Errorf("path traversal attempt on pulled file %s", relPath) @@ -1636,6 +1674,10 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s if isFile, _ := delta.ResolveLocalSaveFilePath(root.Path); isFile { localFilePath = root.Path // single-file save mode } + if canBatch && len(indices) > 0 && changedBytes(remoteFile, indices) <= batchFileMaxBytes { + batched = append(batched, batchJob{relPath: relPath, localPath: localFilePath, remote: remoteFile, indices: indices}) + continue + } if err := e.pullFile(ctx, peer, FileRef{GameID: gameID, Root: root.Name, RelPath: relPath}, localFilePath, remoteFile, indices, throttle, tracker, reportProgress); err != nil { return err @@ -1650,6 +1692,13 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s // File-boundary progress reporting (always fires). reportProgress(true) } + if len(batched) > 0 { + done, err := e.pullBatched(ctx, batcher, peer, gameID, root.Name, batched, throttle, tracker, reportProgress) + pulled = append(pulled, done...) + if err != nil { + return err + } + } // Said once, plainly, at the end. A sync that quietly leaves files behind // is worse than one that fails: the save looks synced and is not, and the diff --git a/internal/p2p/syncengine/engine_test.go b/internal/p2p/syncengine/engine_test.go index 1ed654ff..22a2be33 100644 --- a/internal/p2p/syncengine/engine_test.go +++ b/internal/p2p/syncengine/engine_test.go @@ -224,6 +224,59 @@ func TestSync_PullNewRemoteFile(t *testing.T) { } } +// Saves identical on both sides from the start (Steam Cloud, a copied folder) +// never pull anything — and used to never share the peer's history entry +// either, so the game showed a snapshot on one device and none on the other. +func TestSync_InSyncMirrorsPeerSnapshot(t *testing.T) { + env := setupEngine(t) + write(t, env.localDir, "save.dat", "same progress") + write(t, env.remoteDir, "save.dat", "same progress") + env.transport.latestSnap = &SnapshotInfo{ID: "snap_888", Timestamp: "2026-06-01T00:00:00.000Z", Comment: "Initial snapshot"} + + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatalf("SyncWithPeer error = %v", err) + } + if res.Status != "in_sync" { + t.Fatalf("result = %+v, want in_sync", res) + } + snap, err := env.store.GetSnapshot("snap_888") + if err != nil { + t.Fatalf("peer snapshot not mirrored on an in-sync pass: %v", err) + } + if !strings.Contains(snap.Comment, "Synced from peer: Remote Device") { + t.Errorf("mirror comment = %q", snap.Comment) + } + + // A second in-sync pass must not take it again. + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + game, err := env.store.GetGame("game1") + if err != nil { + t.Fatal(err) + } + snaps, err := env.store.ListSnapshots("game1", game.ActiveBranch) + if err != nil { + t.Fatal(err) + } + if len(snaps) != 1 { + t.Errorf("snapshots after two in-sync passes = %d, want 1", len(snaps)) + } +} + +// An empty save folder in sync with an empty peer has nothing to mirror. +func TestSync_InSyncEmptyFolderTakesNoMirror(t *testing.T) { + env := setupEngine(t) + env.transport.latestSnap = &SnapshotInfo{ID: "snap_999", Timestamp: "2026-06-01T00:00:00.000Z", Comment: "Initial snapshot"} + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + if _, err := env.store.GetSnapshot("snap_999"); err == nil { + t.Error("an empty folder was mirrored") + } +} + func TestSync_InsufficientDiskSpace(t *testing.T) { env := setupEngine(t) write(t, env.remoteDir, "save.dat", "remote progress that needs space") diff --git a/internal/p2p/syncengine/partial_safety_test.go b/internal/p2p/syncengine/partial_safety_test.go index fa85845d..59588c39 100644 --- a/internal/p2p/syncengine/partial_safety_test.go +++ b/internal/p2p/syncengine/partial_safety_test.go @@ -15,7 +15,7 @@ import ( // its partial copy back, the missing files were read as "deleted over there", // and 23k files were deleted on the device that still had them. 2.4 records // as shared only what both manifests showed (persistLineage), which should -// rule that out; these hold it there. +// rule that out; these hold it there, for the per-file and the batched pull. func countFiles(t *testing.T, dir string) int { t.Helper() @@ -79,6 +79,18 @@ func TestPartialPull_InterruptedThenResumed(t *testing.T) { interruptedThenResumed(t, env, 200, 60) } +func TestPartialPull_InterruptedThenResumed_Batched(t *testing.T) { + // One batch in flight, so the first batch of 256 is written before the + // interruption lands in the second: a deterministic partial copy. + old := batchWorkers + batchWorkers = 1 + t.Cleanup(func() { batchWorkers = old }) + env, _ := setupBatchEngine(t, ProtoBatchFiles) + // Batches fetch through the same fake reader, one call per file; the + // 300th is inside the second batch. + interruptedThenResumed(t, env, 600, 300) +} + // The other side of the same situation: this device has everything, the peer // holds only part of it and never synced. The part it lacks is new to it, not // deleted by it. diff --git a/internal/p2p/syncengine/quick.go b/internal/p2p/syncengine/quick.go new file mode 100644 index 00000000..435eba66 --- /dev/null +++ b/internal/p2p/syncengine/quick.go @@ -0,0 +1,73 @@ +package syncengine + +import ( + "context" + "fmt" + + "github.com/opensave/opensave/internal/store" +) + +// quickInSync settles the common case of a sync in one small exchange: this +// device's save still hashes to the base both devices last agreed on, and +// the peer, asked whether it does too (ManifestQuery.IfHash), says +// "unchanged". Then there is nothing to decide, and neither manifest needs +// to cross the wire — for a save of a quarter-million files that is tens of +// MB of JSON per peer, which the minute-long reconcile used to fetch every +// minute to learn that nothing had changed. +// +// It only ever answers "in sync" when both sides provably hold the agreed +// state. In every other case it reports not done and the full sync runs; a +// full manifest that came back anyway (a peer that predates the question) +// is handed over in prefetched so it is not fetched twice. +// +// Deliberately narrow: +// - no exclusion rules: the agreed base is taken over the filtered +// manifest, the peer hashes its unfiltered one; +// - no extra save locations: those carry bases of their own (RootHash); +// - no conflict waiting, and the same branch on both sides. +func (e *Engine) quickInSync(ctx context.Context, gameID string, game store.Game, peer Peer, q ManifestQuery) (res Result, done bool, prefetched *ManifestResponse) { + agreed := e.Store.GetAgreedHash(gameID, peer.ID) + if agreed == "" || !e.rulesFor(gameID).Empty() { + return Result{}, false, nil + } + if roots, err := e.Store.GameRootPaths(gameID); err != nil || len(roots) > 0 { + return Result{}, false, nil + } + e.mu.Lock() + conflicted := e.activeConflicts[gameID] != nil + e.mu.Unlock() + if conflicted { + return Result{}, false, nil + } + local, err := e.ReadManifest(ctx, gameID, game.SavePath) + if err != nil || len(local.Extra) > 0 || local.ManifestHash() != agreed { + return Result{}, false, nil + } + + q.IfHash = agreed + resp, err := e.Transport.FetchManifest(ctx, peer, gameID, q) + if err != nil { + // Let the full path fetch again and classify the error as it always has. + return Result{}, false, nil + } + if !resp.Unchanged { + return Result{}, false, &resp + } + if resp.ManifestHash != agreed || (resp.ActiveBranch != "" && resp.ActiveBranch != game.ActiveBranch) { + // Unchanged, but not in a way this shortcut may act on: the full + // exchange decides. + return Result{}, false, nil + } + + // Same as the in-sync outcome of a full sync, minus what is already + // recorded: the base is unchanged and the lineage with it. + e.Transport.ReportSyncEvent(peer, gameID, "in-sync", map[string]any{ + "peerName": e.deviceName(), + "manifestHash": agreed, + }) + if resp.LatestSnapshot != nil && len(local.Files) > 0 { + e.recordMirrorSnapshot(gameID, game, peer, *resp.LatestSnapshot, + fmt.Sprintf("Synced from peer: %s (%s)", peer.Name, resp.LatestSnapshot.Comment)) + } + return Result{Status: "in_sync", Direction: "none"}, true, nil +} diff --git a/internal/p2p/syncengine/quick_test.go b/internal/p2p/syncengine/quick_test.go new file mode 100644 index 00000000..3cfdf8d8 --- /dev/null +++ b/internal/p2p/syncengine/quick_test.go @@ -0,0 +1,117 @@ +package syncengine + +import ( + "context" + "testing" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/snapshot" +) + +// ifHashTransport answers IfHash the way a current peer does, and counts +// how many full manifests it had to send. +type ifHashTransport struct { + *fakeTransport + full, unchanged int +} + +func (t *ifHashTransport) FetchManifest(ctx context.Context, peer Peer, gameID string, q ManifestQuery) (ManifestResponse, error) { + resp, err := t.fakeTransport.FetchManifest(ctx, peer, gameID, q) + if err != nil { + return resp, err + } + if q.IfHash != "" && resp.Manifest.ManifestHash() == q.IfHash { + t.unchanged++ + return ManifestResponse{ActiveBranch: resp.ActiveBranch, LatestSnapshot: resp.LatestSnapshot, + Unchanged: true, ManifestHash: q.IfHash, Manifest: delta.Manifest{}}, nil + } + t.full++ + return resp, nil +} + +func setupIfHashEngine(t *testing.T) (*engineEnv, *ifHashTransport) { + t.Helper() + env := setupEngine(t) + tr := &ifHashTransport{fakeTransport: env.transport} + env.engine = New(env.store, snapshot.New(env.store), tr) + return env, tr +} + +func TestQuickInSync_SkipsTheManifestWhenNothingChanged(t *testing.T) { + env, tr := setupIfHashEngine(t) + write(t, env.localDir, "save.dat", "same") + write(t, env.remoteDir, "save.dat", "same") + + // First sync converges and records the agreed base. + if res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil || res.Status != "in_sync" { + t.Fatalf("first sync = %+v, %v", res, err) + } + if tr.full != 1 { + t.Fatalf("setup: full manifests = %d, want 1", tr.full) + } + + // Nothing changed anywhere: settled without a full manifest. + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil || res.Status != "in_sync" { + t.Fatalf("second sync = %+v, %v", res, err) + } + if tr.full != 1 || tr.unchanged != 1 { + t.Errorf("second sync fetched %d full manifests and %d unchanged answers; want 0 more full, 1 unchanged", tr.full-1, tr.unchanged) + } +} + +// A change on either side must go the full way — the shortcut may only ever +// say "in sync" when both sides provably hold the agreed state. +func TestQuickInSync_ChangesStillSync(t *testing.T) { + env, tr := setupIfHashEngine(t) + write(t, env.localDir, "save.dat", "same") + write(t, env.remoteDir, "save.dat", "same") + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + + // The peer changes. + write(t, env.remoteDir, "new.dat", "made on the other device") + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatal(err) + } + if res.Status != "updated" || res.Direction != "pull" { + t.Fatalf("peer change: result = %+v, want updated/pull", res) + } + + // This side changes. + write(t, env.localDir, "mine.dat", "made here") + before := env.transport.pullTriggers + res, err = env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatal(err) + } + if res.Status == "in_sync" { + t.Fatalf("local change was reported in sync: %+v", res) + } + if env.transport.pullTriggers == before { + t.Error("the peer was not told to pull this side's change") + } + if tr.unchanged != 0 { + t.Errorf("unchanged answers = %d with a change on one side, want 0", tr.unchanged) + } +} + +// A peer that predates the question sends the full manifest; it is used, +// not fetched a second time. +func TestQuickInSync_OldPeerManifestIsNotFetchedTwice(t *testing.T) { + env := setupEngine(t) // plain fake: ignores IfHash + write(t, env.localDir, "save.dat", "same") + write(t, env.remoteDir, "save.dat", "same") + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + before := env.transport.manifestCalls + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + if got := env.transport.manifestCalls - before; got != 1 { + t.Errorf("manifest requests for one sync = %d, want 1", got) + } +} diff --git a/internal/p2p/syncengine/transport.go b/internal/p2p/syncengine/transport.go index c39e315c..1db8c5df 100644 --- a/internal/p2p/syncengine/transport.go +++ b/internal/p2p/syncengine/transport.go @@ -2,6 +2,7 @@ package syncengine import ( "context" + "errors" "github.com/opensave/opensave/internal/delta" ) @@ -31,6 +32,38 @@ type SnapshotInfo struct { // with more than one save location. const ProtoMultiRoot = 1 +// ProtoBatchFiles is the revision at which a peer serves the blocks of many +// files in one request (FetchFileBatch). Before it, every file cost its own +// round trip — ~200ms each in practice, which for a save of a quarter-million +// small files (Project Zomboid map chunks) meant half a day for 50MB. +const ProtoBatchFiles = 2 + +// FileBlocksRequest asks for some blocks of one file, as one entry of a batch. +type FileBlocksRequest struct { + RelPath string `json:"relPath"` + BlockIndices []int `json:"blockIndices"` + BlockSize int `json:"blockSize"` +} + +// FileBlocks is one file's answer inside a batch. Error is set, and Blocks +// empty, when the responder could not read that file. +type FileBlocks struct { + RelPath string `json:"relPath"` + Blocks []BlockData `json:"blocks"` + Error string `json:"error,omitempty"` +} + +// BatchFetcher is implemented by transports that can fetch the blocks of +// many files, all in one save location, in a single request. Optional: the +// engine checks for it, and for a peer that advertised ProtoBatchFiles. +type BatchFetcher interface { + FetchFileBatch(ctx context.Context, peer Peer, gameID, root string, files []FileBlocksRequest) ([]FileBlocks, error) +} + +// ErrBatchUnsupported is what a transport returns when it cannot batch for +// this peer (the relay path); the engine then pulls file by file. +var ErrBatchUnsupported = errors.New("batched file fetch is not supported for this peer") + // ManifestResponse is what a peer returns for a manifest request. type ManifestResponse struct { Manifest delta.Manifest `json:"manifest"` @@ -56,6 +89,13 @@ type ManifestResponse struct { // did (see hold.go). Without it an empty location is not taken as every // file in it deleted — see emptiedUnconfirmed. DeletionConfirmed bool `json:"deletionConfirmed,omitempty"` + + // Unchanged answers a request that carried ManifestQuery.IfHash: the + // responder's save still hashes to exactly that, so Manifest is left + // empty. Only ever set when asked; a peer that predates it ignores the + // question and sends the full manifest, which is handled as before. + Unchanged bool `json:"unchanged,omitempty"` + ManifestHash string `json:"manifestHash,omitempty"` } // FileRef identifies one file inside one of a game's save locations. @@ -81,6 +121,9 @@ type ManifestQuery struct { // cover art, instead of a blank tile. AppID string CoverURL string + // IfHash asks the peer to answer "unchanged" instead of its manifest when + // its save still hashes to this — the base both devices last agreed on. + IfHash string } // BlockData is one fetched block. diff --git a/internal/p2p/transport_lan.go b/internal/p2p/transport_lan.go index 1955b640..4405f141 100644 --- a/internal/p2p/transport_lan.go +++ b/internal/p2p/transport_lan.go @@ -35,6 +35,23 @@ type lanTransport struct { var lanClient = &http.Client{Timeout: 30 * time.Second} +// lanBulkClient is for the two requests whose size grows with the save: a +// manifest, and a batch of files. Thirty seconds was fine for a few hundred +// files and not for a few hundred thousand over a VPN; the manifest then never +// arrived, every sync of that game failed, and the devices never converged. +// The sync's own context still bounds it (perPeerSyncTimeout), and the +// transport still gives up on a peer that does not answer at all. +var lanBulkClient = &http.Client{ + Timeout: 10 * time.Minute, + Transport: func() http.RoundTripper { + t := http.DefaultTransport.(*http.Transport).Clone() + // Headers arrive once the peer has built its manifest, which for a big + // save not hashed recently takes a while — but not this long. + t.ResponseHeaderTimeout = 3 * time.Minute + return t + }(), +} + func peerURL(peer syncengine.Peer, route string) string { return fmt.Sprintf("http://%s:%d/api/p2p%s", peer.Address, peer.Port, route) } @@ -54,9 +71,17 @@ func (t *lanTransport) FetchManifest(ctx context.Context, peer syncengine.Peer, if q.CoverURL != "" { params.Set("coverUrl", q.CoverURL) } + if q.IfHash != "" { + params.Set("ifHash", q.IfHash) + } var resp syncengine.ManifestResponse - err := t.getJSON(ctx, peer, peerURL(peer, "/manifest/"+gameID)+"?"+params.Encode(), &resp) + req, err := http.NewRequestWithContext(ctx, http.MethodGet, peerURL(peer, "/manifest/"+gameID)+"?"+params.Encode(), nil) + if err != nil { + return resp, err + } + t.sign(req, peer, nil) + err = doJSONWith(lanBulkClient, req, &resp) return resp, err } @@ -76,6 +101,38 @@ func (t *lanTransport) FetchBlocks(ctx context.Context, peer syncengine.Peer, re return decodeBlocks(resp.Blocks) } +// FetchFileBatch fetches the blocks of many files in one request. Only asked +// of a peer that advertised syncengine.ProtoBatchFiles. +func (t *lanTransport) FetchFileBatch(ctx context.Context, peer syncengine.Peer, gameID, root string, files []syncengine.FileBlocksRequest) ([]syncengine.FileBlocks, error) { + var resp struct { + Files []syncengine.FileBlocks `json:"files"` + } + raw, err := json.Marshal(map[string]any{"root": root, "files": files}) + if err != nil { + return nil, err + } + req, err := http.NewRequestWithContext(ctx, http.MethodPost, peerURL(peer, "/files/"+gameID), bytes.NewReader(raw)) + if err != nil { + return nil, err + } + req.Header.Set("Content-Type", "application/json") + t.sign(req, peer, raw) + if err := doJSONWith(lanBulkClient, req, &resp); err != nil { + return nil, err + } + for i := range resp.Files { + if resp.Files[i].Error != "" { + continue + } + blocks, err := decodeBlocks(resp.Files[i].Blocks) + if err != nil { + return nil, fmt.Errorf("%s: %w", resp.Files[i].RelPath, err) + } + resp.Files[i].Blocks = blocks + } + return resp.Files, nil +} + func (t *lanTransport) DeleteRemote(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef) error { return t.postJSON(ctx, peer, peerURL(peer, "/delete-file/"+ref.GameID), map[string]any{"relPath": ref.RelPath, "root": ref.Root}, nil) } @@ -224,7 +281,11 @@ func (t *lanTransport) sign(req *http.Request, peer syncengine.Peer, body []byte } func doJSON(req *http.Request, out any) error { - resp, err := lanClient.Do(req) + return doJSONWith(lanClient, req, out) +} + +func doJSONWith(client *http.Client, req *http.Request, out any) error { + resp, err := client.Do(req) if err != nil { return err } diff --git a/internal/p2p/transport_wan.go b/internal/p2p/transport_wan.go index 859e91ef..e16e545e 100644 --- a/internal/p2p/transport_wan.go +++ b/internal/p2p/transport_wan.go @@ -154,6 +154,20 @@ func (t *routingTransport) FetchBlocks(ctx context.Context, peer syncengine.Peer return t.pick(peer).FetchBlocks(ctx, peer, ref, blockIndices, blockSize) } +// FetchFileBatch batches over the LAN only. Relay messages are sealed and +// size-limited per client (see BatchIndices), so the relay keeps pulling file +// by file. +func (t *routingTransport) FetchFileBatch(ctx context.Context, peer syncengine.Peer, gameID, root string, files []syncengine.FileBlocksRequest) ([]syncengine.FileBlocks, error) { + if peer.Wan() { + return nil, syncengine.ErrBatchUnsupported + } + b, ok := t.lan.(syncengine.BatchFetcher) + if !ok { + return nil, syncengine.ErrBatchUnsupported + } + return b.FetchFileBatch(ctx, peer, gameID, root, files) +} + func (t *routingTransport) DeleteRemote(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef) error { return t.pick(peer).DeleteRemote(ctx, peer, ref) } From cf9269a80ec50ebd8c14c0fe6246a41497834876 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:53:42 +0200 Subject: [PATCH 03/31] presence: do not trust "online" left over from the previous run Every peer started each run with whatever status the database held, and a peer counts as offline only after three failed probes. For a PC switched off overnight that meant being shown online after every start until three probes - the first ten seconds after the background loops started - had failed. - The presence loop probes at once on start, then every 10s. - The three-miss tolerance is for devices heard from during this run (answered a probe, or seen by discovery, a request or the relay since start). A peer with no sign of life since start goes offline on its first unanswered probe. Adds an opt-in e2e measurement (OPENSAVE_PRESENCE_TIMING) of how long a stopped device is shown online: 30s, by design, for one seen this run. --- e2e/presence_timing_test.go | 53 +++++++++++++++++++++++++++++++++++ internal/p2p/engine.go | 37 +++++++++++++++++++++++- internal/p2p/presence_test.go | 43 ++++++++++++++++++++++++++++ 3 files changed, 132 insertions(+), 1 deletion(-) create mode 100644 e2e/presence_timing_test.go diff --git a/e2e/presence_timing_test.go b/e2e/presence_timing_test.go new file mode 100644 index 00000000..4d9402d6 --- /dev/null +++ b/e2e/presence_timing_test.go @@ -0,0 +1,53 @@ +package e2e + +import ( + "os" + "testing" + "time" + + "github.com/opensave/opensave/testutil" +) + +// TestPresence_GoneDeviceShowsOffline measures how long a device that went +// away keeps being shown online. Opt-in measurement: +// OPENSAVE_PRESENCE_TIMING=1 go test ./e2e -run TestPresence_ -v +func TestPresence_GoneDeviceShowsOffline(t *testing.T) { + if os.Getenv("OPENSAVE_PRESENCE_TIMING") == "" { + t.Skip("set OPENSAVE_PRESENCE_TIMING to run") + } + a := testutil.NewTestDaemon(t, "Presence-A") + b := testutil.NewTestDaemon(t, "Presence-B") + a.PairWith(b) + bID := b.NodeID() + + status := func() string { + p, err := a.Daemon.Store.GetPeer(bID) + if err != nil { + return "?" + } + return p.Status + } + if !testutil.WaitFor(60*time.Second, func() bool { return status() == "online" }) { + t.Fatalf("B never showed online on A: %q", status()) + } + + gone := time.Now() + b.Server.Stop() + b.Daemon.Stop() + + last := status() + for time.Since(gone) < 10*time.Minute { + if s := status(); s != last { + t.Logf("%6.1fs after B stopped: %s -> %s", time.Since(gone).Seconds(), last, s) + last = s + } + if last == "offline" { + break + } + time.Sleep(500 * time.Millisecond) + } + t.Logf("B shown online for %.1fs after it stopped", time.Since(gone).Seconds()) + if last != "offline" { + t.Fatalf("B still %q 10 minutes after it stopped", last) + } +} diff --git a/internal/p2p/engine.go b/internal/p2p/engine.go index f99603bf..aae34995 100644 --- a/internal/p2p/engine.go +++ b/internal/p2p/engine.go @@ -125,6 +125,10 @@ type Engine struct { pingMisses map[string]int // probeMu serializes PingPairedPeers rounds. probeMu sync.Mutex + // answered holds the peers that answered a probe during this run, and + // startedMs is when the run began (heardThisRun). + answered map[string]bool + startedMs int64 // Lineage refreshes already running after a peer-applied deletion, keyed // by game+peer. Deletions arrive one file at a time, and each one leaves @@ -289,6 +293,7 @@ func New(s *store.Store, snaps *snapshot.Manager, logf func(level, msg string)) Log: logf, ctx: ctx, cancel: cancel, + startedMs: time.Now().UnixMilli(), } e.Wan = newWanClient(e) e.RelayHost = NewRelayHost(logf) @@ -385,6 +390,26 @@ func (e *Engine) clearPingMisses(peerID string) { delete(e.pingMisses, peerID) } +// noteAnswered records that a peer answered a probe during this run. +func (e *Engine) noteAnswered(peerID string) { + e.pingMu.Lock() + defer e.pingMu.Unlock() + if e.answered == nil { + e.answered = map[string]bool{} + } + e.answered[peerID] = true +} + +// heardThisRun reports whether there is any sign of a peer since this engine +// started: a probe it answered, or a sighting (discovery, a request from it, +// the relay) recent enough to postdate the start. +func (e *Engine) heardThisRun(p store.Peer) bool { + e.pingMu.Lock() + answered := e.answered[p.ID] + e.pingMu.Unlock() + return answered || p.LastSeenMs >= e.startedMs +} + // PingPairedPeers probes every paired LAN peer and updates their // online/offline status. func (e *Engine) PingPairedPeers(ctx context.Context) { @@ -425,11 +450,18 @@ func (e *Engine) PingPairedPeers(ctx context.Context) { // failed cannot lose anything: a device that really has gone is // declared offline a few probes later, and the only cost is a sync // attempt that fails the way it would have anyway. + // + // That patience is for a device seen during this run. A status + // carried over from the previous run is not evidence of anything: a + // PC switched off overnight came back "online" from the database on + // every start, and stayed so until three probes had failed. A device + // not heard from since this start goes offline on its first miss. newStatus := p.Status if ok { e.clearPingMisses(p.ID) + e.noteAnswered(p.ID) newStatus = "online" - } else if e.notePingMiss(p.ID) >= offlineStrikes { + } else if misses := e.notePingMiss(p.ID); misses >= offlineStrikes || !e.heardThisRun(p) { newStatus = "offline" } @@ -603,6 +635,9 @@ var presenceInterval = 10 * time.Second // whatever sync work is running. It stops with the resync loop. func (e *Engine) startPresenceLoop(stop chan struct{}) { e.GoSync(func(ctx context.Context) { + // At once, then on the ticker: until the first round, every peer + // shows whatever the previous run left in the database. + e.PingPairedPeers(ctx) ticker := time.NewTicker(presenceInterval) defer ticker.Stop() for { diff --git a/internal/p2p/presence_test.go b/internal/p2p/presence_test.go index de0397de..aca90a06 100644 --- a/internal/p2p/presence_test.go +++ b/internal/p2p/presence_test.go @@ -25,6 +25,49 @@ func waitUntil(d time.Duration, cond func() bool) bool { return cond() } +// A PC switched off since the last run is in the database as "online". It must +// not stay so for the three probes a device seen this run is allowed: the +// first unanswered probe after start says it is not there. +func TestPresence_StaleOnlineFromLastRunClearsAtOnce(t *testing.T) { + old := presenceInterval + presenceInterval = time.Hour // only the probe at start + t.Cleanup(func() { presenceInterval = old }) + + s, err := store.Open(filepath.Join(t.TempDir(), "opensave.db")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { s.Close() }) + if err := s.EnsureDefaultSettings(t.TempDir(), t.TempDir()); err != nil { + t.Fatal(err) + } + // Nothing listens here: the PC is off. + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + port := ln.Addr().(*net.TCPAddr).Port + ln.Close() + lastRun := time.Now().Add(-10 * time.Hour).UnixMilli() + if err := s.UpsertPeer(store.Peer{ID: "node_off", Name: "Switched off", Address: "127.0.0.1", Port: port, Status: "online", LastSeenMs: lastRun}); err != nil { + t.Fatal(err) + } + + e := New(s, snapshot.New(s), func(string, string) {}) + stop := make(chan struct{}) + start := time.Now() + e.startPresenceLoop(stop) + t.Cleanup(func() { close(stop); e.cancel(); e.bgSyncWG.Wait() }) + + if !waitUntil(5*time.Second, func() bool { + p, err := s.GetPeer("node_off") + return err == nil && p.Status == "offline" + }) { + t.Fatal("a peer online only in the last run's record was still online after the first probe") + } + t.Logf("shown offline %.1fs after start", time.Since(start).Seconds()) +} + // A peer's online state follows it on its own ticker, whatever else the // engine is doing — not only when a sync or the minute-long reconcile happens // to probe. A peer reached over a VPN gets no UDP broadcasts, so without this From 9cbe583c57b33bd2a043e2e047ab7d0e7b07dccb Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:50:18 +0200 Subject: [PATCH 04/31] sync: closest devices first, and no fixed 30-second limit on block fetches Network-aware order: - Each device records how fast its connection to every other one is: from the pulls it makes, and from a short speed test (/api/p2p/speedtest, 2 MB, at most every 6 hours) when nothing has measured it. Before anything is measured, the address kind stands in: home network fast, VPN (Tailscale's range) slower, the internet relay slowest. - A sync goes to the fastest devices first. A device that is behind takes the newer save from the fastest device holding it. - Having asked a close device to take a newer save, a sync waits for it to finish (it reports sync-complete) before asking a much slower one, bounded by how long the transfer should take. The save reaches the close devices at full speed, and the slow ones can then take it from whichever device is nearest to them. - The device list shows each connection's kind and measured speed. Block fetches on direct addresses had a fixed 30-second limit. "Direct" includes VPN addresses, and a few MB of blocks at a VPN relay's hundred-odd KB a second outlasted it: the request failed, and with it the whole pull, over and over. The limit now grows with what is asked for, as it already did over the internet relay. Co-Authored-By: Claude Opus 5.5 --- .../frontend/src/views/Devices.svelte | 14 + internal/api/server.go | 7 +- internal/p2p/engine.go | 34 +++ internal/p2p/routes.go | 36 +++ internal/p2p/syncengine/engine.go | 35 ++- internal/p2p/syncengine/linkspeed.go | 273 ++++++++++++++++++ internal/p2p/syncengine/linkspeed_test.go | 106 +++++++ internal/p2p/transport_lan.go | 48 ++- internal/p2p/transport_wan.go | 11 + internal/p2p/wanclient_handlers.go | 1 + internal/store/migrations/0038_peer_links.sql | 15 + internal/store/peer_links.go | 37 +++ 12 files changed, 613 insertions(+), 4 deletions(-) create mode 100644 internal/p2p/syncengine/linkspeed.go create mode 100644 internal/p2p/syncengine/linkspeed_test.go create mode 100644 internal/store/migrations/0038_peer_links.sql create mode 100644 internal/store/peer_links.go diff --git a/cmd/opensave-app/frontend/src/views/Devices.svelte b/cmd/opensave-app/frontend/src/views/Devices.svelte index 641c5f3b..4b53ad09 100644 --- a/cmd/opensave-app/frontend/src/views/Devices.svelte +++ b/cmd/opensave-app/frontend/src/views/Devices.svelte @@ -76,6 +76,15 @@ }; const fmtTime = (t) => (t ? new Date(t).toLocaleString() : 'never'); + + // How this device reaches the peer, and how fast it has found that to be. + const linkKinds = { lan: 'home network', vpn: 'VPN (e.g. Tailscale)', relay: 'internet relay', internet: 'internet' }; + const fmtRate = (b) => (b >= 1048576 ? (b / 1048576).toFixed(1) + ' MB/s' : Math.max(1, Math.round(b / 1024)) + ' KB/s'); + function linkLabel(link) { + const kind = linkKinds[link.kind] ?? link.kind; + if (!link.bytesPerSec) return `${kind} · speed not measured yet`; + return `${kind} · ${fmtRate(link.bytesPerSec)} (measured ${fmtTime(link.measuredMs)})`; + }
@@ -105,6 +114,11 @@ · last synced {fmtTime(peer.lastSynced)} {#if peer.appVersion}· OpenSave {peer.appVersion}{/if}
+ {#if peer.link} +
+ {linkLabel(peer.link)} +
+ {/if}
diff --git a/internal/api/server.go b/internal/api/server.go index 66a32562..fea038ed 100644 --- a/internal/api/server.go +++ b/internal/api/server.go @@ -98,12 +98,17 @@ func (s *Server) peersPayload() map[string]any { AppVersion string `json:"appVersion,omitempty"` BuildTimeMs int64 `json:"buildTimeMs,omitempty"` HasNewerBuild bool `json:"hasNewerBuild,omitempty"` + // Link is how fast this device has found its connection to it, + // and what kind of address it is reached on: what decides the + // order devices are synced in. + Link syncengine.LinkStat `json:"link"` // What actually protects traffic with this device. Sent per peer // rather than described once in the interface, because the answer // differs per pairing and the reader cannot work out which case // they are in from a general statement. p2p.PeerProtection - }{Peer: p, PeerProtection: s.Daemon.P2P.PeerProtection(p)} + }{Peer: p, PeerProtection: s.Daemon.P2P.PeerProtection(p), + Link: s.Daemon.P2P.Sync.Link(syncengine.Peer{ID: p.ID, Name: p.Name, Address: p.Address, Port: p.Port, IsWan: p.Address == "relay"})} if b, ok := builds[p.ID]; ok { entry.AppVersion = b.AppVersion entry.BuildTimeMs = b.BuildTimeMs diff --git a/internal/p2p/engine.go b/internal/p2p/engine.go index aae34995..1dd69bbb 100644 --- a/internal/p2p/engine.go +++ b/internal/p2p/engine.go @@ -125,6 +125,10 @@ type Engine struct { pingMisses map[string]int // probeMu serializes PingPairedPeers rounds. probeMu sync.Mutex + // probingLinks: peers whose connection is being speed-tested now + // (probeLinkSoon). + probeLinksMu sync.Mutex + probingLinks map[string]bool // answered holds the peers that answered a probe during this run, and // startedMs is when the run began (heardThisRun). answered map[string]bool @@ -461,6 +465,7 @@ func (e *Engine) PingPairedPeers(ctx context.Context) { e.clearPingMisses(p.ID) e.noteAnswered(p.ID) newStatus = "online" + e.probeLinkSoon(p) } else if misses := e.notePingMiss(p.ID); misses >= offlineStrikes || !e.heardThisRun(p) { newStatus = "offline" } @@ -477,6 +482,35 @@ func (e *Engine) PingPairedPeers(ctx context.Context) { } } +// probeLinkSoon tests the connection to a peer that just answered, in the +// background, when nothing has measured it for a while +// (syncengine/linkspeed.go). One test per peer at a time. +func (e *Engine) probeLinkSoon(p store.Peer) { + if e.Sync == nil || p.Address == "relay" { + return + } + e.probeLinksMu.Lock() + if e.probingLinks == nil { + e.probingLinks = map[string]bool{} + } + if e.probingLinks[p.ID] { + e.probeLinksMu.Unlock() + return + } + e.probingLinks[p.ID] = true + e.probeLinksMu.Unlock() + go func() { + defer func() { + e.probeLinksMu.Lock() + delete(e.probingLinks, p.ID) + e.probeLinksMu.Unlock() + }() + if e.Sync.ProbeLinkIfDue(context.Background(), syncengine.Peer{ID: p.ID, Name: p.Name, Address: p.Address, Port: p.Port}) { + e.notifyPeerUpdate() + } + }() +} + // ErrNoPeersOnline is a sync with no other device to sync with: none of the // paired devices answered. Not a failure of anything — the save goes over when // one is back — which is why it is told apart from errors that are. diff --git a/internal/p2p/routes.go b/internal/p2p/routes.go index 2d0ac70b..7d84c1c3 100644 --- a/internal/p2p/routes.go +++ b/internal/p2p/routes.go @@ -3,6 +3,7 @@ package p2p import ( "compress/gzip" "context" + "crypto/rand" "encoding/json" "errors" "fmt" @@ -42,6 +43,7 @@ func (e *Engine) RegisterRoutes(r chi.Router) { r.Get("/api/p2p/games", e.handlePeerGameList) r.Post("/api/p2p/sync-event/{gameId}", e.handleSyncEvent) r.Get("/api/p2p/app-binary", e.handleAppBinary) + r.Get("/api/p2p/speedtest", e.handleSpeedTest) // What moves save data, or starts a sync: refused while paused. r.Group(func(r chi.Router) { @@ -55,6 +57,34 @@ func (e *Engine) RegisterRoutes(r chi.Router) { }) } +// speedTestMax bounds one speed test, and speedTestData is what it sends: +// random, so nothing along the way can compress it into a better figure. +const speedTestMax = 8 << 20 + +var speedTestData = func() []byte { + b := make([]byte, 1<<20) + _, _ = rand.Read(b) + return b +}() + +// handleSpeedTest sends n bytes (at most speedTestMax) for a peer to time: +// how it learns how close this device is (syncengine/linkspeed.go). +func (e *Engine) handleSpeedTest(w http.ResponseWriter, r *http.Request) { + n, _ := strconv.Atoi(r.URL.Query().Get("n")) + if n <= 0 || n > speedTestMax { + n = speedTestMax + } + w.Header().Set("Content-Type", "application/octet-stream") + w.Header().Set("Content-Length", strconv.Itoa(n)) + for n > 0 { + chunk := min(n, len(speedTestData)) + if _, err := w.Write(speedTestData[:chunk]); err != nil { + return + } + n -= chunk + } +} + func clientIP(r *http.Request) string { host, _, err := net.SplitHostPort(r.RemoteAddr) if err != nil { @@ -1048,6 +1078,10 @@ func (e *Engine) handleSyncEvent(w http.ResponseWriter, r *http.Request) { if e.Sync.Progress.OnSyncError != nil { e.Sync.Progress.OnSyncError(gameID, ev) } + // A sync here waiting for this peer to finish need not wait on. + if peer, ok := e.peerByAddress(clientIP(r)); ok { + e.Sync.NotePeerPulled(gameID, peer.ID) + } } jsonOK(w, map[string]any{"success": true}) } @@ -1175,6 +1209,8 @@ func SetServedProto(v int) int { // freshly-pushed files deliberately stay out of it (see persistLineage), so // deleting one here would pull it back instead of propagating the delete. func (e *Engine) peerFinishedPulling(gameID string, peer syncengine.Peer, data map[string]any) { + // A sync here waiting for this peer before slower ones goes on. + defer e.Sync.NotePeerPulled(gameID, peer.ID) // Recorded first, and synchronously: the peer said exactly which files it // wrote, so the lineage can be updated now rather than after a manifest // round trip. That round trip is what left a window in which deleting a diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index ac58f604..230f3d65 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -143,6 +143,13 @@ type Engine struct { // rootConflicts holds divergences in a game's EXTRA save locations, keyed // by game and location so several can wait on a decision at once. rootConflicts map[string]*RootConflict + + // linkMu guards links, the measured connection to each peer, and + // pullWaits, syncs waiting for a close peer to finish taking a save + // (linkspeed.go). + linkMu sync.Mutex + links map[string]store.PeerLink + pullWaits map[string]chan struct{} } // New creates an Engine. @@ -310,13 +317,32 @@ func (e *Engine) SyncGame(ctx context.Context, gameID string, onlinePeers []Peer results := map[string]Result{} busy := false - for _, peer := range onlinePeers { + // Fastest connections first (linkspeed.go): a device that is behind takes + // the newer save from the closest device holding it, and a newer save + // here reaches the close devices before the slow ones. + onlinePeers = e.OrderBySpeed(onlinePeers) + for i, peer := range onlinePeers { // Hard per-peer cap: a wedged transport must never hold // activeSyncs forever (which would silently block every future // sync of this game until an app restart). peerCtx, cancel := context.WithTimeout(ctx, perPeerSyncTimeout) res, err := e.SyncWithPeer(peerCtx, gameID, peer) cancel() + // A close device asked to take this save is given the time to, before + // a much slower one is asked: it gets it at full speed, and the slow + // one can then take it from whichever device is nearest to it. + if err == nil && res.Status == "triggered_peer_pull" && e.waitsForClose(onlinePeers, i) { + var size int64 + if game, gerr := e.Store.GetGame(gameID); gerr == nil { + if m, merr := e.ReadManifest(ctx, gameID, game.SavePath); merr == nil { + for _, f := range m.Files { + size += f.Size + } + } + } + e.Log("info", fmt.Sprintf("waiting for %s (close by) to take %s before the slower devices", peer.Name, gameID)) + e.waitForPull(ctx, gameID, peer, size) + } if err == nil && res.Status == "peer_busy" { // Nothing was compared, so nothing is stamped, and it is not a // failure: the peer was writing this game for a sync of its own. @@ -1584,6 +1610,13 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s tracker := newProgressTracker(totalBytes) throttle := e.throttleFor(peer.Wan()) + // What this transfer says about the connection (linkspeed.go), measured + // however it ends: a pull cut off half-way still moved what it moved. + pullStarted := time.Now() + defer func() { + moved, _, _ := tracker.stats() + e.noteLinkRate(peer, moved, time.Since(pullStarted)) + }() // Progress reporter shared by the per-file loop and the block-group // loop inside each file. Without in-file reporting, a single large diff --git a/internal/p2p/syncengine/linkspeed.go b/internal/p2p/syncengine/linkspeed.go new file mode 100644 index 00000000..b19543b1 --- /dev/null +++ b/internal/p2p/syncengine/linkspeed.go @@ -0,0 +1,273 @@ +package syncengine + +import ( + "context" + "fmt" + "net" + "sort" + "time" + + "github.com/opensave/opensave/internal/store" +) + +// Network-aware syncing. +// +// Devices are not equally close. Two on the same home network move a save at +// tens of MB a second; one reached over a VPN relay or the internet relay may +// manage a hundred KB. A save that goes to every device at once spends hours +// crawling over the slow links while the device next door, which could have +// had it in seconds and then handed it on, waits in the same queue. +// +// So each device keeps how fast it has found its connection to every other — +// measured from the transfers it makes, and from a short speed test when it +// has nothing recent — and works fastest first: a sync goes to the close +// devices first, and a device that is behind takes the newer save from the +// fastest device that holds it. Before anything is measured, the kind of +// address stands in: a home-network address is taken for fast, a VPN address +// (Tailscale's range) for slower, the internet relay for slowest. + +// LinkStat is what this device knows about its connection to a peer. +type LinkStat struct { + // BytesPerSec is the measured rate, 0 when nothing is measured yet. + BytesPerSec float64 `json:"bytesPerSec"` + MeasuredMs int64 `json:"measuredMs"` + // Kind is what the address says: "lan", "vpn", "relay" or "internet". + Kind string `json:"kind"` +} + +const ( + // minMeasureBytes and minMeasureTime: smaller transfers say more about + // round trips than about the link, and are not counted. + minMeasureBytes = 512 << 10 + minMeasureTime = 500 * time.Millisecond + // probeEvery is how often a link is tested again when no transfer has + // measured it since; probeBytes how much a test moves. + probeEvery = 6 * time.Hour + probeBytes = 2 << 20 +) + +// prior rates, by address kind, used until a link is measured. +var priorRate = map[string]float64{ + "lan": 40 << 20, + "vpn": 2 << 20, + "internet": 1 << 20, + "relay": 256 << 10, +} + +var ( + _, tailscaleNet, _ = net.ParseCIDR("100.64.0.0/10") + privateNets = func() []*net.IPNet { + var out []*net.IPNet + for _, c := range []string{"10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16", "169.254.0.0/16", "fc00::/7", "fe80::/10"} { + _, n, _ := net.ParseCIDR(c) + out = append(out, n) + } + return out + }() +) + +// linkKind says what a peer's address is. +func linkKind(peer Peer) string { + if peer.Wan() { + return "relay" + } + ip := net.ParseIP(peer.Address) + if ip == nil { + return "internet" + } + if ip.IsLoopback() { + return "lan" + } + if tailscaleNet.Contains(ip) { + return "vpn" + } + for _, n := range privateNets { + if n.Contains(ip) { + return "lan" + } + } + return "internet" +} + +func (e *Engine) loadLinksLocked() { + if e.links != nil { + return + } + e.links = map[string]store.PeerLink{} + if e.Store == nil { + return + } + if links, err := e.Store.PeerLinks(); err == nil { + e.links = links + } +} + +// Link is what this device knows about its connection to a peer. +func (e *Engine) Link(peer Peer) LinkStat { + e.linkMu.Lock() + defer e.linkMu.Unlock() + e.loadLinksLocked() + l := e.links[peer.ID] + return LinkStat{BytesPerSec: l.BytesPerSec, MeasuredMs: l.MeasuredMs, Kind: linkKind(peer)} +} + +// expectedRate is how fast a transfer with a peer is expected to go: what was +// measured, or the address kind's guess. +func (e *Engine) expectedRate(peer Peer) float64 { + l := e.Link(peer) + if l.BytesPerSec > 0 { + return l.BytesPerSec + } + return priorRate[l.Kind] +} + +// noteLinkRate records a transfer with a peer: bytes in d. +func (e *Engine) noteLinkRate(peer Peer, bytes int64, d time.Duration) { + if bytes < minMeasureBytes || d < minMeasureTime { + return + } + rate := float64(bytes) / d.Seconds() + e.linkMu.Lock() + defer e.linkMu.Unlock() + e.loadLinksLocked() + l := e.links[peer.ID] + l.PeerID = peer.ID + if l.BytesPerSec > 0 { + // A running average, weighted to the latest: a link changes (a VPN + // that found a direct path, a laptop that moved networks), and the + // order should follow it within a transfer or two. + l.BytesPerSec = 0.4*l.BytesPerSec + 0.6*rate + } else { + l.BytesPerSec = rate + } + l.MeasuredMs = time.Now().UnixMilli() + e.links[peer.ID] = l + if e.Store != nil { + _ = e.Store.SavePeerLink(l) + } +} + +// OrderBySpeed sorts peers fastest first, keeping the given order among +// equals. +func (e *Engine) OrderBySpeed(peers []Peer) []Peer { + out := append([]Peer(nil), peers...) + rates := make(map[string]float64, len(out)) + for _, p := range out { + rates[p.ID] = e.expectedRate(p) + } + sort.SliceStable(out, func(i, j int) bool { return rates[out[i].ID] > rates[out[j].ID] }) + return out +} + +// SpeedProber is implemented by transports that can test a link: fetch n +// bytes of nothing in particular from the peer and say how long it took. +type SpeedProber interface { + ProbeSpeed(ctx context.Context, peer Peer, n int) (int64, time.Duration, error) +} + +// ProbeLinkIfDue tests the link to a peer when nothing has measured it for a +// while. Not over the internet relay, which is the slowest link whatever a +// test says and would only be loaded further. Reports whether it ran. +func (e *Engine) ProbeLinkIfDue(ctx context.Context, peer Peer) bool { + prober, ok := e.Transport.(SpeedProber) + if !ok || peer.Wan() { + return false + } + e.linkMu.Lock() + e.loadLinksLocked() + l := e.links[peer.ID] + now := time.Now() + due := now.Sub(time.UnixMilli(l.MeasuredMs)) > probeEvery && now.Sub(time.UnixMilli(l.ProbedMs)) > probeEvery + if due { + l.PeerID = peer.ID + l.ProbedMs = now.UnixMilli() + e.links[peer.ID] = l + if e.Store != nil { + _ = e.Store.SavePeerLink(l) + } + } + e.linkMu.Unlock() + if !due { + return false + } + ctx, cancel := context.WithTimeout(ctx, 60*time.Second) + defer cancel() + n, d, err := prober.ProbeSpeed(ctx, peer, probeBytes) + if err != nil { + e.Log("info", fmt.Sprintf("speed test with %s did not finish: %v", peer.Name, err)) + return true + } + e.noteLinkRate(peer, n, d) + e.Log("info", fmt.Sprintf("connection to %s: %s/s (%s)", peer.Name, humanBytes(int64(e.Link(peer).BytesPerSec)), linkKind(peer))) + return true +} + +// Waiting for close devices before slow ones. + +// fastLink is the rate from which a device counts as close, and slowerBy how +// much slower the next device must be for this one to be waited for first. +const ( + fastLink = 4 << 20 + slowerBy = 4.0 + maxWaitFor = 20 * time.Minute +) + +// NotePeerPulled is told that a peer finished (or gave up) pulling a game from +// this device, and lets a sync waiting for it go on. +func (e *Engine) NotePeerPulled(gameID, peerID string) { + e.linkMu.Lock() + ch := e.pullWaits[gameID+"|"+peerID] + delete(e.pullWaits, gameID+"|"+peerID) + e.linkMu.Unlock() + if ch != nil { + close(ch) + } +} + +// waitForPull waits, after asking a close peer to take this device's save, +// until it has — before the slow devices are asked, so the save reaches the +// close one at full speed and the slow ones can then take it from whichever +// device is closest to them. Bounded by how long the transfer should take, +// and given up when the context ends. +func (e *Engine) waitForPull(ctx context.Context, gameID string, peer Peer, bytes int64) { + ch := make(chan struct{}) + e.linkMu.Lock() + if e.pullWaits == nil { + e.pullWaits = map[string]chan struct{}{} + } + e.pullWaits[gameID+"|"+peer.ID] = ch + e.linkMu.Unlock() + defer func() { + e.linkMu.Lock() + if e.pullWaits[gameID+"|"+peer.ID] == ch { + delete(e.pullWaits, gameID+"|"+peer.ID) + } + e.linkMu.Unlock() + }() + wait := 30*time.Second + time.Duration(3*float64(bytes)/e.expectedRate(peer)*float64(time.Second)) + if wait > maxWaitFor { + wait = maxWaitFor + } + select { + case <-ch: + case <-time.After(wait): + e.Log("info", fmt.Sprintf("%s has not finished taking %s yet; going on with the other devices", peer.Name, gameID)) + case <-ctx.Done(): + } +} + +// waitsForClose reports whether, having asked peers[i] to take this device's +// save, the sync should wait for it before going on: it is a close device, +// and a much slower one is still to come. +func (e *Engine) waitsForClose(peers []Peer, i int) bool { + r := e.expectedRate(peers[i]) + if r < fastLink { + return false + } + for _, p := range peers[i+1:] { + if e.expectedRate(p)*slowerBy <= r { + return true + } + } + return false +} diff --git a/internal/p2p/syncengine/linkspeed_test.go b/internal/p2p/syncengine/linkspeed_test.go new file mode 100644 index 00000000..3e8544dc --- /dev/null +++ b/internal/p2p/syncengine/linkspeed_test.go @@ -0,0 +1,106 @@ +package syncengine + +import ( + "context" + "testing" + "time" +) + +func TestLinkKind(t *testing.T) { + cases := map[string]string{ + "192.168.178.20": "lan", + "10.0.0.5": "lan", + "127.0.0.1": "lan", + "100.99.121.82": "vpn", + "203.0.113.9": "internet", + } + for addr, want := range cases { + if got := linkKind(Peer{Address: addr}); got != want { + t.Errorf("%s: %s, want %s", addr, got, want) + } + } + if got := linkKind(Peer{Address: "relay"}); got != "relay" { + t.Errorf("relay: %s", got) + } +} + +// Before anything is measured the kind of address orders the devices; once a +// link is measured, the measurement does. +func TestOrderBySpeed(t *testing.T) { + env := setupEngine(t) + e := env.engine + home := Peer{ID: "home", Address: "192.168.178.20"} + vpn := Peer{ID: "vpn", Address: "100.99.121.82"} + relay := Peer{ID: "relay", Address: "relay"} + + got := e.OrderBySpeed([]Peer{relay, vpn, home}) + if got[0].ID != "home" || got[1].ID != "vpn" || got[2].ID != "relay" { + t.Errorf("by address: %v", ids(got)) + } + + // A VPN link that turns out fast, and a home one that turns out slow. + e.noteLinkRate(vpn, 200<<20, 2*time.Second) + e.noteLinkRate(home, 1<<20, 10*time.Second) + got = e.OrderBySpeed([]Peer{home, vpn, relay}) + if got[0].ID != "vpn" { + t.Errorf("by measurement: %v", ids(got)) + } + // Remembered across restarts. + e2 := New(env.store, nil, env.transport) + if e2.Link(vpn).BytesPerSec == 0 { + t.Error("the measured link was not kept") + } +} + +func TestNoteLinkRate_IgnoresTinyTransfers(t *testing.T) { + env := setupEngine(t) + p := Peer{ID: "p", Address: "192.168.1.2"} + env.engine.noteLinkRate(p, 10<<10, 2*time.Second) + if env.engine.Link(p).BytesPerSec != 0 { + t.Error("a transfer of a few KB was taken as the link's speed") + } +} + +// A close device is waited for before a much slower one, and the wait ends as +// soon as it reports that it has finished. +func TestWaitForClose(t *testing.T) { + env := setupEngine(t) + e := env.engine + home := Peer{ID: "home", Address: "192.168.178.20"} + vpn := Peer{ID: "vpn", Address: "100.99.121.82"} + peers := e.OrderBySpeed([]Peer{vpn, home}) + if !e.waitsForClose(peers, 0) { + t.Fatal("a close device is not waited for before a slow one") + } + if e.waitsForClose(peers, 1) { + t.Error("the last device is waited for, with nothing after it") + } + if e.waitsForClose(e.OrderBySpeed([]Peer{home}), 0) { + t.Error("waited with no slower device to follow") + } + + done := make(chan struct{}) + start := time.Now() + go func() { + e.waitForPull(context.Background(), "game1", home, 1<<30) + close(done) + }() + time.Sleep(100 * time.Millisecond) + e.NotePeerPulled("game1", home.ID) + select { + case <-done: + if time.Since(start) > 5*time.Second { + t.Error("the wait did not end when the device reported") + } + case <-time.After(10 * time.Second): + t.Fatal("the wait did not end when the device reported") + } +} + +func ids(ps []Peer) []string { + out := make([]string, len(ps)) + for i, p := range ps { + out[i] = p.ID + } + return out +} diff --git a/internal/p2p/transport_lan.go b/internal/p2p/transport_lan.go index 4405f141..e926f0b7 100644 --- a/internal/p2p/transport_lan.go +++ b/internal/p2p/transport_lan.go @@ -92,15 +92,59 @@ func (t *lanTransport) FetchBlocks(ctx context.Context, peer syncengine.Peer, re // No encodings advertised: on a LAN the wire is typically faster than the // compressor, so the bytes saved cost more than they're worth. Responses // are still decoded, so a peer that compresses anyway is handled. - err := t.postJSON(ctx, peer, peerURL(peer, "/blocks/"+ref.GameID), map[string]any{ + // + // Given time for what it asks for. "LAN" here is any direct address, + // a VPN one (Tailscale) included, and a few MB of blocks at a VPN relay's + // hundred-odd KB a second outlasted the fixed 30 seconds every request + // used to get: the request failed, and with it the whole pull, over and + // over. Sized like the relay's (slowLinkBytesPerSec), within the bulk + // client's ceiling. + expected := int64(len(blockIndices)) * int64(max(blockSize, 1)) + ctx, cancel := context.WithTimeout(ctx, 30*time.Second+time.Duration(expected/(slowLinkBytesPerSec/2))*time.Second) + defer cancel() + raw, err := json.Marshal(map[string]any{ "relPath": ref.RelPath, "root": ref.Root, "blockIndices": blockIndices, "blockSize": blockSize, - }, &resp) + }) + if err != nil { + return nil, err + } + req, err := http.NewRequestWithContext(ctx, http.MethodPost, peerURL(peer, "/blocks/"+ref.GameID), bytes.NewReader(raw)) if err != nil { return nil, err } + req.Header.Set("Content-Type", "application/json") + t.sign(req, peer, raw) + if err := doJSONWith(lanBulkClient, req, &resp); err != nil { + return nil, err + } return decodeBlocks(resp.Blocks) } +// ProbeSpeed times fetching n bytes from the peer's speed test +// (syncengine.SpeedProber). The clock starts once the answer begins to +// arrive, so the figure is the link's rate rather than its round trip. +func (t *lanTransport) ProbeSpeed(ctx context.Context, peer syncengine.Peer, n int) (int64, time.Duration, error) { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, peerURL(peer, fmt.Sprintf("/speedtest?n=%d", n)), nil) + if err != nil { + return 0, 0, err + } + t.sign(req, peer, nil) + resp, err := lanBulkClient.Do(req) + if err != nil { + return 0, 0, err + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + return 0, 0, fmt.Errorf("peer returned %d", resp.StatusCode) + } + start := time.Now() + got, err := io.Copy(io.Discard, resp.Body) + if err != nil { + return got, time.Since(start), err + } + return got, time.Since(start), nil +} + // FetchFileBatch fetches the blocks of many files in one request. Only asked // of a peer that advertised syncengine.ProtoBatchFiles. func (t *lanTransport) FetchFileBatch(ctx context.Context, peer syncengine.Peer, gameID, root string, files []syncengine.FileBlocksRequest) ([]syncengine.FileBlocks, error) { diff --git a/internal/p2p/transport_wan.go b/internal/p2p/transport_wan.go index e16e545e..2de42d0d 100644 --- a/internal/p2p/transport_wan.go +++ b/internal/p2p/transport_wan.go @@ -3,6 +3,7 @@ package p2p import ( "context" "encoding/json" + "errors" "fmt" "time" @@ -168,6 +169,16 @@ func (t *routingTransport) FetchFileBatch(ctx context.Context, peer syncengine.P return b.FetchFileBatch(ctx, peer, gameID, root, files) } +// ProbeSpeed tests LAN links only (syncengine.SpeedProber); the relay is the +// slowest link whatever a test would say. +func (t *routingTransport) ProbeSpeed(ctx context.Context, peer syncengine.Peer, n int) (int64, time.Duration, error) { + p, ok := t.lan.(syncengine.SpeedProber) + if peer.Wan() || !ok { + return 0, 0, errors.New("no speed test over the relay") + } + return p.ProbeSpeed(ctx, peer, n) +} + func (t *routingTransport) DeleteRemote(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef) error { return t.pick(peer).DeleteRemote(ctx, peer, ref) } diff --git a/internal/p2p/wanclient_handlers.go b/internal/p2p/wanclient_handlers.go index 95203dd3..bd3b8e07 100644 --- a/internal/p2p/wanclient_handlers.go +++ b/internal/p2p/wanclient_handlers.go @@ -172,6 +172,7 @@ func (w *WanClient) handleMessage(ctx context.Context, msg RelayMessage) { if w.engine.Sync.Progress.OnSyncError != nil { w.engine.Sync.Progress.OnSyncError(msg.GameID, ev) } + w.engine.Sync.NotePeerPulled(msg.GameID, msg.From) } case "request": diff --git a/internal/store/migrations/0038_peer_links.sql b/internal/store/migrations/0038_peer_links.sql new file mode 100644 index 00000000..4cff6cfb --- /dev/null +++ b/internal/store/migrations/0038_peer_links.sql @@ -0,0 +1,15 @@ +-- How fast this device has measured its connection to each paired device: +-- from transfers it made, and from a short speed test when there is nothing +-- recent (see internal/p2p/syncengine/linkspeed.go). Syncs go to the fastest +-- connections first, so a save reaches the devices close to it before it +-- crawls over a slow relay — and those can then hand it on. +-- +-- bytes_per_sec is a running average; measured_ms when it last changed; +-- probed_ms when a speed test last ran, so a slow link is not tested over and +-- over. +CREATE TABLE peer_links ( + peer_id TEXT PRIMARY KEY, + bytes_per_sec REAL NOT NULL DEFAULT 0, + measured_ms INTEGER NOT NULL DEFAULT 0, + probed_ms INTEGER NOT NULL DEFAULT 0 +); diff --git a/internal/store/peer_links.go b/internal/store/peer_links.go new file mode 100644 index 00000000..d65b3a37 --- /dev/null +++ b/internal/store/peer_links.go @@ -0,0 +1,37 @@ +package store + +import "fmt" + +// PeerLink is how fast this device has measured its connection to a peer. +// See migrations/0038_peer_links.sql. +type PeerLink struct { + PeerID string `db:"peer_id"` + BytesPerSec float64 `db:"bytes_per_sec"` + MeasuredMs int64 `db:"measured_ms"` + ProbedMs int64 `db:"probed_ms"` +} + +// PeerLinks returns every recorded link, by peer. +func (s *Store) PeerLinks() (map[string]PeerLink, error) { + var rows []PeerLink + if err := s.db.Select(&rows, `SELECT * FROM peer_links`); err != nil { + return nil, fmt.Errorf("list peer links: %w", err) + } + out := make(map[string]PeerLink, len(rows)) + for _, r := range rows { + out[r.PeerID] = r + } + return out, nil +} + +// SavePeerLink records a peer's link, replacing what was there. +func (s *Store) SavePeerLink(l PeerLink) error { + _, err := s.db.Exec(`INSERT INTO peer_links (peer_id, bytes_per_sec, measured_ms, probed_ms) VALUES (?, ?, ?, ?) + ON CONFLICT(peer_id) DO UPDATE SET bytes_per_sec = excluded.bytes_per_sec, + measured_ms = excluded.measured_ms, probed_ms = excluded.probed_ms`, + l.PeerID, l.BytesPerSec, l.MeasuredMs, l.ProbedMs) + if err != nil { + return fmt.Errorf("save peer link %s: %w", l.PeerID, err) + } + return nil +} From d9e7933104caa868035da97478fe349234d293ce Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:07:29 +0200 Subject: [PATCH 05/31] sync: an unmeasured link is speed-tested again soon A speed test that failed - a peer not yet on a build that answers one - counted as done for six hours, so every link stayed unmeasured and the order fell back to address guesses all that time. A link never measured is now tested again ten minutes after a failed test. Co-Authored-By: Claude Opus 5.5 --- internal/p2p/syncengine/linkspeed.go | 14 ++++++- internal/p2p/syncengine/linkspeed_test.go | 47 +++++++++++++++++++++++ 2 files changed, 60 insertions(+), 1 deletion(-) diff --git a/internal/p2p/syncengine/linkspeed.go b/internal/p2p/syncengine/linkspeed.go index b19543b1..179e2eb9 100644 --- a/internal/p2p/syncengine/linkspeed.go +++ b/internal/p2p/syncengine/linkspeed.go @@ -44,6 +44,8 @@ const ( // measured it since; probeBytes how much a test moves. probeEvery = 6 * time.Hour probeBytes = 2 << 20 + // probeRetry is how soon a speed test that failed is tried again. + probeRetry = 10 * time.Minute ) // prior rates, by address kind, used until a link is measured. @@ -177,7 +179,13 @@ func (e *Engine) ProbeLinkIfDue(ctx context.Context, peer Peer) bool { e.loadLinksLocked() l := e.links[peer.ID] now := time.Now() - due := now.Sub(time.UnixMilli(l.MeasuredMs)) > probeEvery && now.Sub(time.UnixMilli(l.ProbedMs)) > probeEvery + // A link never measured is tried again soon after a test that failed; + // one measured before only when that has gone stale. + retryAfter := probeEvery + if l.BytesPerSec == 0 { + retryAfter = probeRetry + } + due := now.Sub(time.UnixMilli(l.MeasuredMs)) > probeEvery && now.Sub(time.UnixMilli(l.ProbedMs)) > retryAfter if due { l.PeerID = peer.ID l.ProbedMs = now.UnixMilli() @@ -194,6 +202,10 @@ func (e *Engine) ProbeLinkIfDue(ctx context.Context, peer Peer) bool { defer cancel() n, d, err := prober.ProbeSpeed(ctx, peer, probeBytes) if err != nil { + // A link never measured is tried again after probeRetry (see due + // above): the peer may simply not have been ready — or not yet on a + // build that answers a speed test, which once left every link + // unmeasured for hours. e.Log("info", fmt.Sprintf("speed test with %s did not finish: %v", peer.Name, err)) return true } diff --git a/internal/p2p/syncengine/linkspeed_test.go b/internal/p2p/syncengine/linkspeed_test.go index 3e8544dc..c87657ab 100644 --- a/internal/p2p/syncengine/linkspeed_test.go +++ b/internal/p2p/syncengine/linkspeed_test.go @@ -2,6 +2,7 @@ package syncengine import ( "context" + "errors" "testing" "time" ) @@ -97,6 +98,52 @@ func TestWaitForClose(t *testing.T) { } } +type probeTransport struct { + *fakeTransport + fail bool + calls int +} + +func (p *probeTransport) ProbeSpeed(ctx context.Context, peer Peer, n int) (int64, time.Duration, error) { + p.calls++ + if p.fail { + return 0, 0, errors.New("peer returned 404") + } + return int64(n), time.Second, nil +} + +// A speed test that fails — a peer not yet updated to answer one — is tried +// again after minutes, not hours; a measured link is not tested again soon. +func TestProbeLink_RetriesAFailureSoon(t *testing.T) { + env := setupEngine(t) + pt := &probeTransport{fakeTransport: env.transport, fail: true} + env.engine.Transport = pt + p := Peer{ID: "p", Name: "P", Address: "192.168.1.9"} + + if !env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Fatal("an unmeasured link was not tested") + } + if env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Fatal("tested again straight after a failure") + } + // Eleven minutes later. + env.engine.linkMu.Lock() + l := env.engine.links[p.ID] + l.ProbedMs = time.Now().Add(-11 * time.Minute).UnixMilli() + env.engine.links[p.ID] = l + env.engine.linkMu.Unlock() + pt.fail = false + if !env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Fatal("a failed test was not tried again after ten minutes") + } + if env.engine.Link(p).BytesPerSec == 0 { + t.Fatal("the successful test was not recorded") + } + if env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Error("a freshly measured link was tested again") + } +} + func ids(ps []Peer) []string { out := make([]string, len(ps)) for i, p := range ps { From 02f0ceb99f3c4884f3545aa895b8ae67142ed6c1 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:31:09 +0200 Subject: [PATCH 06/31] sync: a long pull measures its link every 30 seconds, not only at its end A pull of a large save over a slow link takes an hour, and the link's speed was recorded once, when it ended: all that time the device list said "not measured" and the sync order fell back to address guesses. The link is now measured over each 30-second window while the pull runs, and the rest when it ends. Co-Authored-By: Claude Opus 5.5 --- internal/p2p/syncengine/engine.go | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index 230f3d65..8752e573 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -1610,13 +1610,24 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s tracker := newProgressTracker(totalBytes) throttle := e.throttleFor(peer.Wan()) - // What this transfer says about the connection (linkspeed.go), measured - // however it ends: a pull cut off half-way still moved what it moved. - pullStarted := time.Now() - defer func() { + // What this transfer says about the connection (linkspeed.go): every + // measureWindow while it runs — a pull of a large save takes an hour over a + // slow link, and one measurement at its end left the link unmeasured all + // that time — and what is left when it ends, however it ends: a pull cut + // off half-way still moved what it moved. + const measureWindow = 30 * time.Second + var measureMu sync.Mutex + measuredAt, measuredBytes := time.Now(), int64(0) + measure := func(final bool) { moved, _, _ := tracker.stats() - e.noteLinkRate(peer, moved, time.Since(pullStarted)) - }() + measureMu.Lock() + defer measureMu.Unlock() + if d := time.Since(measuredAt); final || d >= measureWindow { + e.noteLinkRate(peer, moved-measuredBytes, d) + measuredAt, measuredBytes = time.Now(), moved + } + } + defer measure(true) // Progress reporter shared by the per-file loop and the block-group // loop inside each file. Without in-file reporting, a single large @@ -1633,6 +1644,7 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s } lastReport = time.Now() reportMu.Unlock() + measure(false) bytesPulled, speed, pct := tracker.stats() ev := ProgressEvent{PeerName: peer.Name, BytesTransferred: bytesPulled, TotalBytes: totalBytes, SpeedBytesPerSec: speed, Percentage: pct} From f2189abfea0aad70c5fc90a80b16dc577292c8f2 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Fri, 2 Oct 2026 14:50:43 +0200 Subject: [PATCH 07/31] sync: a fast link's speed test is kept, not discarded as too short Over a home network the 2 MB speed test is over in a fraction of a second, and the rule that ignores transfers under half a second - meant for syncs, whose small transfers say more about round trips than about the link - threw it away: the fastest link stayed unmeasured, and with nothing measured it sorted behind the measured slow ones. A test that quick is now run again at 8 MB, and a test counts however short it is. Co-Authored-By: Claude Opus 5.5 --- internal/p2p/syncengine/linkspeed.go | 16 ++++++++++++++-- internal/p2p/syncengine/linkspeed_test.go | 22 ++++++++++++++++++++++ 2 files changed, 36 insertions(+), 2 deletions(-) diff --git a/internal/p2p/syncengine/linkspeed.go b/internal/p2p/syncengine/linkspeed.go index 179e2eb9..21af61f7 100644 --- a/internal/p2p/syncengine/linkspeed.go +++ b/internal/p2p/syncengine/linkspeed.go @@ -125,7 +125,12 @@ func (e *Engine) expectedRate(peer Peer) float64 { // noteLinkRate records a transfer with a peer: bytes in d. func (e *Engine) noteLinkRate(peer Peer, bytes int64, d time.Duration) { - if bytes < minMeasureBytes || d < minMeasureTime { + e.recordLinkRate(peer, bytes, d, minMeasureTime) +} + +// recordLinkRate records bytes in d, when d is at least minTime. +func (e *Engine) recordLinkRate(peer Peer, bytes int64, d, minTime time.Duration) { + if bytes < minMeasureBytes || d < minTime || d <= 0 { return } rate := float64(bytes) / d.Seconds() @@ -201,6 +206,11 @@ func (e *Engine) ProbeLinkIfDue(ctx context.Context, peer Peer) bool { ctx, cancel := context.WithTimeout(ctx, 60*time.Second) defer cancel() n, d, err := prober.ProbeSpeed(ctx, peer, probeBytes) + if err == nil && d < minMeasureTime { + // Over a home network the test is over before it measures much: + // once more, larger, for a figure worth having. + n, d, err = prober.ProbeSpeed(ctx, peer, 4*probeBytes) + } if err != nil { // A link never measured is tried again after probeRetry (see due // above): the peer may simply not have been ready — or not yet on a @@ -209,7 +219,9 @@ func (e *Engine) ProbeLinkIfDue(ctx context.Context, peer Peer) bool { e.Log("info", fmt.Sprintf("speed test with %s did not finish: %v", peer.Name, err)) return true } - e.noteLinkRate(peer, n, d) + // A test is a measurement however short: it moved nothing but the test, + // so no round trips of a sync's own are in it. + e.recordLinkRate(peer, n, d, time.Millisecond) e.Log("info", fmt.Sprintf("connection to %s: %s/s (%s)", peer.Name, humanBytes(int64(e.Link(peer).BytesPerSec)), linkKind(peer))) return true } diff --git a/internal/p2p/syncengine/linkspeed_test.go b/internal/p2p/syncengine/linkspeed_test.go index c87657ab..968dcc87 100644 --- a/internal/p2p/syncengine/linkspeed_test.go +++ b/internal/p2p/syncengine/linkspeed_test.go @@ -102,6 +102,8 @@ type probeTransport struct { *fakeTransport fail bool calls int + // perMB is how long each MB takes; a second when unset. + perMB time.Duration } func (p *probeTransport) ProbeSpeed(ctx context.Context, peer Peer, n int) (int64, time.Duration, error) { @@ -109,6 +111,9 @@ func (p *probeTransport) ProbeSpeed(ctx context.Context, peer Peer, n int) (int6 if p.fail { return 0, 0, errors.New("peer returned 404") } + if p.perMB > 0 { + return int64(n), time.Duration(n>>20) * p.perMB, nil + } return int64(n), time.Second, nil } @@ -144,6 +149,23 @@ func TestProbeLink_RetriesAFailureSoon(t *testing.T) { } } +// A home network finishes the test before it has measured much: it is run +// again larger, and the result is kept — not discarded as too short, which +// left the fastest link unmeasured and sorted behind the slow ones. +func TestProbeLink_AFastLinkIsMeasured(t *testing.T) { + env := setupEngine(t) + pt := &probeTransport{fakeTransport: env.transport, perMB: 20 * time.Millisecond} + env.engine.Transport = pt + p := Peer{ID: "lan", Name: "LAN", Address: "192.168.178.20"} + env.engine.ProbeLinkIfDue(context.Background(), p) + if got := env.engine.Link(p).BytesPerSec; got < 20<<20 { + t.Errorf("a 50 MB/s link measured as %.0f B/s", got) + } + if pt.calls != 2 { + t.Errorf("%d test(s), want the short one and a larger one", pt.calls) + } +} + func ids(ps []Peer) []string { out := make([]string, len(ps)) for i, p := range ps { From 0acaac38037128667959e9a8b76cdb96d6561a79 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:00:22 +0200 Subject: [PATCH 08/31] delta: hash cache and manifest walk that scale to saves of many files On a save of ~240k small files (Project Zomboid map chunks) a repeat manifest build took 13-19s in 2.4.0; now ~1.3s, with an identical hash. - Walk with filepath.WalkDir instead of filepath.Walk. Walk's extra Lstat per entry opens every file on Windows and dominated a warm build. - Reuse block read buffers (sync.Pool) instead of allocating 64KB+ per file: a cold build allocated ~15GB of garbage, now ~0.6GB. - Sort with sort.Slice when evicting from the hash cache. The insertion sort was quadratic in the entry count (map order is random). - Raise the cache budget floor to 256MB and scale it at 1MB per game, ceiling 512MB. 64MB was filled by one large save on its own, so the cache evicted and re-read that save on every pass. A budget, not an allocation: small libraries never come near it. - Share the whole-file hash string with the block hash for single-block files (most save files). - FileEntryForInfo: a cache lookup with the FileInfo a walk already has. --- internal/delta/hashcache.go | 35 +++++++++++++++++++++++++---------- internal/delta/manifest.go | 28 +++++++++++++++++++++++++--- 2 files changed, 50 insertions(+), 13 deletions(-) diff --git a/internal/delta/hashcache.go b/internal/delta/hashcache.go index 5ee3b611..61adcda8 100644 --- a/internal/delta/hashcache.go +++ b/internal/delta/hashcache.go @@ -4,6 +4,7 @@ import ( "os" "path/filepath" "runtime" + "sort" "strings" "sync" "time" @@ -64,11 +65,19 @@ type cacheEntry struct { // the old behaviour. const ( // cacheMinBytes is the floor, and the value a small library gets. - cacheMinBytes = 64 << 20 + // + // A budget, not an allocation: the cache only ever holds entries for + // files that exist, so a small library never comes near it. It has to + // cover the rare library with one enormous save, though — a game that + // writes every map chunk as its own file (Project Zomboid: ~240k files, + // ~53MB by entryCost) filled a 64MB budget on its own, and the cache then + // evicted and re-read that save on every pass, which is the churn it + // exists to prevent. + cacheMinBytes = 256 << 20 // cacheMaxCeiling is the most this will ever hold, however many games are // tracked. A cache is meant to replace repeated reading, not to become // the memory problem it was added to fix. - cacheMaxCeiling = 384 << 20 + cacheMaxCeiling = 512 << 20 // cachePerGameBytes is how much the budget grows per tracked game. // // From measurement rather than taste: a game's manifest costs roughly @@ -77,7 +86,10 @@ const ( // library and is not enough for someone tracking 350 games — the cache // sits at its cap and evicts entries it is about to want again, which // quietly reinstates the repeated reading it exists to remove. - cachePerGameBytes = 512 << 10 + // + // Doubled alongside the floor so a large library still gets more than a + // small one (350 games -> 350MB). + cachePerGameBytes = 1 << 20 ) // cacheMaxBytes is the current budget. Not a constant: it scales with how many @@ -193,6 +205,12 @@ func FileEntryFor(path string) (FileEntry, error) { return cachedHashFile(path, info) } +// FileEntryForInfo is FileEntryFor with the FileInfo a directory walk already +// has, so a cache hit costs no syscall at all. +func FileEntryForInfo(path string, info os.FileInfo) (FileEntry, error) { + return cachedHashFile(path, info) +} + func cachedHashFile(path string, info os.FileInfo) (FileEntry, error) { key := cacheKeyFor(path) stamp := cacheStamp{size: info.Size(), mtimeNs: info.ModTime().UnixNano()} @@ -302,13 +320,10 @@ func evictLocked() { for k, v := range hashCache.entries { all = append(all, aged{k, v.usedAt}) } - // Partial selection would be faster, but this runs only on overflow and - // clarity is worth more here than the microseconds. - for i := 1; i < len(all); i++ { - for j := i; j > 0 && all[j].at.Before(all[j-1].at); j-- { - all[j], all[j-1] = all[j-1], all[j] - } - } + // sort.Slice, not an insertion sort: map iteration order is random, so + // insertion sort was quadratic in the entry count — ~10^10 comparisons for + // a quarter-million-file save, on every overflow. + sort.Slice(all, func(i, j int) bool { return all[i].at.Before(all[j].at) }) target := cacheMaxBytes * 3 / 4 for _, a := range all { if hashCache.bytes <= target { diff --git a/internal/delta/manifest.go b/internal/delta/manifest.go index f70016e5..d008ad76 100644 --- a/internal/delta/manifest.go +++ b/internal/delta/manifest.go @@ -10,6 +10,7 @@ import ( "encoding/json" "fmt" "io" + "io/fs" "os" "path/filepath" "sort" @@ -217,9 +218,17 @@ func HashFile(path string) (FileEntry, error) { } } + wholeHash := hex.EncodeToString(whole.Sum(nil)) + // A single-block file's block hash is the whole-file hash; share the one + // string. Most save files are single-block, and on a save of a quarter + // million of them the duplicate hex strings cost ~20MB per manifest. + if len(blocks) == 1 { + blocks[0].Hash = wholeHash + } + return FileEntry{ Size: info.Size(), - Hash: hex.EncodeToString(whole.Sum(nil)), + Hash: wholeHash, Blocks: blocks, BlockSize: blockSize, MtimeMs: Milli(info.ModTime().UnixMilli()), @@ -263,7 +272,20 @@ func buildManifest(root string) (Manifest, error) { } var dirs []string - err = filepath.Walk(root, func(path string, walkInfo os.FileInfo, walkErr error) error { + // WalkDir, not Walk: Walk issues an extra Lstat per entry, which on + // Windows opens every file and dominated the cost of a warm build of a + // large save (13-19s vs ~1.2s for 240k files). WalkDir takes size and + // mtime from the directory listing itself. + err = filepath.WalkDir(root, func(path string, d fs.DirEntry, walkErr error) error { + var walkInfo os.FileInfo + if walkErr == nil && path != root { + // Reparse points are rejected below from the listing's type bits, + // before Info is asked for anything. + if d.Type()&(os.ModeSymlink|os.ModeIrregular) != 0 { + return nil + } + walkInfo, walkErr = d.Info() + } if walkErr != nil { // The save folder itself could not be read — its listing refused, // or the folder gone since it was looked at. That is not an empty @@ -282,7 +304,7 @@ func buildManifest(root string) (Manifest, error) { // skipped instead of failing the whole manifest — one // unreadable directory must not abort every sync of the game. if os.IsPermission(walkErr) { - if walkInfo != nil && walkInfo.IsDir() { + if d != nil && d.IsDir() { return filepath.SkipDir } return nil From 93354a43799787125b3051be3881c598158bad49 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:00:22 +0200 Subject: [PATCH 09/31] snapshot: copy unchanged files from the previous snapshot instead of compressing them again Every snapshot stays a complete zip that restores on its own. What changes is how it is written: a file whose content (SHA-256, as already recorded per snapshot and held by the hash cache) matches the previous snapshot of the same branch is copied in as its already-compressed entry (zip.Writer.Copy). Only new and changed files are compressed. On a 204k-file save: a full snapshot 80.5s, a follow-up 8.9s, same size. The archive walk also moves to filepath.WalkDir for the same reason as the manifest walk. --- internal/snapshot/reuse.go | 124 ++++++++++++++++++++++++++ internal/snapshot/reuse_bench_test.go | 61 +++++++++++++ internal/snapshot/reuse_test.go | 69 ++++++++++++++ internal/snapshot/snapshot.go | 9 +- internal/snapshot/zip.go | 21 ++++- internal/snapshot/zip_roots.go | 28 ++++-- 6 files changed, 299 insertions(+), 13 deletions(-) create mode 100644 internal/snapshot/reuse.go create mode 100644 internal/snapshot/reuse_bench_test.go create mode 100644 internal/snapshot/reuse_test.go diff --git a/internal/snapshot/reuse.go b/internal/snapshot/reuse.go new file mode 100644 index 00000000..90da079e --- /dev/null +++ b/internal/snapshot/reuse.go @@ -0,0 +1,124 @@ +package snapshot + +import ( + "archive/zip" + "os" + "sync/atomic" + + "github.com/opensave/opensave/internal/delta" +) + +// Snapshots that re-pack only what changed. +// +// Every snapshot is a complete zip, restorable on its own — that stays. What +// changes is how it is written: a file whose content is exactly what the +// previous snapshot of the same branch holds is copied into the new archive +// as its already-compressed entry (zip.Writer.Copy), not read and deflated +// again. Only new and changed files are compressed. +// +// For a save of a quarter-million small files (Project Zomboid map chunks) +// where a play session touches a few hundred, that turns a snapshot from a +// minute of reading and deflating 0.76GB into writing mostly copied bytes. +// +// Content is identified by SHA-256 — the hash each snapshot already records +// per file (store.SnapshotFiles) and the one the manifest's hash cache holds, +// so an unchanged file is recognised without being read. Size is checked as +// well, cheaply, before the hash is asked for. +type reuseSource struct { + zr *zip.ReadCloser + files map[string]*zip.File // entry name -> previous entry + hashes map[string]string // entry name -> sha256 recorded for it + reused int // entries copied, for tests +} + +// lastReused is how many entries the most recent snapshot copied from its +// predecessor. Tests read it; nothing else does. +var lastReused atomic.Int64 + +// openReuseSource prepares the newest snapshot of a branch for reuse. Any +// problem — no snapshot, no recorded file list (an older snapshot, or a +// mirror), a zip that does not open — means no reuse, never a failed snapshot. +func (m *Manager) openReuseSource(gameID, branch string) *reuseSource { + prev, err := m.LatestSnapshot(gameID, branch) + if err != nil || prev.ZipPath == "" { + return nil + } + recorded, err := m.Store.SnapshotFiles(prev.ID) + if err != nil || len(recorded) == 0 { + return nil + } + zr, err := zip.OpenReader(prev.ZipPath) + if err != nil { + return nil + } + r := &reuseSource{ + zr: zr, + files: make(map[string]*zip.File, len(zr.File)), + hashes: make(map[string]string, len(recorded)), + } + for _, f := range zr.File { + r.files[f.Name] = f + } + for _, c := range recorded { + name := c.Path + if c.Root != "" { + name = RootPrefix + c.Root + "/" + c.Path + } + r.hashes[name] = c.Hash + } + return r +} + +func (r *reuseSource) Close() { + if r == nil { + lastReused.Store(0) + return + } + lastReused.Store(int64(r.reused)) + if r.zr != nil { + r.zr.Close() + } +} + +// addFileEntry adds filePath under entryName, copying the previous +// snapshot's entry when the content is the same, and archiving it afresh +// otherwise. Returns the file's sha256 either way. A nil receiver archives +// afresh, as every snapshot did before. +func (r *reuseSource) addFileEntry(w *zip.Writer, filePath, entryName string, info os.FileInfo) (string, error) { + if r != nil { + if hash, ok := r.reusable(filePath, entryName, info); ok { + if err := w.Copy(r.files[entryName]); err != nil { + return "", err + } + r.reused++ + return hash, nil + } + } + return addFileEntry(w, filePath, entryName) +} + +// reusable reports whether the previous snapshot holds exactly this file's +// current content under the same name. +func (r *reuseSource) reusable(filePath, entryName string, info os.FileInfo) (string, bool) { + prevHash, ok := r.hashes[entryName] + prev := r.files[entryName] + if !ok || prev == nil || prev.Mode().IsDir() { + return "", false + } + if info == nil { + var err error + if info, err = os.Stat(filePath); err != nil { + return "", false + } + } + if uint64(info.Size()) != prev.UncompressedSize64 { + return "", false + } + // From the hash cache when the file is unchanged since it was last + // hashed, which after the watcher's manifest build it nearly always is. + entry, err := delta.FileEntryForInfo(filePath, info) + if err != nil || entry.Hash != prevHash { + return "", false + } + return prevHash, true +} diff --git a/internal/snapshot/reuse_bench_test.go b/internal/snapshot/reuse_bench_test.go new file mode 100644 index 00000000..641a0335 --- /dev/null +++ b/internal/snapshot/reuse_bench_test.go @@ -0,0 +1,61 @@ +package snapshot + +import ( + "archive/zip" + "os" + "path/filepath" + "testing" + "time" + + "github.com/opensave/opensave/internal/delta" +) + +// TestBench_SnapshotReuse times a full snapshot of a real save folder and a +// second one that reuses it. Opt-in: OPENSAVE_BENCH_SAVE=. +// Reads the folder only; archives go to a temp dir. +func TestBench_SnapshotReuse(t *testing.T) { + src := os.Getenv("OPENSAVE_BENCH_SAVE") + if src == "" { + t.Skip("set OPENSAVE_BENCH_SAVE to run") + } + dir := t.TempDir() + + first := filepath.Join(dir, "first.zip") + start := time.Now() + skipped, captured, err := zipRootsCapturing(src, nil, first, nil) + if err != nil { + t.Fatal(err) + } + full := time.Since(start) + t.Logf("full snapshot skipped %d unreadable files", len(skipped)) + + // As in the app: the watcher's manifest build has hashed the save just + // before a snapshot is taken, so the cache is warm. + if _, err := delta.BuildManifest(src); err != nil { + t.Fatal(err) + } + + zr, err := zip.OpenReader(first) + if err != nil { + t.Fatal(err) + } + r := &reuseSource{zr: zr, files: map[string]*zip.File{}, hashes: map[string]string{}} + for _, f := range zr.File { + r.files[f.Name] = f + } + for _, c := range captured { + r.hashes[c.Path] = c.Hash + } + second := filepath.Join(dir, "second.zip") + start = time.Now() + if _, _, err := zipRootsCapturing(src, nil, second, r); err != nil { + t.Fatal(err) + } + incr := time.Since(start) + r.Close() + + fi1, _ := os.Stat(first) + fi2, _ := os.Stat(second) + t.Logf("%d files: full snapshot %s (%d MB), reusing snapshot %s (%d MB), %d entries copied", + len(captured), full.Round(time.Millisecond), fi1.Size()>>20, incr.Round(time.Millisecond), fi2.Size()>>20, lastReused.Load()) +} diff --git a/internal/snapshot/reuse_test.go b/internal/snapshot/reuse_test.go new file mode 100644 index 00000000..104a8efd --- /dev/null +++ b/internal/snapshot/reuse_test.go @@ -0,0 +1,69 @@ +package snapshot + +import ( + "fmt" + "testing" +) + +// A snapshot copies the files that did not change from the previous one, and +// still restores to exactly the save it was taken of. +func TestSnapshotReusesUnchangedFiles(t *testing.T) { + env := setup(t) + for i := 0; i < 50; i++ { + write(t, env.saveDir, fmt.Sprintf("map/chunk_%02d.bin", i), fmt.Sprintf("chunk %d %s", i, "................................................")) + } + first, err := env.mgr.Create("game1", "first", false) + if err != nil { + t.Fatal(err) + } + if got := lastReused.Load(); got != 0 { + t.Errorf("first snapshot reused %d entries from nothing", got) + } + + // A session changes two chunks and adds one. + write(t, env.saveDir, "map/chunk_03.bin", "changed three") + write(t, env.saveDir, "map/chunk_40.bin", "changed forty") + write(t, env.saveDir, "map/chunk_99.bin", "new") + second, err := env.mgr.Create("game1", "second", false) + if err != nil { + t.Fatal(err) + } + if got := lastReused.Load(); got != 48 { + t.Errorf("second snapshot copied %d unchanged entries, want 48", got) + } + + // Each snapshot restores its own state, independently of the other. + write(t, env.saveDir, "map/chunk_03.bin", "ruined") + if _, err := env.mgr.Restore("game1", second.ID); err != nil { + t.Fatal(err) + } + for name, want := range map[string]string{ + "map/chunk_03.bin": "changed three", + "map/chunk_40.bin": "changed forty", + "map/chunk_99.bin": "new", + "map/chunk_07.bin": "chunk 7 ................................................", + } { + if got := read(env.saveDir, name); got != want { + t.Errorf("after restoring the second snapshot, %s = %q, want %q", name, got, want) + } + } + if _, err := env.mgr.Restore("game1", first.ID); err != nil { + t.Fatal(err) + } + if got := read(env.saveDir, "map/chunk_03.bin"); got != "chunk 3 ................................................" { + t.Errorf("after restoring the first snapshot, chunk_03 = %q", got) + } + if got := read(env.saveDir, "map/chunk_99.bin"); got != "" { + t.Errorf("restoring the first snapshot left the later file: %q", got) + } + + // And the recorded contents of the second snapshot are right for the + // copied entries too. + files, err := env.store.SnapshotFiles(second.ID) + if err != nil { + t.Fatal(err) + } + if len(files) != 51 { + t.Errorf("second snapshot records %d files, want 51", len(files)) + } +} diff --git a/internal/snapshot/snapshot.go b/internal/snapshot/snapshot.go index 77d0cf6f..f483bde4 100644 --- a/internal/snapshot/snapshot.go +++ b/internal/snapshot/snapshot.go @@ -236,7 +236,14 @@ func (m *Manager) createOnBranchFrom(gameID, branch, from, comment string, isSys } else { src = from } - skipped, captured, err := ZipRootsCapturing(src, extraRoots, stagingPath) + // The game's own save, onto its branch: unchanged files are copied from + // the branch's previous snapshot rather than compressed again (reuse.go). + var reuse *reuseSource + if from == "" { + reuse = m.openReuseSource(gameID, branch) + } + skipped, captured, err := zipRootsCapturing(src, extraRoots, stagingPath, reuse) + reuse.Close() if err != nil { os.Remove(stagingPath) return store.Snapshot{}, fmt.Errorf("zip save data: %w", err) diff --git a/internal/snapshot/zip.go b/internal/snapshot/zip.go index 1ea2920c..803c60a8 100644 --- a/internal/snapshot/zip.go +++ b/internal/snapshot/zip.go @@ -8,6 +8,7 @@ import ( "encoding/hex" "fmt" "io" + "io/fs" "os" "path/filepath" "strings" @@ -38,6 +39,10 @@ func ZipPath(sourcePath, outPath string) (skipped []string, err error) { // archived so the caller can record what the snapshot holds. The hashes come // free: the bytes pass through a hasher on their way into the archive. func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured []store.CapturedFile, err error) { + return zipPathCapturing(sourcePath, outPath, nil) +} + +func zipPathCapturing(sourcePath, outPath string, reuse *reuseSource) (skipped []string, captured []store.CapturedFile, err error) { info, err := os.Stat(sourcePath) if err != nil { return nil, nil, fmt.Errorf("source path does not exist: %w", err) @@ -54,7 +59,7 @@ func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured [] if !info.IsDir() { name := filepath.Base(sourcePath) - hash, addErr := addFileEntry(w, sourcePath, name) + hash, addErr := reuse.addFileEntry(w, sourcePath, name, info) if addErr != nil { return nil, nil, addErr } @@ -62,7 +67,10 @@ func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured [] } archived := 0 - walkErr := filepath.Walk(sourcePath, func(path string, walkInfo os.FileInfo, walkErr error) error { + // WalkDir, not Walk: Walk's extra Lstat per entry opens every file on + // Windows, which on a save of a quarter-million files cost more than the + // archiving itself once unchanged files are copied (reuse.go). + walkErr := filepath.WalkDir(sourcePath, func(path string, d fs.DirEntry, walkErr error) error { if walkErr != nil { // Unreadable subtree — record it and move on. if path != sourcePath { @@ -80,14 +88,19 @@ func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured [] } rel = filepath.ToSlash(rel) - if walkInfo.IsDir() { + if d.IsDir() { // Explicit directory entries keep empty dirs restorable. if _, err := w.CreateHeader(&zip.FileHeader{Name: rel + "/", Method: zip.Store}); err != nil { return err } return nil } - hash, addErr := addFileEntry(w, path, rel) + info, infoErr := d.Info() + if infoErr != nil { + skipped = append(skipped, path) + return nil + } + hash, addErr := reuse.addFileEntry(w, path, rel, info) if addErr != nil { skipped = append(skipped, path) return nil diff --git a/internal/snapshot/zip_roots.go b/internal/snapshot/zip_roots.go index 8f70ff2f..c3618b6c 100644 --- a/internal/snapshot/zip_roots.go +++ b/internal/snapshot/zip_roots.go @@ -3,6 +3,7 @@ package snapshot import ( "archive/zip" "fmt" + "io/fs" "os" "path/filepath" "sort" @@ -44,8 +45,14 @@ func ZipRoots(primary string, extra map[string]string, outPath string) (skipped // instead of inferred from a single whole-save hash that nothing reliably // kept current. func ZipRootsCapturing(primary string, extra map[string]string, outPath string) (skipped []string, captured []store.CapturedFile, err error) { + return zipRootsCapturing(primary, extra, outPath, nil) +} + +// zipRootsCapturing is ZipRootsCapturing, copying unchanged files straight +// out of reuse (the previous snapshot) when it is given. See reuse.go. +func zipRootsCapturing(primary string, extra map[string]string, outPath string, reuse *reuseSource) (skipped []string, captured []store.CapturedFile, err error) { if len(extra) == 0 { - return ZipPathCapturing(primary, outPath) + return zipPathCapturing(primary, outPath, reuse) } // One archive, written in a single pass. @@ -69,7 +76,7 @@ func ZipRootsCapturing(primary string, extra map[string]string, outPath string) defer f.Close() w := newSnapshotWriter(f) - primarySkipped, primaryFiles, err := archiveInto(w, primary, "", "") + primarySkipped, primaryFiles, err := archiveInto(w, primary, "", "", reuse) skipped = append(skipped, primarySkipped...) captured = append(captured, primaryFiles...) if err != nil { @@ -82,7 +89,7 @@ func ZipRootsCapturing(primary string, extra map[string]string, outPath string) if strings.TrimSpace(path) == "" { continue } - sub, subFiles, subErr := archiveInto(w, path, RootPrefix+name+"/", name) + sub, subFiles, subErr := archiveInto(w, path, RootPrefix+name+"/", name, reuse) skipped = append(skipped, sub...) captured = append(captured, subFiles...) if subErr != nil { @@ -94,14 +101,14 @@ func ZipRootsCapturing(primary string, extra map[string]string, outPath string) } // archiveInto writes one directory tree into an open zip under prefix. -func archiveInto(w *zip.Writer, sourcePath, prefix, root string) (skipped []string, captured []store.CapturedFile, err error) { +func archiveInto(w *zip.Writer, sourcePath, prefix, root string, reuse *reuseSource) (skipped []string, captured []store.CapturedFile, err error) { info, err := os.Stat(sourcePath) if err != nil { return nil, nil, err } if !info.IsDir() { name := filepath.Base(sourcePath) - hash, addErr := addFileEntry(w, sourcePath, prefix+name) + hash, addErr := reuse.addFileEntry(w, sourcePath, prefix+name, info) if addErr != nil { return nil, nil, addErr } @@ -117,7 +124,7 @@ func archiveInto(w *zip.Writer, sourcePath, prefix, root string) (skipped []stri } } - walkErr := filepath.Walk(sourcePath, func(path string, walkInfo os.FileInfo, walkErr error) error { + walkErr := filepath.WalkDir(sourcePath, func(path string, d fs.DirEntry, walkErr error) error { if walkErr != nil { if path != sourcePath { skipped = append(skipped, path) @@ -133,11 +140,16 @@ func archiveInto(w *zip.Writer, sourcePath, prefix, root string) (skipped []stri return err } rel = filepath.ToSlash(rel) - if walkInfo.IsDir() { + if d.IsDir() { _, err := w.CreateHeader(&zip.FileHeader{Name: prefix + rel + "/", Method: zip.Store}) return err } - hash, addErr := addFileEntry(w, path, prefix+rel) + info, infoErr := d.Info() + if infoErr != nil { + skipped = append(skipped, path) + return nil + } + hash, addErr := reuse.addFileEntry(w, path, prefix+rel, info) if addErr != nil { skipped = append(skipped, path) return nil From 829f65a910ea51972cf7b5357620626330c0e1dd Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Sun, 4 Oct 2026 01:49:32 +0200 Subject: [PATCH 10/31] Find a save folder on another drive letter; a drive this PC lacks is no warning - A missing save folder is looked for at the identical path on every other drive letter (a Steam library on D: on one PC and E: on another). Exactly one match is adopted for this device and logged; none or several are logged once. Searched at most once an hour. - A save folder on a drive this device does not have is shown as "Not on this device" (muted) instead of a warning, kept out of the warning headline and the notification bell. A folder gone from a drive that exists is still a warning. Co-Authored-By: Claude Opus 5.5 --- .../frontend/src/lib/gamestatus.js | 5 +- .../frontend/src/lib/gamestatus.test.js | 12 ++ .../frontend/src/lib/notifications.js | 3 +- .../frontend/src/lib/notifications.test.js | 5 + .../frontend/src/views/game/GameHeader.svelte | 21 ++- internal/api/server.go | 12 +- internal/daemon/daemon.go | 3 + internal/daemon/missing.go | 123 +++++++++++++++++- internal/daemon/missing_test.go | 40 ++++++ internal/daemon/relocate_test.go | 82 ++++++++++++ 10 files changed, 298 insertions(+), 8 deletions(-) create mode 100644 internal/daemon/missing_test.go create mode 100644 internal/daemon/relocate_test.go diff --git a/cmd/opensave-app/frontend/src/lib/gamestatus.js b/cmd/opensave-app/frontend/src/lib/gamestatus.js index 4d920c01..2be2f839 100644 --- a/cmd/opensave-app/frontend/src/lib/gamestatus.js +++ b/cmd/opensave-app/frontend/src/lib/gamestatus.js @@ -21,6 +21,9 @@ import { playLength } from './format.js'; export function gameStatus(game, { peers = {}, activity, conflicted = false, now = Date.now() } = {}) { // Before anything else: with its folder gone nothing else about it can // move — no snapshot, no sync, no decision taken. + // …unless the whole drive is absent here: the game's saves live on another + // device's D:\ or E:\, and there is nothing on this one to worry about. + if (game.savePathMissing && game.saveDriveMissing) return { state: 'elsewhere', label: 'Not on this device', tone: 'muted' }; if (game.savePathMissing) return { state: 'missing', label: 'Save folder missing', tone: 'warn' }; // Every save file deleted here at once, held back from the other devices // until someone says whether that was meant (lib/emptied.js). @@ -177,7 +180,7 @@ const stamp = (s) => (s ? Date.parse(s) : null); const snapshotsOf = (game) => Object.values(game.branches ?? {}).flatMap((b) => b.snapshots ?? []); // What needs looking at, in the order the summary above the library reads it. -const URGENCY = { missing: 0, emptied: 1, conflict: 2, error: 3, syncing: 4, restoring: 4, unsynced: 5, empty: 6, paused: 7 }; +const URGENCY = { missing: 0, emptied: 1, conflict: 2, error: 3, syncing: 4, restoring: 4, unsynced: 5, empty: 6, paused: 7, elsewhere: 10 }; // unlisted states (synced, local…) rank 9 /** * The orders the library can be put in. Each compares two rows, {game, diff --git a/cmd/opensave-app/frontend/src/lib/gamestatus.test.js b/cmd/opensave-app/frontend/src/lib/gamestatus.test.js index d144943c..810b197b 100644 --- a/cmd/opensave-app/frontend/src/lib/gamestatus.test.js +++ b/cmd/opensave-app/frontend/src/lib/gamestatus.test.js @@ -119,6 +119,18 @@ describe('a missing save folder', () => { expect(librarySummary(rows)).toEqual({ tone: 'warn', headline: "Hades's save folder is missing" }); expect(sortRows(rows, 'attention').map((r) => r.game.name)).toEqual(['Hades', 'Celeste']); }); + + it('on a drive this device does not have is not a warning, and stays out of the headline', () => { + const g = game({ id: 'a', name: 'ARK', savePathMissing: true, saveDriveMissing: true }); + const s = gameStatus(g); + expect(s).toMatchObject({ state: 'elsewhere', tone: 'muted', label: 'Not on this device' }); + const rows = [ + { game: g, status: s }, + { game: game({ id: 'c', name: 'Celeste', branches: snaps(minsAgo(5)) }), status: { state: 'synced' } } + ]; + expect(librarySummary(rows).tone).not.toBe('warn'); + expect(sortRows(rows, 'attention').map((r) => r.game.name)).toEqual(['Celeste', 'ARK']); + }); }); describe('SORTS', () => { diff --git a/cmd/opensave-app/frontend/src/lib/notifications.js b/cmd/opensave-app/frontend/src/lib/notifications.js index e8633fd5..ece29614 100644 --- a/cmd/opensave-app/frontend/src/lib/notifications.js +++ b/cmd/opensave-app/frontend/src/lib/notifications.js @@ -46,7 +46,8 @@ export function waitingOnYou({ games = {}, conflicts = {}, locationConflicts = [ out.push({ id: `offer:${o.gameId}:${o.snapshotId}`, tone: 'info', gameId: o.gameId, title: `A newer save for ${o.gameName ?? name(o.gameId)}`, detail: `From ${o.deviceName ?? 'another device'}, through the cloud`, go: { view: 'home' } }); } for (const g of Object.values(games)) { - if (g.savePathMissing) { + // A drive this device doesn't have is not waiting on anyone. + if (g.savePathMissing && !g.saveDriveMissing) { out.push({ id: `missing:${g.id}`, tone: 'warn', gameId: g.id, title: `${g.name}'s save folder is missing`, detail: 'Nothing is synced for it until it is back', go: { view: 'game', params: { gameId: g.id } } }); } } diff --git a/cmd/opensave-app/frontend/src/lib/notifications.test.js b/cmd/opensave-app/frontend/src/lib/notifications.test.js index a5828324..1e32a9f7 100644 --- a/cmd/opensave-app/frontend/src/lib/notifications.test.js +++ b/cmd/opensave-app/frontend/src/lib/notifications.test.js @@ -21,6 +21,11 @@ describe('notifications', () => { expect(w.find((x) => x.id.startsWith('damaged')).title).toBe("1 snapshot of Hades can't be restored"); }); + it('does not ask about a game whose drive this device does not have', () => { + const w = waitingOnYou({ games: { ark: { id: 'ark', name: 'ARK', savePathMissing: true, saveDriveMissing: true, branches: {} } } }); + expect(w).toEqual([]); + }); + it('turns activity into events, unread since the bell was last opened', () => { const now = Date.UTC(2026, 8, 25, 12); const items = [ diff --git a/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte b/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte index 19750020..8eb05a2e 100644 --- a/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte +++ b/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte @@ -17,6 +17,7 @@ import Pencil from 'lucide-svelte/icons/pencil'; import Star from 'lucide-svelte/icons/star'; import TriangleAlert from 'lucide-svelte/icons/triangle-alert'; + import Info from 'lucide-svelte/icons/info'; import { collections, isFavourite, toggleFavourite } from '../../lib/collections.js'; export let game; @@ -188,7 +189,18 @@
{/if} -{#if game.savePathMissing && !editPath} +{#if game.savePathMissing && game.saveDriveMissing && !editPath} + +
+ +
+ This game isn't on this device. + Its saves are on {game.savePath.slice(0, 2)}, which this PC doesn't have, so it's skipped here and keeps syncing + between your other devices. If it is installed here, point the game at its save folder with Edit. +
+
+{:else if game.savePathMissing && !editPath} @@ -288,6 +300,13 @@ .missing strong { color: var(--text); } + .missing.elsewhere { + border-color: var(--border); + background: transparent; + } + .missing.elsewhere :global(svg) { + color: var(--text-faint); + } .emptied-actions { display: flex; flex-wrap: wrap; diff --git a/internal/api/server.go b/internal/api/server.go index fea038ed..4a24555c 100644 --- a/internal/api/server.go +++ b/internal/api/server.go @@ -572,10 +572,14 @@ func (s *Server) gamePayload(g store.Game) map[string]any { // The save folder is not there — gone, moved, or on a drive not // plugged in. Nothing is watched or synced for it until it is back. "savePathMissing": daemon.SaveFolderMissing(g.SavePath), - "lastPlayedAt": lastPlayed, - "playingSince": playingSince, - "playtimeMs": play.PlaytimeMs, - "playSessions": play.Sessions, + // …and of those, the ones whose whole drive is absent here: the game + // lives on another device's D:\ or E:\, which is not a problem to + // warn about (daemon.SaveDriveMissing). + "saveDriveMissing": daemon.SaveFolderMissing(g.SavePath) && daemon.SaveDriveMissing(g.SavePath), + "lastPlayedAt": lastPlayed, + "playingSince": playingSince, + "playtimeMs": play.PlaytimeMs, + "playSessions": play.Sessions, // Whether the game is installed on this device: "found", "not-found", // or "" when there is no telling (daemon.InstallState). A game that is // not here is one nobody plays here, which is why it shows no play. diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 306ce6ff..5683cd88 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -62,6 +62,9 @@ type Daemon struct { // Games whose save folder is not there; see missing.go. missingMu sync.Mutex missing map[string]bool + // relocateChecked is when each missing save folder was last looked for on + // another drive (missing.go). + relocateChecked map[string]time.Time // Play sessions; see sessions.go. sessions sessionState diff --git a/internal/daemon/missing.go b/internal/daemon/missing.go index aa083447..b55c7b1f 100644 --- a/internal/daemon/missing.go +++ b/internal/daemon/missing.go @@ -6,6 +6,8 @@ import ( "io/fs" "os" "path/filepath" + "strings" + "time" "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/logging" @@ -34,10 +36,126 @@ func SaveFolderMissing(savePath string) bool { return errors.Is(err, fs.ErrNotExist) } +// SaveDriveMissing reports whether a save path lives on a drive this device +// does not have at all — a D:\ or E:\ Steam library that exists on another +// PC. That is where a game auto-tracked from a peer lands when the peer's +// library is on a drive letter this machine lacks, and it is not something +// to warn about: nothing was lost here, the game just isn't on this device. +// Only a volume that is plainly absent counts; an unreachable network share +// reports some other error and stays a warning. +func SaveDriveMissing(savePath string) bool { + vol := filepath.VolumeName(savePath) + if vol == "" { + return false + } + _, err := os.Stat(vol + string(filepath.Separator)) + return errors.Is(err, fs.ErrNotExist) +} + +// The same save folder under another drive letter. +// +// A save inside a game's install folder — a Steam library, most often — sits +// on whichever drive that library is on, and that differs between devices: D: +// on one, E: on another. A game tracked on one device and auto-tracked on +// another arrives with the first device's drive letter, and on the second the +// folder is simply "missing" though it is right there on another drive. +// +// So a missing folder is looked for at the identical path on every other drive +// letter. Exactly one match is adopted for this device, and said once. None, or +// more than one (which would be a guess), is said once too, and the search is +// not repeated more often than relocateRecheck: it touches every drive, and the +// minute-long watch reconcile asks about missing folders constantly. +const relocateRecheck = time.Hour + +// relocateCandidates lists the folders at savePath's exact path under the other +// drive letters that exist here. Windows drive letters only. +func relocateCandidates(savePath string) []string { + vol := filepath.VolumeName(savePath) + if len(vol) != 2 || vol[1] != ':' { + return nil + } + rest := savePath[2:] + own := strings.ToUpper(vol[:1]) + var found []string + for c := 'C'; c <= 'Z'; c++ { + letter := string(c) + if letter == own { + continue + } + candidate := letter + ":" + rest + if !SaveFolderMissing(candidate) { + found = append(found, candidate) + } + } + return found +} + +// relocateToOtherDrive switches a game whose save folder is missing to the same +// path on another drive, when there is exactly one. Reports whether it did. +func (d *Daemon) relocateToOtherDrive(gameID, name, savePath string) bool { + d.missingMu.Lock() + last, checked := d.relocateChecked[gameID] + if checked && time.Since(last) < relocateRecheck { + d.missingMu.Unlock() + return false + } + if d.relocateChecked == nil { + d.relocateChecked = map[string]time.Time{} + } + d.relocateChecked[gameID] = time.Now() + d.missingMu.Unlock() + + if name == "" { + name = gameID + } + found := relocateCandidates(savePath) + switch { + case len(found) == 0: + if !checked { + d.Log.Log("info", fmt.Sprintf("the save folder of %q is not on another drive here either (looked for %s on every drive letter)", + name, logging.Quote(savePath[min(2, len(savePath)):]))) + } + return false + case len(found) > 1: + if !checked { + d.Log.Log("warn", fmt.Sprintf("the save folder of %q is missing at %s, and the same path exists on %d other drives (%s) — not guessing which; set it with Edit", + name, logging.Quote(savePath), len(found), strings.Join(found, ", "))) + } + return false + } + + newPath, err := d.ValidateSavePath(found[0]) + if err != nil { + d.Log.Log("warn", fmt.Sprintf("the save folder of %q exists at %s, but it cannot be used: %v", name, logging.Quote(found[0]), err)) + return false + } + game, err := d.Store.GetGame(gameID) + if err != nil || game.SavePath != savePath { + return false // changed meanwhile; the next reconcile sees the new path + } + game.SavePath = newPath + if err := d.Store.UpdateGame(game); err != nil { + d.Log.Log("warn", fmt.Sprintf("could not switch %q to %s: %v", name, logging.Quote(newPath), err)) + return false + } + d.Log.Log("success", fmt.Sprintf("the save folder of %q was not at %s but is at %s on this device — using that", + name, logging.Quote(savePath), logging.Quote(newPath))) + d.missingMu.Lock() + delete(d.missing, gameID) + d.missingMu.Unlock() + if d.OnGameChanged != nil { + d.OnGameChanged(gameID) + } + return true +} + // noteMissing records whether a game's save folder is missing, and says so // once when that changes — not on every minute's check — so the log and the // screen follow the folder going and coming back. func (d *Daemon) noteMissing(gameID, name, savePath string, missing bool) { + if missing && d.relocateToOtherDrive(gameID, name, savePath) { + return // found under another drive letter and switched to it + } d.missingMu.Lock() was := d.missing[gameID] if missing { @@ -55,7 +173,10 @@ func (d *Daemon) noteMissing(gameID, name, savePath string, missing bool) { if name == "" { name = gameID } - if missing { + if missing && SaveDriveMissing(savePath) { + d.Log.Log("info", fmt.Sprintf("the save folder of %q is on %s, which this device doesn't have (%s) — the game is skipped here "+ + "and keeps syncing between the devices that have it", name, filepath.VolumeName(savePath), logging.Quote(savePath))) + } else if missing { d.Log.Log("warn", fmt.Sprintf("the save folder of %q is not there (%s) — it is not watched or synced until it is back; "+ "OpenSave will not create it, since an empty folder in its place would read as every file deleted", name, logging.Quote(savePath))) } else { diff --git a/internal/daemon/missing_test.go b/internal/daemon/missing_test.go new file mode 100644 index 00000000..0b5fdf6c --- /dev/null +++ b/internal/daemon/missing_test.go @@ -0,0 +1,40 @@ +package daemon + +import ( + "os" + "path/filepath" + "runtime" + "testing" +) + +// A save path on a drive letter this machine does not have is "not on this +// device", not "folder gone": the UI informs rather than warns about it. +func TestSaveDriveMissing(t *testing.T) { + existing := t.TempDir() + if SaveDriveMissing(filepath.Join(existing, "saves")) { + t.Error("a folder on an existing drive was reported as on a missing drive") + } + if SaveDriveMissing("relative/path") { + t.Error("a path without a volume was reported as on a missing drive") + } + + if runtime.GOOS != "windows" { + return + } + // Find a drive letter that is not mounted. + for c := 'Z'; c >= 'D'; c-- { + root := string(c) + `:\` + if _, err := os.Stat(root); err == nil { + continue + } + p := string(c) + `:\SteamLibrary\steamapps\common\Game\save` + if !SaveDriveMissing(p) { + t.Errorf("%s: drive %c: is absent but was not reported missing", p, c) + } + if !SaveFolderMissing(p) { + t.Errorf("%s: SaveFolderMissing should still report the folder missing", p) + } + return + } + t.Skip("every drive letter D-Z is mounted") +} diff --git a/internal/daemon/relocate_test.go b/internal/daemon/relocate_test.go new file mode 100644 index 00000000..152755fc --- /dev/null +++ b/internal/daemon/relocate_test.go @@ -0,0 +1,82 @@ +package daemon + +import ( + "os" + "path/filepath" + "runtime" + "strings" + "testing" + + "github.com/opensave/opensave/internal/store" +) + +// freeDriveLetter returns a drive letter not mounted here, or "". +func freeDriveLetter() string { + for c := 'Z'; c >= 'D'; c-- { + if _, err := os.Stat(string(c) + `:\`); err != nil { + return string(c) + } + } + return "" +} + +// A game tracked elsewhere with its save on a drive this device lacks, while +// the identical path exists on another drive here, is switched to it — once. +func TestRelocateToOtherDrive(t *testing.T) { + if runtime.GOOS != "windows" { + t.Skip("drive letters are a Windows thing") + } + free := freeDriveLetter() + if free == "" { + t.Skip("no free drive letter") + } + d := newTestDaemon(t) + real := filepath.Join(t.TempDir(), "SteamLibrary", "steamapps", "common", "Game", "save") + if err := os.MkdirAll(real, 0o777); err != nil { + t.Fatal(err) + } + elsewhere := free + ":" + real[2:] // same path, absent drive + if err := d.Store.CreateGame(store.Game{ID: "g", Name: "Game", SavePath: elsewhere, ActiveBranch: "main", AutoSync: true}); err != nil { + t.Fatal(err) + } + + d.noteMissing("g", "Game", elsewhere, true) + + g, err := d.Store.GetGame("g") + if err != nil { + t.Fatal(err) + } + if !strings.EqualFold(g.SavePath, real) { + t.Fatalf("save path = %s, want it switched to %s", g.SavePath, real) + } + d.missingMu.Lock() + stillMissing := d.missing["g"] + d.missingMu.Unlock() + if stillMissing { + t.Error("the game is still marked missing after being found") + } +} + +// Not found: said once, and not searched again on every reconcile. +func TestRelocateNotFound_IsNotRepeated(t *testing.T) { + if runtime.GOOS != "windows" { + t.Skip("drive letters are a Windows thing") + } + free := freeDriveLetter() + if free == "" { + t.Skip("no free drive letter") + } + d := newTestDaemon(t) + path := free + `:\NoSuchLibrary\steamapps\common\Nothing\save` + if err := d.Store.CreateGame(store.Game{ID: "n", Name: "Nothing", SavePath: path, ActiveBranch: "main", AutoSync: true}); err != nil { + t.Fatal(err) + } + if d.relocateToOtherDrive("n", "Nothing", path) { + t.Fatal("relocated to a folder that does not exist") + } + first := d.relocateChecked["n"] + d.relocateToOtherDrive("n", "Nothing", path) + if !d.relocateChecked["n"].Equal(first) { + t.Error("searched again straight away; it should wait relocateRecheck") + } +} From 7731db8fa214a56bbeb3146cc5858969763ed556 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:00:25 +0200 Subject: [PATCH 11/31] app: smaller memory footprint (GOGC 50, 1 GiB soft limit) OpenSave runs in the background, and with Go's default (let the heap double before collecting) a large save's manifests showed up as twice their size in Task Manager. GOGC 50 and a 1 GiB soft memory limit make the runtime collect and return memory sooner, for a little CPU. GOGC and GOMEMLIMIT set in the environment still win. Co-Authored-By: Claude Opus 5.5 --- cmd/opensave-app/main.go | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/cmd/opensave-app/main.go b/cmd/opensave-app/main.go index 09e46f07..4f8ca678 100644 --- a/cmd/opensave-app/main.go +++ b/cmd/opensave-app/main.go @@ -7,6 +7,7 @@ import ( "fmt" "os" "runtime" + "runtime/debug" "github.com/wailsapp/wails/v2" "github.com/wailsapp/wails/v2/pkg/options" @@ -45,6 +46,8 @@ func main() { // removes its leftover binary) before claiming the single-instance lock. cleanupReplacedBinary() + tuneGC() + app := NewApp() err := wails.Run(&options.App{ @@ -97,3 +100,17 @@ func main() { os.Exit(1) } } + +// tuneGC trades a little CPU for a smaller footprint. OpenSave is a +// background app, and Go's default (let the heap double before collecting) +// meant a large save's manifests showed up as twice their size in Task +// Manager. The soft limit makes the runtime collect and hand memory back to +// the OS harder as it approaches 1GiB. Explicit GOGC / GOMEMLIMIT win. +func tuneGC() { + if os.Getenv("GOGC") == "" { + debug.SetGCPercent(50) + } + if os.Getenv("GOMEMLIMIT") == "" { + debug.SetMemoryLimit(1 << 30) + } +} From 40c76750080639f62168066feadaacd54ccb31dc Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:47:34 +0200 Subject: [PATCH 12/31] watcher: changes OpenSave makes itself are not new saves New package internal/owntouch records the paths OpenSave itself writes, removes, creates or re-dates: pulled files, a peer's deletions, the folders a pull creates, the files "keep theirs" removes and the mtimes "keep mine" touches. The watcher classifies every event by it: a burst made only of OpenSave's own changes moves the recorded hash and nothing else - no auto-snapshot, and no OnChanged sending the sync straight back out. Before, every change a sync applied looked exactly like the game saving. A peer deleting files one request at a time produced an auto-snapshot every couple of seconds, each of a save that was neither the old one nor the new one; they filled the retention budget and pushed out the snapshots that mattered. - a file counts as OpenSave's only if nothing wrote it after the mark (its mtime is not later), so a game saving straight after a sync is still the game's change. - once OpenSave is done with a file (renamed into place, re-dated to the peer's time) it records the time and size it left it with, and from then on the file is OpenSave's only while it has exactly those - a game saving within the two seconds the time comparison allows is still the game's. - a path found gone counts as OpenSave's only if OpenSave removed it (MarkRemoved): a file it pulled and the game or a person then deleted is their change. - the rescan after new folders come under watch is OpenSave's own when it created those folders (a pull of a save with subfolders), and someone else's otherwise. Co-Authored-By: Claude Opus 5.5 --- internal/delta/patch.go | 5 + internal/owntouch/owntouch.go | 137 ++++++++++++++++++++++++++++ internal/owntouch/owntouch_test.go | 132 +++++++++++++++++++++++++++ internal/p2p/routes.go | 3 + internal/p2p/syncengine/conflict.go | 6 ++ internal/p2p/syncengine/engine.go | 17 +++- internal/watcher/owntouch_test.go | 127 ++++++++++++++++++++++++++ internal/watcher/watcher.go | 62 ++++++++++++- 8 files changed, 485 insertions(+), 4 deletions(-) create mode 100644 internal/owntouch/owntouch.go create mode 100644 internal/owntouch/owntouch_test.go create mode 100644 internal/watcher/owntouch_test.go diff --git a/internal/delta/patch.go b/internal/delta/patch.go index 7976fc45..98e0104b 100644 --- a/internal/delta/patch.go +++ b/internal/delta/patch.go @@ -9,6 +9,7 @@ import ( "time" "github.com/opensave/opensave/internal/fsx" + "github.com/opensave/opensave/internal/owntouch" ) // BlockSource supplies the bytes for one block index, either from the @@ -77,7 +78,11 @@ func replaceWithRetry(tmpPath, filePath string) error { delay *= 2 } clearReadOnlyIfSet(filePath) + // Marked before the rename, so the watcher's event finds it already + // known as a change OpenSave made (owntouch). + owntouch.Mark(filePath) if err = renameFile(tmpPath, filePath); err == nil { + owntouch.Settled(filePath) return nil } } diff --git a/internal/owntouch/owntouch.go b/internal/owntouch/owntouch.go new file mode 100644 index 00000000..e63987a5 --- /dev/null +++ b/internal/owntouch/owntouch.go @@ -0,0 +1,137 @@ +// Package owntouch remembers which paths in save folders OpenSave itself has +// just written or removed, so the watcher can tell those changes apart from +// ones a game or a person made. +// +// Without it every change OpenSave applied — a file pulled from a peer, a +// deletion a peer asked for, the mtimes "keep mine" touches — looked to the +// watcher exactly like the game saving. A peer deleting files one request at +// a time over several minutes produced a full auto-snapshot every couple of +// seconds, each of a save that was neither the old one nor the new one, and +// those filled the retention budget and pushed out the snapshots that mattered. +package owntouch + +import ( + "os" + "path/filepath" + "runtime" + "strings" + "sync" + "time" +) + +// window is how long a mark counts. Long enough to cover the watcher's +// debounce and a slow burst of events behind it; short enough that the same +// path changed again by a game a little later is not mistaken for ours. +const window = 2 * time.Minute + +var ( + mu sync.Mutex + marked = map[string]mark{} +) + +// mark is when OpenSave touched a path, whether it removed it, and - once +// OpenSave is done with a file (Settled) - the time and size it left it with. +type mark struct { + at time.Time + removed bool + settled bool + mod time.Time + size int64 +} + +func key(path string) string { + p := filepath.Clean(path) + if runtime.GOOS == "windows" { + p = strings.ToLower(p) + } + return p +} + +// Mark records that OpenSave itself just created, wrote, removed or re-dated +// path. Its parent folder is marked too: creating or removing an entry is an +// event on the folder as well. +func Mark(path string) { record(path, false) } + +// MarkRemoved records that OpenSave itself is about to remove path. Only a +// path marked this way counts as OpenSave's when it is found gone: a file +// OpenSave has just written and a game or a person then deleted is their +// change, and has to sync and be snapshotted as one. +func MarkRemoved(path string) { record(path, true) } + +// Settled records the time and size OpenSave has left a file it marked with, +// once it is done with it (written, renamed into place, re-dated). From then on +// the file is OpenSave's only while it still has exactly those: a game saving +// over it a moment later - inside any slack a time comparison would need - is +// the game's change. +func Settled(path string) { + fi, err := os.Lstat(path) + if err != nil || fi.IsDir() { + return + } + k := key(path) + mu.Lock() + defer mu.Unlock() + if m, ok := marked[k]; ok { + m.settled, m.mod, m.size = true, fi.ModTime(), fi.Size() + marked[k] = m + } +} + +func record(path string, removed bool) { + if path == "" { + return + } + now := time.Now() + mu.Lock() + defer mu.Unlock() + marked[key(path)] = mark{at: now, removed: removed} + marked[key(filepath.Dir(path))] = mark{at: now} + if len(marked) > 50000 { + for k, m := range marked { + if now.Sub(m.at) > window { + delete(marked, k) + } + } + } +} + +// slack is how much later than its mark a file OpenSave wrote may be dated: +// re-dating to now happens just after marking, and some filesystems keep +// times to two seconds. +const slack = 2 * time.Second + +// Recent reports whether the change just seen at path is one OpenSave made: +// the path was marked within the window, and nothing has written it since. +// +// The second half matters as much as the first. A game often saves straight +// after a sync — the same file OpenSave has just put there — and a mark alone +// would take that save for the sync's for two minutes: no snapshot of it, and +// no new version to hand to anyone. Whatever OpenSave leaves behind is dated +// no later than its mark (a pulled file keeps the time it was written to its +// temporary name, or is given the peer's; re-dating sets now), so a file +// dated later was written by someone else. Once OpenSave is done with a file +// (Settled) the test is exact: the time and size it left, or not ours - the +// slack would take a game's save within two seconds of a pull for the pull's. +// A path that is gone is OpenSave's +// only if OpenSave removed it (MarkRemoved): one it wrote and someone then +// deleted is theirs. A folder carries no save of its own, so its mark is +// enough. +func Recent(path string) bool { + mu.Lock() + m, ok := marked[key(path)] + mu.Unlock() + if !ok || time.Since(m.at) > window { + return false + } + fi, err := os.Lstat(path) + if err != nil { + return m.removed + } + if fi.IsDir() { + return true + } + if m.settled { + return fi.ModTime().Equal(m.mod) && fi.Size() == m.size + } + return !fi.ModTime().After(m.at.Add(slack)) +} diff --git a/internal/owntouch/owntouch_test.go b/internal/owntouch/owntouch_test.go new file mode 100644 index 00000000..824f04a4 --- /dev/null +++ b/internal/owntouch/owntouch_test.go @@ -0,0 +1,132 @@ +package owntouch + +import ( + "os" + "path/filepath" + "testing" + "time" +) + +func TestMarkAndRecent(t *testing.T) { + dir := t.TempDir() + file := filepath.Join(dir, "sub", "slot1.sav") + if Recent(file) { + t.Fatal("unmarked path reported as ours") + } + Mark(file) + if err := os.MkdirAll(filepath.Dir(file), 0o777); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(file, []byte("pulled"), 0o666); err != nil { + t.Fatal(err) + } + if !Recent(file) { + t.Error("marked path not reported as ours") + } + if !Recent(filepath.Dir(file)) { + t.Error("the parent folder of a marked path not reported as ours") + } + if Recent(filepath.Join(dir, "sub", "slot2.sav")) { + t.Error("a sibling of a marked path reported as ours") + } +} + +// A game saving over a file OpenSave has just written is the game's change, +// however soon after. +func TestRecent_AWriteAfterTheMarkIsNotOurs(t *testing.T) { + dir := t.TempDir() + file := filepath.Join(dir, "slot1.sav") + Mark(file) + if err := os.WriteFile(file, []byte("pulled"), 0o666); err != nil { + t.Fatal(err) + } + if !Recent(file) { + t.Fatal("the file OpenSave wrote is not reported as ours") + } + // The game saves a little later. + later := time.Now().Add(5 * time.Second) + if err := os.WriteFile(file, []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if err := os.Chtimes(file, later, later); err != nil { + t.Fatal(err) + } + if Recent(file) { + t.Error("a save written after OpenSave's was taken for OpenSave's") + } +} + +// A file OpenSave has just written, then deleted by a game or a person, is +// their deletion: it has to sync and be snapshotted. Only a path OpenSave +// removed itself is its own when found gone. +func TestRecent_AFileDeletedAfterOurWriteIsNotOurs(t *testing.T) { + dir := t.TempDir() + pulled := filepath.Join(dir, "pulled.sav") + Mark(pulled) + if err := os.WriteFile(pulled, []byte("from the peer"), 0o666); err != nil { + t.Fatal(err) + } + if err := os.Remove(pulled); err != nil { + t.Fatal(err) + } + if Recent(pulled) { + t.Error("a file deleted by someone else after OpenSave wrote it was taken for OpenSave's") + } + + removed := filepath.Join(dir, "removed.sav") + if err := os.WriteFile(removed, []byte("old"), 0o666); err != nil { + t.Fatal(err) + } + MarkRemoved(removed) + if err := os.Remove(removed); err != nil { + t.Fatal(err) + } + if !Recent(removed) { + t.Error("a file OpenSave removed was not reported as its own") + } +} + +// A game saving over a file OpenSave has just pulled is the game's change +// however soon it comes - inside the slack a time comparison allows too. +func TestRecent_AWriteRightAfterASettledPullIsNotOurs(t *testing.T) { + file := filepath.Join(t.TempDir(), "progress.sav") + Mark(file) + if err := os.WriteFile(file, []byte("from A"), 0o666); err != nil { + t.Fatal(err) + } + Settled(file) + if !Recent(file) { + t.Fatal("the file OpenSave pulled is not reported as ours") + } + // Same size, a moment later: only the time tells them apart. + time.Sleep(20 * time.Millisecond) + if err := os.WriteFile(file, []byte("from B"), 0o666); err != nil { + t.Fatal(err) + } + if Recent(file) { + t.Error("a save written straight after the pull was taken for OpenSave's") + } +} + +// A pull renames a file into place, then re-dates it to the peer's time. In +// between, and before it is Settled again, the file is still OpenSave's. +func TestRecent_ARedatedPullBeforeItSettlesAgainIsOurs(t *testing.T) { + file := filepath.Join(t.TempDir(), "auto.sav") + Mark(file) + if err := os.WriteFile(file, []byte("pulled"), 0o666); err != nil { + t.Fatal(err) + } + Settled(file) + Mark(file) + peers := time.Now().Add(-time.Hour) + if err := os.Chtimes(file, peers, peers); err != nil { + t.Fatal(err) + } + if !Recent(file) { + t.Error("a pulled file re-dated to the peer's time was taken for someone else's change") + } + Settled(file) + if !Recent(file) { + t.Error("a pulled file settled at the peer's time was taken for someone else's change") + } +} diff --git a/internal/p2p/routes.go b/internal/p2p/routes.go index 7d84c1c3..65168ecd 100644 --- a/internal/p2p/routes.go +++ b/internal/p2p/routes.go @@ -20,6 +20,7 @@ import ( "github.com/go-chi/chi/v5" "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/logging" + "github.com/opensave/opensave/internal/owntouch" "github.com/opensave/opensave/internal/p2p/pairing" "github.com/opensave/opensave/internal/p2p/syncengine" "github.com/opensave/opensave/internal/store" @@ -968,6 +969,8 @@ func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { entry, _ = delta.FileEntryFor(full) } // Empty dirs only, for a folder, like rmdirSync. + // A change made at a peer's request, not by the game (owntouch). + owntouch.MarkRemoved(full) if os.Remove(full) == nil && (info.IsDir() || entry.Hash != "") { e.Sync.NotePeerDeletion(game.ID, body.Root, body.RelPath, entry, info.IsDir()) } diff --git a/internal/p2p/syncengine/conflict.go b/internal/p2p/syncengine/conflict.go index 1cef4b5b..4167ad72 100644 --- a/internal/p2p/syncengine/conflict.go +++ b/internal/p2p/syncengine/conflict.go @@ -8,6 +8,7 @@ import ( "time" "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" "github.com/opensave/opensave/internal/snapshot" ) @@ -222,14 +223,18 @@ func (e *Engine) touchSaveMtimes(gameID, root string) { return } if !info.IsDir() { + owntouch.Mark(root) _ = os.Chtimes(root, now, now) + owntouch.Settled(root) return } _ = filepath.Walk(root, func(path string, fi os.FileInfo, walkErr error) error { if walkErr != nil || fi.IsDir() { return nil } + owntouch.Mark(path) _ = os.Chtimes(path, now, now) + owntouch.Settled(path) return nil }) } @@ -362,6 +367,7 @@ func (e *Engine) overwriteLocalWithRemote(ctx context.Context, gameID string, pe } full := filepath.Join(game.SavePath, filepath.FromSlash(relPath)) _ = os.Chmod(full, 0o666) + owntouch.MarkRemoved(full) _ = os.Remove(full) } diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index 8752e573..5e71bdcc 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -12,6 +12,7 @@ import ( "time" "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" "github.com/opensave/opensave/internal/snapshot" "github.com/opensave/opensave/internal/store" "github.com/opensave/opensave/internal/syncpause" @@ -1458,6 +1459,7 @@ func (e *Engine) applyLocalDeletions(gameID string, root syncRoot, d Decision) { // spelling this disk actually holds. full := delta.LocalNameFor(root.Path, relPath) _ = os.Chmod(full, 0o666) + owntouch.MarkRemoved(full) if err := os.Remove(full); err == nil { e.Log("info", "deleted locally (peer deleted): "+relPath) } @@ -1472,6 +1474,7 @@ func (e *Engine) applyLocalDeletions(gameID string, root syncRoot, d Decision) { } full := delta.LocalNameFor(root.Path, relDir) if info, err := os.Stat(full); err == nil && info.IsDir() { + owntouch.MarkRemoved(full) if err := os.Remove(full); err == nil { // only removes empty dirs, matching rmdirSync e.Log("info", "deleted directory locally (peer deleted): "+relDir) } @@ -1514,7 +1517,9 @@ func (e *Engine) createPulledDirsIn(gameID string, root syncRoot, dirsToPull []s if !delta.IsSafePath(root.Path, relDir) { continue } - _ = os.MkdirAll(filepath.Join(root.Path, filepath.FromSlash(relDir)), 0o777) + full := filepath.Join(root.Path, filepath.FromSlash(relDir)) + owntouch.Mark(full) + _ = os.MkdirAll(full, 0o777) } } @@ -1562,7 +1567,11 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s // Make sure every remote directory exists before patching into it. for _, dir := range remoteData.Manifest.Dirs { if delta.IsSafePath(root.Path, dir) { - _ = os.MkdirAll(filepath.Join(root.Path, filepath.FromSlash(dir)), 0o777) + full := filepath.Join(root.Path, filepath.FromSlash(dir)) + if _, err := os.Stat(full); err != nil { + owntouch.Mark(full) + _ = os.MkdirAll(full, 0o777) + } } } @@ -1729,7 +1738,11 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s } if remoteFile.MtimeMs > 0 { mtime := time.UnixMilli(int64(remoteFile.MtimeMs)) + // Marked again first: until Settled below, the time the rename left + // is not the one to compare (owntouch). + owntouch.Mark(localFilePath) _ = os.Chtimes(localFilePath, mtime, mtime) + owntouch.Settled(localFilePath) } pulled = append(pulled, relPath) e.Log("info", "file updated: "+relPath) diff --git a/internal/watcher/owntouch_test.go b/internal/watcher/owntouch_test.go new file mode 100644 index 00000000..6a1e8134 --- /dev/null +++ b/internal/watcher/owntouch_test.go @@ -0,0 +1,127 @@ +package watcher + +import ( + "fmt" + "os" + "path/filepath" + "testing" + "time" + + "github.com/opensave/opensave/internal/owntouch" +) + +// Files OpenSave writes or deletes itself — a pull, a peer's deletions — are +// not a new save. Only the recorded hash moves; no auto-snapshot, and no +// OnChanged echoing the sync back out. +func TestOwnChanges_AreNotSnapshotted(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + eng := New(col.callbacks()) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + + // A peer's deletions arriving one at a time, as they do, plus pulled files. + for i := 0; i < 5; i++ { + p := filepath.Join(saveDir, fmt.Sprintf("map_%d.bin", i)) + owntouch.Mark(p) + if err := os.WriteFile(p, []byte(fmt.Sprint(i)), 0o666); err != nil { + t.Fatal(err) + } + time.Sleep(300 * time.Millisecond) + } + // Past the debounce, with room to spare. + if !waitFor(t, 8*time.Second, func() bool { + col.mu.Lock() + defer col.mu.Unlock() + return col.manifestHashes["game1"] != "" + }) { + t.Fatal("the recorded hash was not moved to the new content") + } + time.Sleep(3 * time.Second) + if n := col.snapshotCount(); n != 0 { + t.Errorf("%d auto-snapshot(s) for changes OpenSave made itself, want 0", n) + } + if n := col.changedCount(); n != 0 { + t.Errorf("OnChanged fired %d time(s) for changes OpenSave made itself", n) + } + + // The game then saves: that is snapshotted, as always. + if err := os.WriteFile(filepath.Join(saveDir, "slot1.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("a change the game made was not snapshotted") + } +} + +// A pull that creates folders and writes into them before the watcher has put +// them under watch is still the pull: the rescan that follows registering +// them is not someone else's change. +func TestOwnNewFolders_AreNotSnapshotted(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + eng := New(col.callbacks()) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + for i := 0; i < 20; i++ { + dir := filepath.Join(saveDir, fmt.Sprintf("slot%02d", i)) + owntouch.Mark(dir) + if err := os.MkdirAll(dir, 0o777); err != nil { + t.Fatal(err) + } + p := filepath.Join(dir, "data.sav") + owntouch.Mark(p) + if err := os.WriteFile(p, []byte(fmt.Sprint(i)), 0o666); err != nil { + t.Fatal(err) + } + } + if !waitFor(t, 10*time.Second, func() bool { + col.mu.Lock() + defer col.mu.Unlock() + return col.manifestHashes["game1"] != "" + }) { + t.Fatal("the recorded hash was not moved to the new content") + } + time.Sleep(4 * time.Second) + if n := col.snapshotCount(); n != 0 { + t.Errorf("%d auto-snapshot(s) for folders and files a pull created", n) + } + + // A folder the game creates is a change, as before. + if err := os.MkdirAll(filepath.Join(saveDir, "newslot"), 0o777); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(saveDir, "newslot", "data.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("a folder the game created was not snapshotted") + } +} + +// A burst that mixes OpenSave's own changes with one from the game is a real +// change and is snapshotted. +func TestMixedChanges_AreSnapshotted(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + eng := New(col.callbacks()) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + own := filepath.Join(saveDir, "pulled.bin") + owntouch.Mark(own) + if err := os.WriteFile(own, []byte("from a peer"), 0o666); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(saveDir, "slot1.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("a burst including the game's own change was not snapshotted") + } +} diff --git a/internal/watcher/watcher.go b/internal/watcher/watcher.go index 612ebfb6..3c9b7ee9 100644 --- a/internal/watcher/watcher.go +++ b/internal/watcher/watcher.go @@ -32,6 +32,7 @@ import ( "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/ignore" "github.com/opensave/opensave/internal/logging" + "github.com/opensave/opensave/internal/owntouch" ) const ( @@ -188,6 +189,17 @@ type gameWatch struct { // registerFolders when an Add fails. rewatch atomic.Bool + // What the burst since the last snapshot check was made of: changes + // OpenSave applied itself (owntouch), anything else, or both. Read and + // written only by the event loop. + ownEvents, otherEvents bool + // foreignFolders: a folder queued for watching since the last rescan was + // not one OpenSave created, or the watch started on a busy folder. The + // rescan that follows registration counts as someone else's change only + // then — a pull creates folders and writes into them before they are + // watched, and that is still the pull. Event loop only. + foreignFolders bool + // folders carries directories to put under watch to registerFolders. See // there for why the event loop does not register them itself. folders chan string @@ -450,6 +462,11 @@ func (e *Engine) WatchWithLocations(gameID, savePath string, extra map[string]st } if busy { gw.rewatch.Store(true) + // What the folder holds is nobody's change OpenSave knows of: this + // rescan must not record it as its own, or a change made while the + // app was closed would become the baseline before the catch-up + // below could see it. Set before the event loop starts. + gw.foreignFolders = true gw.rescan <- struct{}{} // buffered and empty: cannot block } e.games[gameID] = gw @@ -676,10 +693,19 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { if !gw.eventRelevant(event) { continue } + own := owntouch.Recent(event.Name) + if own { + gw.ownEvents = true + } else { + gw.otherEvents = true + } // New subdirectory in directory mode: extend the watch — on // registerFolders, never here. See registerFolders. if !gw.isFile && event.Has(fsnotify.Create) { if info, err := os.Stat(event.Name); err == nil && info.IsDir() { + if !own { + gw.foreignFolders = true + } gw.queueFolder(event.Name) } } @@ -689,7 +715,14 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { // Folders just came under watch, or the watch started on a folder // that was busy. Anything written into a folder before its watch // existed raised no event of its own, so read the tree again once - // things are quiet. + // things are quiet. Whose change that is follows from whose + // folders they were: what OpenSave created, it also filled. + if gw.foreignFolders { + gw.otherEvents = true + } else { + gw.ownEvents = true + } + gw.foreignFolders = false resetDebounce() case err, ok := <-gw.fsw.Errors: @@ -718,6 +751,7 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { // directory and allocates no second buffer for one it already // has. gw.rewatch.Store(true) + gw.otherEvents = true // the missed events could be anyone's } else { e.log("warn", fmt.Sprintf("watching %q: %v", gw.gameID, err)) continue @@ -758,7 +792,9 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { for _, path := range gw.extra { delta.InvalidateRoot(path) } - e.handleChange(ctx, gw) + ownOnly := gw.ownEvents && !gw.otherEvents + gw.ownEvents, gw.otherEvents = false, false + e.handleChangeFrom(ctx, gw, ownOnly) } } } @@ -788,6 +824,22 @@ func (gw *gameWatch) eventRelevant(event fsnotify.Event) bool { // (gameplay guard), skip if content is unchanged since the last // auto-snapshot, then snapshot with retries and notify. func (e *Engine) handleChange(ctx context.Context, gw *gameWatch) { + e.handleChangeFrom(ctx, gw, false) +} + +// handleChangeFrom is handleChange told whether every change in the burst was +// one OpenSave applied itself — a pull, a deletion a peer asked for, mtimes a +// conflict resolution touched (owntouch). +// +// Those are not a new save. They are a sync, which keeps its own history +// (the mirror snapshot of a pull, the safety snapshot before files are +// replaced). Snapshotting them as well took a full auto-snapshot every couple +// of seconds while a peer deleted files one request at a time, each of a save +// half-way between two states, until they had filled the retention budget and +// pushed out the snapshots worth keeping. So the burst only moves the recorded +// hash, which keeps the next real change measured against what is actually on +// disk; anything not known to be ours is snapshotted as before. +func (e *Engine) handleChangeFrom(ctx context.Context, gw *gameWatch, ownOnly bool) { // Gameplay guard: the game may still be mid-write. for anyFileLocked(gw.savePath) { e.log("info", fmt.Sprintf("save files for %q are in use; waiting (gameplay guard)", gw.gameID)) @@ -821,6 +873,12 @@ func (e *Engine) handleChange(ctx context.Context, gw *gameWatch) { e.log("info", fmt.Sprintf("no content change for %q; skipping auto-snapshot", gw.gameID)) return } + if ownOnly { + if err := e.cb.SetLastManifestHash(gw.gameID, currentHash); err != nil { + e.log("warn", fmt.Sprintf("failed to record manifest hash for %q: %v", gw.gameID, err)) + } + return + } for attempt := 1; attempt <= snapshotMaxRetries; attempt++ { err = e.cb.CreateSnapshot(gw.gameID) From 33a6abf1d2f7c4b2a83623234d85c958701de8cf Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:34:32 +0200 Subject: [PATCH 13/31] sync: batched pulls tell owntouch what they write The batched pull re-dates each file to the peer's time like the per-file pull does, and like it marks the file first and settles it after, so the watcher takes a batch for OpenSave's own change. Co-Authored-By: Claude Opus 5.5 --- internal/p2p/syncengine/batch.go | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/internal/p2p/syncengine/batch.go b/internal/p2p/syncengine/batch.go index 7a15d94e..c1b1ff23 100644 --- a/internal/p2p/syncengine/batch.go +++ b/internal/p2p/syncengine/batch.go @@ -10,6 +10,7 @@ import ( "time" "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" ) // Pulling many small files, many to a request. @@ -270,7 +271,9 @@ func writeBatchedFile(j batchJob, blocks []BlockData) error { committed = true if j.remote.MtimeMs > 0 { mtime := time.UnixMilli(int64(j.remote.MtimeMs)) + owntouch.Mark(j.localPath) // see pullFiles: re-dated, then Settled _ = os.Chtimes(j.localPath, mtime, mtime) + owntouch.Settled(j.localPath) } return nil } @@ -288,7 +291,9 @@ func (e *Engine) pullBatchOneByOne(ctx context.Context, peer Peer, gameID, root } if j.remote.MtimeMs > 0 { mtime := time.UnixMilli(int64(j.remote.MtimeMs)) + owntouch.Mark(j.localPath) _ = os.Chtimes(j.localPath, mtime, mtime) + owntouch.Settled(j.localPath) } written = append(written, j.relPath) } From c0cf59d518513b26f21326b2bc28f58b7168c7e6 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:47:41 +0200 Subject: [PATCH 14/31] sync: decide by save versions, not by guessing from files Every device now records, per game, which version of the save it holds as a version vector: one counter per device, moved only when the save really changes there (never by a sync; see internal/owntouch). The version travels in the manifest response, and two devices compare versions before anything else: - one older than the other: that device is behind. It takes the newer save whole - every file and every deletion - and only from a device holding it completely; only then does it adopt that version. Until then it is a source for nobody: nothing is taken from it, pushed by it, or deleted on its word. A pull that stops part-way is remembered across restarts, so the half-copy is never mistaken for a change made there. - a device remembers the newest version it has heard of (its target). Two devices that both know a newer one exists do nothing with each other until a device holding it is back. This is what stopped a half-finished copy being passed on, with everything it lacked read as deletions. - an empty folder holds no version, so it fills itself and never raises a conflict. - neither newer: both changed independently. Only this asks. Resolution stays mutual: "keep mine" on one device does not make the other's save simply older - the other device is asked in turn, and the first is not asked again (the answer travels with the version). If both keep their own, the later answer stands. Every sync and every manifest served first checks the save against the hash recorded with its version, so a change the watcher has not reported yet (it waits for the game to finish writing) is a version before anyone compares - otherwise a device with unrecorded work would look merely behind and take the other's save over it. Changes to excluded files are not versions, and neither is writing an exclusion rule: the recorded hash is tagged with the rules it was taken under and simply re-taken when they change. A device fetching back a save it was asked to put back after it was emptied (hold.go) follows the hold's rules. Saves from before versions get one named after their content, so devices that already agreed agree at once; ones that did not are compared as before until they agree. Games with automatic sync off are not watched, so they keep the old file comparison; so do peers that predate this. Also: - the mass-deletion guard is removed. It held syncs that would delete many files for a decision - a stopgap for deletions read from a half-finished copy, which versions now rule out at the source. - the sync engine marks what it writes and removes (owntouch), and a burst of deletions a peer asks for takes one safety snapshot, at most every ten minutes, instead of one auto-snapshot per file. --- internal/cliapp/syncsummary.go | 4 + internal/daemon/daemon.go | 3 + internal/p2p/engine.go | 4 + internal/p2p/routes.go | 37 + internal/p2p/syncengine/conflict.go | 14 + internal/p2p/syncengine/engine.go | 50 +- internal/p2p/syncengine/massdelete.go | 28 - internal/p2p/syncengine/massdelete_test.go | 69 -- internal/p2p/syncengine/quick.go | 10 + internal/p2p/syncengine/transport.go | 4 + internal/p2p/syncengine/version.go | 827 ++++++++++++++++++ internal/p2p/syncengine/version_test.go | 575 ++++++++++++ .../store/migrations/0039_game_versions.sql | 42 + internal/store/versions.go | 71 ++ 14 files changed, 1624 insertions(+), 114 deletions(-) delete mode 100644 internal/p2p/syncengine/massdelete.go delete mode 100644 internal/p2p/syncengine/massdelete_test.go create mode 100644 internal/p2p/syncengine/version.go create mode 100644 internal/p2p/syncengine/version_test.go create mode 100644 internal/store/migrations/0039_game_versions.sql create mode 100644 internal/store/versions.go diff --git a/internal/cliapp/syncsummary.go b/internal/cliapp/syncsummary.go index b5518341..b07eeb07 100644 --- a/internal/cliapp/syncsummary.go +++ b/internal/cliapp/syncsummary.go @@ -101,6 +101,10 @@ func outcomeFromPeers(gameID string, peers map[string]peerSyncResult) syncOutcom waiting = who + " is waiting for a folder to be chosen for it" case "peer_holding": waiting = who + " is holding it back: its save was emptied there" + case "waiting": + if waiting == "" { + waiting = "a newer version is on a device that is not online" + } case "peer_missing": if waiting == "" { waiting = who + " does not track it" diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 5683cd88..4e2ffcf0 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -233,6 +233,9 @@ func New(opts Options) (*Daemon, error) { return err }, OnChanged: func(gameID string) { + // A change made here, by the game or the user: a new version of + // the save, recorded before anyone is offered it. + d.P2P.Sync.NoteLocalChange(gameID) // An emptied save is noticed as it happens, and said on screen, // even with no other device online to hold it back from. _, _ = d.P2P.Sync.CheckHold(gameID, false) diff --git a/internal/p2p/engine.go b/internal/p2p/engine.go index 1dd69bbb..83e68ea7 100644 --- a/internal/p2p/engine.go +++ b/internal/p2p/engine.go @@ -129,6 +129,10 @@ type Engine struct { // (probeLinkSoon). probeLinksMu sync.Mutex probingLinks map[string]bool + // When each game last got a safety snapshot before a peer's deletions + // (snapshotBeforePeerDeletions). + peerDeleteSnapMu sync.Mutex + peerDeleteSnap map[string]time.Time // answered holds the peers that answered a probe during this run, and // startedMs is when the run began (heardThisRun). answered map[string]bool diff --git a/internal/p2p/routes.go b/internal/p2p/routes.go index 65168ecd..4dd9a74f 100644 --- a/internal/p2p/routes.go +++ b/internal/p2p/routes.go @@ -754,6 +754,7 @@ func (e *Engine) handleManifest(w http.ResponseWriter, r *http.Request) { ActiveBranch: game.ActiveBranch, Proto: ServedProto(), DeletionConfirmed: e.Sync.DeletionConfirmed(game.ID), + Version: e.Sync.LocalVersion(game, manifest), } if latest, err := e.Snapshots.LatestSnapshot(gameID, ""); err == nil { resp.LatestSnapshot = &syncengine.SnapshotInfo{ID: latest.ID, Timestamp: latest.Timestamp, Comment: latest.Comment} @@ -927,6 +928,37 @@ func (e *Engine) handleFileBatch(w http.ResponseWriter, r *http.Request) { jsonOK(w, map[string]any{"files": out}) } +// peerDeleteSnapshotEvery is how far apart two safety snapshots before a +// peer's deletions may be: one for a burst of deletions, however many files +// it removes one request at a time. +const peerDeleteSnapshotEvery = 10 * time.Minute + +// snapshotBeforePeerDeletions takes a safety snapshot of a game before the +// first of a burst of deletions a peer asks for. +func (e *Engine) snapshotBeforePeerDeletions(gameID string, r *http.Request) { + if e.Snapshots == nil { + return + } + e.peerDeleteSnapMu.Lock() + if e.peerDeleteSnap == nil { + e.peerDeleteSnap = map[string]time.Time{} + } + if last, ok := e.peerDeleteSnap[gameID]; ok && time.Since(last) < peerDeleteSnapshotEvery { + e.peerDeleteSnapMu.Unlock() + return + } + e.peerDeleteSnap[gameID] = time.Now() + e.peerDeleteSnapMu.Unlock() + + asker := "another device" + if peer, ok := e.peerByAddress(clientIP(r)); ok { + asker = peer.Name + } + if _, err := e.Snapshots.CreateBeforeReplacing(gameID, fmt.Sprintf("Before %s deleted files here", asker)); err != nil { + e.Log("warn", fmt.Sprintf("safety snapshot before %s's deletions failed: %v", asker, err)) + } +} + func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") var body struct { @@ -961,6 +993,11 @@ func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { _ = os.Chmod(full, 0o666) deleting := time.Now() if info, statErr := os.Stat(full); statErr == nil { + // One copy of the save before a peer starts deleting from it, not + // one per file: the watcher no longer snapshots changes OpenSave + // applies (owntouch), and it was those, one every few seconds of a + // half-deleted save, that had been the only copy before. + e.snapshotBeforePeerDeletions(game.ID, r) // What is removed is remembered, so a sync of this device's own that // lands before the rest of the batch does not take it for a change // made here (syncengine/peerdeleted.go). diff --git a/internal/p2p/syncengine/conflict.go b/internal/p2p/syncengine/conflict.go index 4167ad72..99553538 100644 --- a/internal/p2p/syncengine/conflict.go +++ b/internal/p2p/syncengine/conflict.go @@ -45,6 +45,9 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution if err := e.markResolvedLocal(ctx, gameID, peer); err != nil { return "", err } + // This device's version becomes newer than both, so the peer — and any + // device holding either side's — takes it whole (version.go). + e.versionAfterResolution(gameID, conflict.remoteVersion, true) e.clearConflict(gameID) e.Transport.TriggerPeerPull(peer, gameID) return "", nil @@ -64,8 +67,18 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution e.Log("warn", fmt.Sprintf("safety snapshot before keep-remote failed: %v", err)) } if err := e.overwriteLocalWithRemote(ctx, gameID, peer, remoteData, "Resolved conflict: Overwrite with remote"); err != nil { + // This device's own version is given up either way; what is still + // missing comes from the peer like any outdated device's would. + e.discardLocalVersion(gameID, remoteData.Version) return "", err } + if game, err := e.Store.GetGame(gameID); err == nil { + if now, err := delta.BuildManifest(game.SavePath); err == nil && sameFiles(now, remoteData.Manifest) { + e.versionAfterResolution(gameID, remoteData.Version, false) + } else { + e.discardLocalVersion(gameID, remoteData.Version) + } + } e.markResolvedConverged(gameID, peer) e.clearConflict(gameID) return "", nil @@ -107,6 +120,7 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution if err := e.markResolvedLocal(ctx, gameID, peer); err != nil { return "", err } + e.versionAfterResolution(gameID, conflict.remoteVersion, true) e.clearConflict(gameID) e.Transport.TriggerPeerPull(peer, gameID) return created, nil diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index 5e71bdcc..2c10af0f 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -35,6 +35,10 @@ type Conflict struct { OnlyLocalTotal int `json:"onlyLocalTotal"` OnlyRemoteTotal int `json:"onlyRemoteTotal"` ChangedTotal int `json:"changedTotal"` + + // remoteVersion is the peer's version when the conflict was found, which + // the answer is recorded against (version.go). + remoteVersion *VersionInfo } // SideStats summarises one side's save state for the conflict UI. @@ -151,6 +155,12 @@ type Engine struct { linkMu sync.Mutex links map[string]store.PeerLink pullWaits map[string]chan struct{} + // versionMu serialises reading and writing the save versions + // (version.go); pullingNow is the games this run is pulling a newer + // version of, and loggedOnce the last "waiting" line said per game/peer. + versionMu sync.Mutex + pullingNow map[string]bool + loggedOnce map[string]string } // New creates an Engine. @@ -368,7 +378,7 @@ func (e *Engine) SyncGame(ctx context.Context, gameID string, onlinePeers []Peer // divergence from the NEXT sync, causing the peer to silently overwrite // its own changes instead of detecting the conflict and asking. switch res.Status { - case "conflict", "error": + case "conflict", "error", versionStatusWaiting: case "peer_missing", "peer_awaiting_folder", "peer_holding": // The two devices talked and finished, which is what the // per-device stamp has always recorded. But nothing of THIS game @@ -561,6 +571,27 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re } e.mu.Unlock() + // 3b. The versions decide, when both devices keep them (version.go): who + // is behind takes the newer save whole, from a device that holds it, and + // nothing is ever taken from — or deleted on the word of — a device that + // is behind. Left to the file comparison below only when the two hold + // the same save, or both saves predate versions. + // + // Not while this device is fetching back a save someone emptied here and + // asked to have put back (hold.go): the emptying was recorded as a version + // of its own, and by versions this device would be the newer one, with + // nothing to fetch. The hold's own rules bring the files back. + fetchingBack := false + if h, ok, err := e.Store.GetDeletionHold(gameID); err == nil && ok && h.State == store.HoldFetching { + fetchingBack = true + } + if remoteData.Version != nil && game.AutoSync && !fetchingBack { + if res, handled, err := e.syncByVersion(ctx, game, peer, localManifest, unfilteredLocal, remoteData); handled { + releaseAligned() + return res, err + } + } + // 4. Conflict detection (lineage + skew-tolerant mtimes). lastSyncMs := e.lastSyncTimeMs(peer.ID) agreedHash := e.Store.GetAgreedHash(gameID, peer.ID) @@ -740,22 +771,6 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re } handedOverDeletion := e.handOverEmptying(gameID, peer, localManifest.Files, &decision) - // A sync about to delete a large part of a save — here or on the peer — - // stops and asks instead. emptiedUnconfirmed above covers a save emptied - // completely; this covers one gutted partly, which is what a partial copy - // on one device turns into once it is read as the other's deletions. Two - // waves of that took tens of thousands of files out of a save before - // anyone was asked. A deletion someone confirmed on purpose - // (handedOverDeletion, DeletionConfirmed) is not second-guessed. - if !handedOverDeletion && !remoteData.DeletionConfirmed { - if n, total := massDeletion(localManifest, remoteData.Manifest, decision); n > 0 { - e.Log("warn", fmt.Sprintf("syncing %q with %q would delete %d of %d files — holding it for a decision instead", - game.Name, peer.Name, n, total)) - e.registerConflict(gameID, peer, localManifest, remoteData) - return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, nil - } - } - // Nothing below is about files arriving, only about local files leaving. // A pull that brings files this device never held destroys nothing. atRisk := filesAtRisk(localManifest, decision) @@ -1361,6 +1376,7 @@ func (e *Engine) registerConflict(gameID string, peer Peer, localManifest delta. DiffFiles: diffs, DiffTotal: total, OnlyLocalTotal: counts["only-local"], OnlyRemoteTotal: counts["only-remote"], ChangedTotal: counts["changed"], + remoteVersion: remoteData.Version, } e.mu.Unlock() diff --git a/internal/p2p/syncengine/massdelete.go b/internal/p2p/syncengine/massdelete.go deleted file mode 100644 index aea0cc26..00000000 --- a/internal/p2p/syncengine/massdelete.go +++ /dev/null @@ -1,28 +0,0 @@ -package syncengine - -import "github.com/opensave/opensave/internal/delta" - -// A sync is held for a decision when it would delete more than both of these: -// enough files that it is not a game tidying its slots, and enough of the save -// that it is not one corner of a big one. Ordinary play deletes a handful of -// files; a device holding part of a save being read as the other's deletions -// deletes thousands. -const ( - massDeleteMinFiles = 100 - massDeleteFraction = 0.10 -) - -// massDeletion reports how many files the decision would delete, on either -// side, and how many the save holds, when that crosses the thresholds above; -// zero otherwise. -func massDeletion(local, remote delta.Manifest, d Decision) (deleting, total int) { - n := len(d.FilesToDeleteLocally) + len(d.FilesToDeleteOnPeer) - total = len(local.Files) - if len(remote.Files) > total { - total = len(remote.Files) - } - if n <= massDeleteMinFiles || total == 0 || float64(n) <= massDeleteFraction*float64(total) { - return 0, total - } - return n, total -} diff --git a/internal/p2p/syncengine/massdelete_test.go b/internal/p2p/syncengine/massdelete_test.go deleted file mode 100644 index a18b34c7..00000000 --- a/internal/p2p/syncengine/massdelete_test.go +++ /dev/null @@ -1,69 +0,0 @@ -package syncengine - -import ( - "context" - "fmt" - "os" - "path/filepath" - "testing" -) - -// Both sides converge on n files, then the peer loses some of them. -func convergedThenPeerLoses(t *testing.T, n, lost int) *engineEnv { - t.Helper() - env := setupEngine(t) - writeSmallFiles(t, env.localDir, n) - writeSmallFiles(t, env.remoteDir, n) - if res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil || res.Status != "in_sync" { - t.Fatalf("setup sync = %+v, %v", res, err) - } - for i := 0; i < lost; i++ { - if err := os.Remove(filepath.Join(env.remoteDir, "world", fmt.Sprintf("map_%04d.bin", i))); err != nil { - t.Fatal(err) - } - } - return env -} - -func TestMassDeletion_IsHeldForADecision(t *testing.T) { - env := convergedThenPeerLoses(t, 1000, 500) - res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) - if err != nil { - t.Fatal(err) - } - if res.Status != "conflict" { - t.Fatalf("result = %+v, want the sync held as a conflict", res) - } - if n := countFiles(t, env.localDir); n != 1000 { - t.Fatalf("this side has %d files; the 500 deletions were applied instead of held", n) - } - if _, ok := env.engine.ActiveConflicts()["game1"]; !ok { - t.Error("no conflict registered to ask about") - } - - // "Keep mine": the peer's missing files go back to it, nothing is deleted here. - if _, err := env.engine.ResolveConflict(context.Background(), "game1", env.peer.ID, "keep-local"); err != nil { - t.Fatal(err) - } - if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { - t.Fatal(err) - } - if n := countFiles(t, env.localDir); n != 1000 { - t.Fatalf("after keep-mine this side has %d files, want 1000", n) - } -} - -// Ordinary play deletes a few files; that still just syncs. -func TestMassDeletion_SmallDeletionsStillSync(t *testing.T) { - env := convergedThenPeerLoses(t, 1000, 20) - res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) - if err != nil { - t.Fatal(err) - } - if res.Status == "conflict" { - t.Fatalf("20 of 1000 deletions were held: %+v", res) - } - if n := countFiles(t, env.localDir); n != 980 { - t.Fatalf("this side has %d files, want the peer's 20 deletions applied (980)", n) - } -} diff --git a/internal/p2p/syncengine/quick.go b/internal/p2p/syncengine/quick.go index 435eba66..74f5b5f2 100644 --- a/internal/p2p/syncengine/quick.go +++ b/internal/p2p/syncengine/quick.go @@ -58,6 +58,16 @@ func (e *Engine) quickInSync(ctx context.Context, gameID string, game store.Game // exchange decides. return Result{}, false, nil } + // The same files; the versions still have to say the same (version.go). + if resp.Version != nil && game.AutoSync { + res, final := e.quickVersionCheck(game, peer, local, *resp.Version) + if !final { + return Result{}, false, nil + } + if res.Status == versionStatusWaiting { + return res, true, nil + } + } // Same as the in-sync outcome of a full sync, minus what is already // recorded: the base is unchanged and the lineage with it. diff --git a/internal/p2p/syncengine/transport.go b/internal/p2p/syncengine/transport.go index 1db8c5df..ed8f8466 100644 --- a/internal/p2p/syncengine/transport.go +++ b/internal/p2p/syncengine/transport.go @@ -96,6 +96,10 @@ type ManifestResponse struct { // question and sends the full manifest, which is handled as before. Unchanged bool `json:"unchanged,omitempty"` ManifestHash string `json:"manifestHash,omitempty"` + + // Version is the responder's version of the save (version.go), absent + // from builds that predate it and for games it does not version. + Version *VersionInfo `json:"version,omitempty"` } // FileRef identifies one file inside one of a game's save locations. diff --git a/internal/p2p/syncengine/version.go b/internal/p2p/syncengine/version.go new file mode 100644 index 00000000..684f0c09 --- /dev/null +++ b/internal/p2p/syncengine/version.go @@ -0,0 +1,827 @@ +package syncengine + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "fmt" + "sort" + "strings" + "time" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/ignore" + "github.com/opensave/opensave/internal/store" +) + +// Save versions. +// +// Every device knows which version of a game's save it holds, as a version +// vector: one counter per device that has ever changed the save, moved only +// when the save really changes there (the watcher saw the game or the user +// write it — never a sync). Comparing two vectors answers, exactly, the +// question the file-by-file comparison could only guess at from what each side +// happens to hold: +// +// - one is older than the other: that device is simply behind. It takes the +// newer save as a whole — every file, and every deletion — and only from a +// device that holds it completely. Nothing is ever taken from it, pushed +// by it, or deleted on its say-so, and it is a source for nobody. +// - equal: the same save. +// - neither: both changed it independently. Only then is anyone asked. +// +// A device also remembers the newest version it has heard of (the target) +// until it holds it. That is what keeps two outdated devices from syncing with +// each other: each knows something newer exists, so each waits for a device +// that holds it, instead of reading the other's half-finished copy as the save +// and passing on what it lacks as deletions. +// +// An empty folder holds no version at all, which is older than every version, +// so a new or emptied device fills itself from a complete one and never raises +// a conflict. A save that existed before this was introduced gets a version +// named after its content (genesisKey), so devices that already held the same +// save agree at once, and ones that did not are compared the way they always +// were, until they agree. + +// VersionVector is writer key → counter. +type VersionVector map[string]int64 + +type versionOrder int + +const ( + versionEqual versionOrder = iota + versionOlder + versionNewer + versionConcurrent +) + +// compareVersions says how a relates to b. +func compareVersions(a, b VersionVector) versionOrder { + aBehind, bBehind := false, false + for k, bv := range b { + if a[k] < bv { + aBehind = true + break + } + } + for k, av := range a { + if b[k] < av { + bBehind = true + break + } + } + switch { + case !aBehind && !bBehind: + return versionEqual + case aBehind && !bBehind: + return versionOlder + case !aBehind && bBehind: + return versionNewer + default: + return versionConcurrent + } +} + +// covers reports whether a is b or newer. +func covers(a, b VersionVector) bool { + o := compareVersions(a, b) + return o == versionEqual || o == versionNewer +} + +func mergeVersions(a, b VersionVector) VersionVector { + out := make(VersionVector, len(a)+len(b)) + for k, v := range a { + out[k] = v + } + for k, v := range b { + if v > out[k] { + out[k] = v + } + } + return out +} + +func (v VersionVector) clone() VersionVector { return mergeVersions(v, nil) } + +// genesisOnly reports whether every entry names a save that predates version +// tracking (genesisKey), i.e. nobody has changed it since. +func (v VersionVector) genesisOnly() bool { + if len(v) == 0 { + return false + } + for k := range v { + if !strings.HasPrefix(k, "g:") { + return false + } + } + return true +} + +// String is a short, stable form for the log. +func (v VersionVector) String() string { + if len(v) == 0 { + return "none" + } + keys := make([]string, 0, len(v)) + for k := range v { + keys = append(keys, k) + } + sort.Strings(keys) + parts := make([]string, len(keys)) + for i, k := range keys { + short := k + if len(short) > 8 { + short = short[:8] + } + parts[i] = fmt.Sprintf("%s=%d", short, v[k]) + } + return strings.Join(parts, ",") +} + +// genesisKey names the version of a save that existed before versions did, +// after its content: devices that hold the same files get the same name. +// Directories are left out — they are created by any sync, and are not what +// makes two saves the same. +func genesisKey(m delta.Manifest) string { + return "g:" + filesKey(m, ignore.Rules{})[:16] +} + +// filesKey hashes which files a save holds and what is in them: paths and +// content hashes, nothing else — not folders, not times. Excluded paths are +// left out. +func filesKey(m delta.Manifest, rules ignore.Rules) string { + paths := make([]string, 0, len(m.Files)) + for p := range m.Files { + if rules.Empty() || !rules.Match(p) { + paths = append(paths, p) + } + } + sort.Strings(paths) + h := sha256.New() + for _, p := range paths { + h.Write([]byte(p)) + h.Write([]byte{0}) + h.Write([]byte(m.Files[p].Hash)) + h.Write([]byte{0}) + } + return hex.EncodeToString(h.Sum(nil)) +} + +// versionHashOf is what is recorded with a version: the save's files as this +// device syncs them, tagged with the exclusion rules it was taken under. A +// save that no longer hashes to it was changed here since, whether or not the +// watcher said so. +// +// Filtered, because a file nobody syncs — a device's own config — changing +// is not a new version of the save. Tagged, because writing a rule is not one +// either: it changes the filtered hash on every device at once, and each took +// that as a change of its own, so the next sync asked about a divergence +// nobody made. A hash taken under other rules is re-taken, not compared +// (sameSaveAs). +func (e *Engine) versionHashOf(gameID string, primary delta.Manifest) string { + text := "" + if game, err := e.Store.GetGame(gameID); err == nil { + text = game.SyncIgnore + } + tag := sha256.Sum256([]byte(text)) + return hex.EncodeToString(tag[:4]) + ":" + filesKey(primary, ignore.Parse(text)) +} + +// sameSaveAs reports whether a hash recorded with a version still describes +// the save that now hashes to current: the same, or taken under different +// exclusion rules, which says nothing about the save. +func sameSaveAs(recorded, current string) bool { + if recorded == current { + return true + } + r, _, okR := strings.Cut(recorded, ":") + c, _, okC := strings.Cut(current, ":") + return okR && okC && r != c +} + +// VersionInfo is a device's version of one game, as it travels in a manifest +// response. Absent from devices that predate it, which are synced the way they +// always were. +type VersionInfo struct { + Vector VersionVector `json:"vector"` + // Target is the newer version the device knows of and does not hold yet. + Target VersionVector `json:"target,omitempty"` + // HasFiles: the device holds at least one save file for the game. + HasFiles bool `json:"hasFiles"` + // Answered is the version someone on the device kept its own save over + // in a conflict (Unix ms in AnsweredAt). Resolution is mutual: the device + // holding that version is asked in turn, not overwritten. + Answered VersionVector `json:"answered,omitempty"` + AnsweredAt int64 `json:"answeredAt,omitempty"` +} + +// Outdated reports whether the device knows its save is behind another's. +func (v VersionInfo) Outdated() bool { + return len(v.Target) > 0 && compareVersions(v.Vector, v.Target) == versionOlder +} + +// gameVersion is the stored record with its vectors decoded. +type gameVersion struct { + rec store.GameVersion + vec VersionVector + target VersionVector + answered VersionVector +} + +func (e *Engine) loadVersionLocked(gameID string) (gameVersion, error) { + rec, err := e.Store.GetGameVersion(gameID) + if err != nil { + return gameVersion{}, err + } + gv := gameVersion{rec: rec, vec: VersionVector{}} + _ = json.Unmarshal([]byte(rec.Vector), &gv.vec) + if gv.vec == nil { + gv.vec = VersionVector{} + } + if rec.Target != "" { + _ = json.Unmarshal([]byte(rec.Target), &gv.target) + } + if rec.Answered != "" { + _ = json.Unmarshal([]byte(rec.Answered), &gv.answered) + } + return gv, nil +} + +func (e *Engine) saveVersionLocked(gv gameVersion) error { + if gv.rec.GameID == "" { + return nil + } + raw, _ := json.Marshal(gv.vec) + gv.rec.Vector = string(raw) + gv.rec.Target = "" + if len(gv.target) > 0 { + t, _ := json.Marshal(gv.target) + gv.rec.Target = string(t) + } + gv.rec.Answered = "" + if len(gv.answered) > 0 { + a, _ := json.Marshal(gv.answered) + gv.rec.Answered = string(a) + } else { + gv.rec.AnsweredAt = 0 + } + return e.Store.SaveGameVersion(gv.rec) +} + +// ensureGenesisLocked names a save that has files and no version yet. Not +// while the device is taking a newer version: its folder is then a part-way +// copy of someone else's, and naming that would make it a version of its own. +func (gv *gameVersion) ensureGenesis(local delta.Manifest, hash string) bool { + if len(gv.vec) > 0 || len(local.Files) == 0 || len(gv.target) > 0 || gv.rec.Pulling { + return false + } + gv.vec = VersionVector{genesisKey(local): 1} + gv.rec.Hash = hash + return true +} + +// refreshLocked brings this device's version up to date with its save before +// anyone compares it: a save that predates versions is named, and one that no +// longer hashes to what its version recorded was changed here — by the game, +// whether or not the watcher has said so yet (it waits for the game to finish +// writing, and while a game holds its files open that can be a long time). It +// is a version of its own from now. Without this, a device whose change was +// still unrecorded looked merely behind, and took the other's save over it. +// +// Not while a pull towards a newer version is unfinished: the folder is then a +// part-way copy of someone else's save, not a change made here. Reports +// whether the record changed. +func (e *Engine) refreshLocked(gv *gameVersion, game store.Game, local delta.Manifest) bool { + hash := e.versionHashOf(game.ID, local) + if gv.ensureGenesis(local, hash) { + return true + } + if gv.rec.Pulling || len(gv.vec) == 0 || gv.rec.Hash == "" || gv.rec.Hash == hash { + return false + } + if sameSaveAs(gv.rec.Hash, hash) { + // Taken under other exclusion rules, so it cannot say whether the + // save changed: re-taken under the new ones. A change made at the + // same time still becomes a version — the watcher reports it + // (NoteLocalChange). + gv.rec.Hash = hash + return true + } + e.bumpLocked(gv, game.Name, hash) + return true +} + +// learn takes in what a peer knows: a version newer than this device's +// becomes its target, unless it already knows of one newer still. +func (gv *gameVersion) learn(theirs VersionInfo) { + for _, c := range []VersionVector{theirs.Vector, theirs.Target} { + if len(c) == 0 || compareVersions(gv.vec, c) != versionOlder { + continue + } + if len(gv.target) == 0 || compareVersions(gv.target, c) == versionOlder { + gv.target = c.clone() + } + } + gv.dropReachedTarget() +} + +// dropReachedTarget forgets a target this device's version covers — or has +// gone past it sideways, by changing on its own: that is no longer "behind", +// it is a conflict, and the version comparison says so by itself. +func (gv *gameVersion) dropReachedTarget() { + if len(gv.target) > 0 && compareVersions(gv.vec, gv.target) != versionOlder { + gv.target = nil + } +} + +func (gv gameVersion) info(hasFiles bool) VersionInfo { + v := VersionInfo{Vector: gv.vec.clone(), Target: gv.target.clone(), HasFiles: hasFiles} + if len(gv.answered) > 0 { + v.Answered, v.AnsweredAt = gv.answered.clone(), gv.rec.AnsweredAt + } + return v +} + +// adopt makes this device hold theirs: their version, and their answer to a +// conflict if they carry one — the device holding the version they kept +// theirs over is still to be asked, whoever it meets. +func (gv *gameVersion) adopt(vec VersionVector, theirs VersionInfo) { + gv.vec = vec.clone() + gv.answered, gv.rec.AnsweredAt = nil, 0 + if len(theirs.Answered) > 0 { + gv.answered, gv.rec.AnsweredAt = theirs.Answered.clone(), theirs.AnsweredAt + } + gv.dropReachedTarget() +} + +// LocalVersion is this device's version of a game, for a manifest response. +// manifest is the save being served; a save that predates versions is named +// here if it has not been yet. Nil when the game takes no part in versions: +// one with automatic sync turned off is not watched, so nothing would ever +// move its version — the peer then syncs it the way it always has. +func (e *Engine) LocalVersion(game store.Game, manifest delta.Manifest) *VersionInfo { + if !game.AutoSync { + return nil + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(game.ID) + if err != nil { + return nil + } + if e.refreshLocked(&gv, game, manifest) { + _ = e.saveVersionLocked(gv) + } + v := gv.info(len(manifest.Files) > 0) + return &v +} + +// NoteLocalChange moves this device's version on: the watcher saw the game +// or the user change the save (never a sync — those are told apart by +// internal/owntouch). Called before the change is offered to anyone. +func (e *Engine) NoteLocalChange(gameID string) { + game, err := e.Store.GetGame(gameID) + if err != nil || !game.AutoSync { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + return + } + if gv.rec.Pulling { + // Files a pull towards a newer version wrote — this run's own, still + // arriving, or left by a run that stopped part-way (after a restart + // the watcher reports those as a change: it cannot know who wrote + // them). Not a version: the folder is a part-way copy of someone + // else's, and naming it would hand that copy to every other device. + // The pull finishing is what moves the version. Anything written here + // meanwhile is in the snapshot taken before the pull replaces it. + return + } + m, err := delta.BuildManifest(game.SavePath) + if err != nil { + return + } + hash := e.versionHashOf(gameID, m) + if len(gv.vec) == 0 { + // First change ever seen here, on a save that predates versions or a + // new one. Named after its content, so a device holding the same files + // agrees with it rather than conflicting. + if gv.ensureGenesis(m, hash) { + _ = e.saveVersionLocked(gv) + } + return + } + if hash == gv.rec.Hash { + // Already a version: a sync noticed it first (refreshLocked), or the + // change was only to files nobody syncs. + return + } + // The watcher saw the game or the user write the save: a version, even if + // the exclusion rules changed since the last one (writing a rule raises + // no file event, so that alone never arrives here). + e.bumpLocked(&gv, game.Name, hash) +} + +// bumpLocked gives this device a new version of its own, after hash. +func (e *Engine) bumpLocked(gv *gameVersion, name, hash string) { + gv.rec.Counter++ + gv.vec = gv.vec.clone() + gv.vec[gv.rec.Key] = gv.rec.Counter + gv.rec.Hash = hash + gv.dropReachedTarget() + if err := e.saveVersionLocked(*gv); err == nil { + e.Log("info", fmt.Sprintf("%s changed on this device — version %s", name, gv.vec)) + } +} + +// versionAfterResolution records a conflict's answer as a version. +// +// Keeping the peer's save makes this device hold what both held: newer than +// both, so the peer finds nothing to take and devices behind either take it. +// +// Keeping this device's own is a new version of its own — but deliberately +// not one that includes the peer's: that would make the peer simply behind, +// and its save would be replaced without anyone there being asked. Resolution +// is mutual. The answer is recorded instead (Answered), so this device is not +// asked again about the same version, and the peer, still independent of it, +// asks in turn. Whoever answers last decides (syncByVersion). +func (e *Engine) versionAfterResolution(gameID string, theirs *VersionInfo, keptLocal bool) { + if theirs == nil { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + return + } + if keptLocal { + gv.rec.Counter++ + gv.vec = gv.vec.clone() + gv.vec[gv.rec.Key] = gv.rec.Counter + gv.answered, gv.rec.AnsweredAt = theirs.Vector.clone(), time.Now().UnixMilli() + } else { + gv.vec = mergeVersions(gv.vec, theirs.Vector) + gv.answered, gv.rec.AnsweredAt = nil, 0 + } + gv.rec.Pulling = false // whatever was being taken, the answer replaces it + gv.rec.Hash = "" + if game, err := e.Store.GetGame(gameID); err == nil { + if m, err := delta.BuildManifest(game.SavePath); err == nil { + gv.rec.Hash = e.versionHashOf(gameID, m) + } + } + gv.dropReachedTarget() + _ = e.saveVersionLocked(gv) +} + +// discardLocalVersion is for a "keep theirs" that did not finish: this +// device's own version is given up, and it takes the peer's like any outdated +// device would — whole, and only from a device that holds it. +func (e *Engine) discardLocalVersion(gameID string, theirs *VersionInfo) { + if theirs == nil { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + return + } + gv.vec = VersionVector{} + gv.target = theirs.Vector.clone() + gv.rec.Hash = "" + _ = e.saveVersionLocked(gv) +} + +// versionStatusWaiting: this device or the peer is behind, and the other +// cannot give it the newer version either. Nothing happens; both wait for a +// device that holds it. +const versionStatusWaiting = "waiting" + +// syncByVersion decides a sync from the two devices' versions. handled false +// leaves it to the file comparison: the two hold the same version and the +// same files, or both saves predate versions and have not met since. +// +// local and remote are filtered by the exclusion rules; unfilteredLocal is +// this device's save as it is on disk, which names a version. +func (e *Engine) syncByVersion(ctx context.Context, game store.Game, peer Peer, + local, unfilteredLocal delta.Manifest, remote ManifestResponse) (Result, bool, error) { + + gameID := game.ID + theirs := *remote.Version + + e.versionMu.Lock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + e.versionMu.Unlock() + return Result{}, false, nil + } + before := gv.target.clone() + changed := e.refreshLocked(&gv, game, unfilteredLocal) + gv.learn(theirs) + if changed || compareVersions(before, gv.target) != versionEqual || len(before) != len(gv.target) { + _ = e.saveVersionLocked(gv) + } + mine := gv.info(len(local.Files) > 0) + e.versionMu.Unlock() + + waiting := func(why string) (Result, bool, error) { + e.logOnce(gameID+"|"+peer.ID, why) + return Result{Status: versionStatusWaiting, PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + pushTo := func() (Result, bool, error) { + e.forgetLogOnce(gameID + "|" + peer.ID) + e.Log("info", fmt.Sprintf("%q: %s has an older version (%s, this device %s) — asking it to take this one", + game.Name, peer.Name, theirs.Vector, mine.Vector)) + e.Transport.TriggerPeerPull(peer, gameID) + return Result{Status: "triggered_peer_pull", Direction: "push", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + + switch { + case mine.Outdated(): + if theirs.Outdated() || !covers(theirs.Vector, mine.Target) { + return waiting(fmt.Sprintf("%q: this device is behind (holds %s, newest known %s); %s does not hold that version either — waiting for a device that does", + game.Name, mine.Vector, mine.Target, peer.Name)) + } + return e.pullVersion(ctx, game, peer, local, remote, theirs.Vector) + case theirs.Outdated(): + if covers(mine.Vector, theirs.Target) { + return pushTo() + } + return waiting(fmt.Sprintf("%q: %s is behind and this device does not hold the version it is waiting for either", + game.Name, peer.Name)) + } + + switch compareVersions(mine.Vector, theirs.Vector) { + case versionNewer: + return pushTo() + case versionOlder: + // learn made this device outdated above, so this is only reached if + // the record could not be read back; the peer holds the newer version + // completely, so take it. + return e.pullVersion(ctx, game, peer, local, remote, theirs.Vector) + case versionEqual: + e.forgetLogOnce(gameID + "|" + peer.ID) + if sameFiles(local, remote.Manifest) { + return Result{}, false, nil + } + // One version, different files. Each side checks its own save against + // its version before answering (refreshLocked), so this is a change + // that landed after the peer answered; its sync carries it. + return waiting(fmt.Sprintf("%q: same version as %s but the files differ — waiting for the change to be recorded", + game.Name, peer.Name)) + } + + // Concurrent: both changed independently. + e.forgetLogOnce(gameID + "|" + peer.ID) + if sameFiles(local, remote.Manifest) { + // …into the same save. Both record the union, which each computes the + // same way, and are equal from then on. + e.versionMu.Lock() + if gv, err := e.loadVersionLocked(gameID); err == nil { + gv.vec = mergeVersions(gv.vec, theirs.Vector) + gv.rec.Hash = e.versionHashOf(gameID, unfilteredLocal) + gv.dropReachedTarget() + _ = e.saveVersionLocked(gv) + } + e.versionMu.Unlock() + return Result{}, false, nil + } + if !mine.HasFiles { + // An empty folder is never one side of a conflict: it takes the save. + return e.pullVersion(ctx, game, peer, local, remote, mergeVersions(mine.Vector, theirs.Vector)) + } + if !theirs.HasFiles { + return pushTo() + } + if mine.Vector.genesisOnly() || theirs.Vector.genesisOnly() { + // A save from before versions existed, never compared with this one + // since: the file comparison and the shared history decide, as before. + return Result{}, false, nil + } + + // Someone already answered: here, keeping this device's save over the + // peer's version (or one it has since moved on from), or there, keeping + // theirs over this one's. + iAnswered := len(mine.Answered) > 0 && covers(theirs.Vector, mine.Answered) + theyAnswered := len(theirs.Answered) > 0 && covers(mine.Vector, theirs.Answered) + switch { + case iAnswered && theyAnswered: + // Both were answered, each keeping its own. The later answer stands, + // as the person who gave it saw the other one too; the earlier one's + // save is in the snapshot taken before it is replaced. + // Answers in the same millisecond are settled the same way on both + // devices, or each would keep asking the other to take its own. + if mine.AnsweredAt > theirs.AnsweredAt || + (mine.AnsweredAt == theirs.AnsweredAt && mine.Vector.String() > theirs.Vector.String()) { + return pushTo() + } + e.Log("info", fmt.Sprintf("%q: %s's answer to the conflict came later — taking its version", game.Name, peer.Name)) + return e.pullVersion(ctx, game, peer, local, remote, mergeVersions(mine.Vector, theirs.Vector)) + case iAnswered: + // Answered here; the peer is asked in turn when it syncs. + return pushTo() + } + e.Log("warn", fmt.Sprintf("%q was changed independently on this device (%s) and on %s (%s)", + game.Name, mine.Vector, peer.Name, theirs.Vector)) + e.registerConflict(gameID, peer, local, remote) + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, true, nil +} + +// pullVersion makes this device's save exactly the peer's — every file it +// holds, and none it does not — and only then records the peer's version as +// this device's. Until that, the device stays behind and keeps waiting. +// +// local and remote are filtered by the exclusion rules, so excluded files are +// neither taken nor removed. +func (e *Engine) pullVersion(ctx context.Context, game store.Game, peer Peer, + local delta.Manifest, remote ManifestResponse, adopt VersionVector) (Result, bool, error) { + + gameID := game.ID + if len(remote.Manifest.Files) == 0 && len(local.Files) > 0 && !remote.DeletionConfirmed { + // The newer version is an empty folder nobody has confirmed emptying. + e.Log("info", fmt.Sprintf("%q holds none of %q's save files now, and has not confirmed deleting them — keeping this device's copies", + peer.Name, game.Name)) + return Result{Status: "peer_holding", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + + var d Decision + for p, lf := range local.Files { + rf, ok := remote.Manifest.Files[p] + switch { + case !ok: + d.FilesToDeleteLocally = append(d.FilesToDeleteLocally, p) + case rf.Hash != lf.Hash: + d.FilesToPull = append(d.FilesToPull, p) + } + } + for p := range remote.Manifest.Files { + if _, ok := local.Files[p]; !ok { + d.FilesToPull = append(d.FilesToPull, p) + } + } + localDirs, remoteDirs := toSet(local.Dirs), toSet(remote.Manifest.Dirs) + for dir := range remoteDirs { + if _, ok := localDirs[dir]; !ok { + d.DirsToPull = append(d.DirsToPull, dir) + } + } + for dir := range localDirs { + if _, ok := remoteDirs[dir]; !ok { + d.DirsToDeleteLocally = append(d.DirsToDeleteLocally, dir) + } + } + sort.Strings(d.FilesToPull) + + e.Log("info", fmt.Sprintf("%q: taking %s's newer version (%s): %d file(s) to fetch, %d to remove", + game.Name, peer.Name, adopt, len(d.FilesToPull), len(d.FilesToDeleteLocally))) + + // Whatever this device holds is kept in a snapshot before any of it is + // replaced or removed. + replacing := len(d.FilesToDeleteLocally) + for _, p := range d.FilesToPull { + if _, ok := local.Files[p]; ok { + replacing++ + } + } + if replacing > 0 { + if _, err := e.Snapshots.CreateBeforeReplacing(gameID, fmt.Sprintf("Before taking %s's newer version", peer.Name)); err != nil { + return Result{}, true, fmt.Errorf("refusing to replace %q's files: they could not be snapshotted first: %w", game.Name, err) + } + } + + e.setPulling(gameID, true) + defer e.setPulling(gameID, false) + + applied := e.Writing(gameID) + started := time.Now() + e.applyLocalDeletions(gameID, primaryRootOf(game), d) + if n := len(d.FilesToDeleteLocally); n > 0 { + e.RecordActivity(store.ActivityEvent{GameID: gameID, Kind: store.ActivityDeleted, Device: peer.Name, Files: n}) + e.noteEmptiedByPeer(gameID, started) + } + e.createPulledDirs(gameID, game, d.DirsToPull) + if len(d.FilesToPull) > 0 { + if err := e.pullFiles(ctx, peer, gameID, game, primaryRootOf(game), local, remote, d.FilesToPull); err != nil { + applied() + return Result{}, true, fmt.Errorf("taking %s's version of %q did not finish (it is resumed on the next sync): %w", peer.Name, game.Name, err) + } + } + applied() + + // Only a save that is now exactly the peer's takes its version. + fresh, err := e.ReadManifest(ctx, gameID, game.SavePath) + if err != nil { + return Result{}, true, err + } + freshHash := e.versionHashOf(gameID, fresh) + if rules := e.rulesFor(gameID); !rules.Empty() { + fresh = filterManifest(fresh, rules) + } + if !sameFiles(fresh, remote.Manifest) { + return Result{}, true, fmt.Errorf("taking %s's version of %q did not finish: the save still differs; it is resumed on the next sync", peer.Name, game.Name) + } + + e.versionMu.Lock() + if gv, err := e.loadVersionLocked(gameID); err == nil { + theirs := VersionInfo{} + if remote.Version != nil { + theirs = *remote.Version + } + gv.adopt(adopt, theirs) + gv.rec.Hash = freshHash + gv.rec.Pulling = false + _ = e.saveVersionLocked(gv) + e.Log("success", fmt.Sprintf("%q now holds version %s from %s", game.Name, gv.vec, peer.Name)) + } + e.versionMu.Unlock() + + // The file comparison's own records, so it agrees if it is ever asked. + e.persistLineage(gameID, peer.ID, fresh, remote.Manifest) + _ = e.Store.SetAgreedHash(gameID, peer.ID, remote.Manifest.ManifestHash()) + + e.syncExtraRoots(ctx, gameID, game, peer, remote) + if !d.HasChanges() { + return Result{Status: "in_sync", Direction: "none", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + return Result{Status: "updated", Direction: "pull", PeerID: peer.ID, PeerName: peer.Name}, true, nil +} + +// setPulling marks a pull towards a newer version as running, so the files it +// writes are never taken for a change made here. In memory for this run; in +// the record from the start until the pull has completed, which only adopting +// the version clears — a pull that stopped part-way leaves a folder that is +// neither this device's version nor the peer's, and it must not be read as a +// change made here (refreshLocked, NoteLocalChange) while it waits to resume. +func (e *Engine) setPulling(gameID string, on bool) { + e.versionMu.Lock() + defer e.versionMu.Unlock() + if e.pullingNow == nil { + e.pullingNow = map[string]bool{} + } + if !on { + delete(e.pullingNow, gameID) + return + } + e.pullingNow[gameID] = true + if gv, err := e.loadVersionLocked(gameID); err == nil && gv.rec.GameID != "" && !gv.rec.Pulling { + gv.rec.Pulling = true + _ = e.saveVersionLocked(gv) + } +} + +// quickVersionCheck is the version half of quickInSync: both saves are the +// agreed one, so the question is only whether the versions say the same. +// done reports a final answer; otherwise the full sync decides. +func (e *Engine) quickVersionCheck(game store.Game, peer Peer, local delta.Manifest, theirs VersionInfo) (res Result, done bool) { + e.versionMu.Lock() + gv, err := e.loadVersionLocked(game.ID) + if err != nil || gv.rec.GameID == "" { + e.versionMu.Unlock() + return Result{}, false + } + before := gv.target.clone() + changed := e.refreshLocked(&gv, game, local) + gv.learn(theirs) + if changed || compareVersions(before, gv.target) != versionEqual || len(before) != len(gv.target) { + _ = e.saveVersionLocked(gv) + } + mine := gv.info(len(local.Files) > 0) + e.versionMu.Unlock() + + if compareVersions(mine.Vector, theirs.Vector) != versionEqual { + return Result{}, false + } + if mine.Outdated() || theirs.Outdated() { + // The same save, and at least one of the two knows a newer one exists + // that neither holds: nothing to do between these two. + return Result{Status: versionStatusWaiting, PeerID: peer.ID, PeerName: peer.Name}, true + } + return Result{}, true +} + +// logOnce logs a line the first time it is the answer for key, and not again +// while it stays the answer: waiting is re-decided every reconcile. +func (e *Engine) logOnce(key, msg string) { + e.versionMu.Lock() + if e.loggedOnce == nil { + e.loggedOnce = map[string]string{} + } + same := e.loggedOnce[key] == msg + e.loggedOnce[key] = msg + e.versionMu.Unlock() + if !same { + e.Log("info", msg) + } +} + +func (e *Engine) forgetLogOnce(key string) { + e.versionMu.Lock() + delete(e.loggedOnce, key) + e.versionMu.Unlock() +} diff --git a/internal/p2p/syncengine/version_test.go b/internal/p2p/syncengine/version_test.go new file mode 100644 index 00000000..6f2d7f12 --- /dev/null +++ b/internal/p2p/syncengine/version_test.go @@ -0,0 +1,575 @@ +package syncengine + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "sort" + "sync" + "testing" + "time" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/snapshot" + "github.com/opensave/opensave/internal/store" +) + +func TestCompareVersions(t *testing.T) { + cases := []struct { + a, b VersionVector + want versionOrder + }{ + {VersionVector{}, VersionVector{}, versionEqual}, + {VersionVector{}, VersionVector{"x": 1}, versionOlder}, + {VersionVector{"x": 2}, VersionVector{"x": 1}, versionNewer}, + {VersionVector{"x": 1, "y": 1}, VersionVector{"x": 1, "y": 1}, versionEqual}, + {VersionVector{"x": 2}, VersionVector{"x": 1, "y": 1}, versionConcurrent}, + {VersionVector{"x": 1}, VersionVector{"x": 1, "y": 1}, versionOlder}, + } + for _, c := range cases { + if got := compareVersions(c.a, c.b); got != c.want { + t.Errorf("compare(%v, %v) = %d, want %d", c.a, c.b, got, c.want) + } + } +} + +// --- a network of real engines ------------------------------------------- + +// meshNode is one device: its own store, save folder and engine. +type meshNode struct { + name string + dir string + st *store.Store + eng *Engine + // blockBudget, when >= 0, is how many more files this device serves + // before every further block request fails: a pull that stops part-way. + blockBudget int +} + +func (n *meshNode) peer() Peer { + return Peer{ID: "node_" + n.name, Name: n.name, Address: "127.0.0.1", Port: 1} +} + +type mesh struct { + mu sync.Mutex + nodes map[string]*meshNode // by peer id + offline map[string]bool + // deletesSent counts DeleteRemote calls: the version path never makes one. + deletesSent int +} + +type meshTransport struct { + m *mesh + self *meshNode +} + +var errUnreachable = errors.New("peer unreachable") + +func (t *meshTransport) target(peer Peer) (*meshNode, error) { + t.m.mu.Lock() + defer t.m.mu.Unlock() + n := t.m.nodes[peer.ID] + if n == nil || t.m.offline[peer.ID] { + return nil, errUnreachable + } + return n, nil +} + +// FetchManifest answers the way the real route does: the save, and the +// device's version of it. +func (t *meshTransport) FetchManifest(ctx context.Context, peer Peer, gameID string, q ManifestQuery) (ManifestResponse, error) { + n, err := t.target(peer) + if err != nil { + return ManifestResponse{}, err + } + game, err := n.st.GetGame(gameID) + if err != nil { + return ManifestResponse{}, errors.New("Game not found.") + } + m, err := delta.BuildManifest(n.dir) + if err != nil { + return ManifestResponse{}, err + } + return ManifestResponse{ + Manifest: m, + ActiveBranch: game.ActiveBranch, + DeletionConfirmed: n.eng.DeletionConfirmed(gameID), + Version: n.eng.LocalVersion(game, m), + }, nil +} + +func (t *meshTransport) FetchBlocks(ctx context.Context, peer Peer, ref FileRef, blockIndices []int, blockSize int) ([]BlockData, error) { + n, err := t.target(peer) + if err != nil { + return nil, err + } + t.m.mu.Lock() + if n.blockBudget == 0 { + t.m.mu.Unlock() + return nil, errors.New("connection reset") + } + if n.blockBudget > 0 { + n.blockBudget-- + } + t.m.mu.Unlock() + full := filepath.Join(n.dir, filepath.FromSlash(ref.RelPath)) + entry, err := delta.HashFile(full) + if err != nil { + return nil, err + } + raw, err := os.ReadFile(full) + if err != nil { + return nil, err + } + var out []BlockData + for _, idx := range blockIndices { + if idx >= len(entry.Blocks) { + continue + } + start := idx * entry.BlockSize + end := start + entry.Blocks[idx].Length + if end > len(raw) { + end = len(raw) + } + out = append(out, BlockData{Index: idx, Data: raw[start:end], Length: end - start}) + } + return out, nil +} + +func (t *meshTransport) DeleteRemote(ctx context.Context, peer Peer, ref FileRef) error { + n, err := t.target(peer) + if err != nil { + return err + } + t.m.mu.Lock() + t.m.deletesSent++ + t.m.mu.Unlock() + return os.RemoveAll(filepath.Join(n.dir, filepath.FromSlash(ref.RelPath))) +} + +func (t *meshTransport) TriggerPeerPull(peer Peer, gameID string) {} +func (t *meshTransport) ReportSyncEvent(peer Peer, gameID, eventType string, data map[string]any) {} + +func newMesh(t *testing.T, names ...string) (*mesh, map[string]*meshNode) { + t.Helper() + m := &mesh{nodes: map[string]*meshNode{}, offline: map[string]bool{}} + byName := map[string]*meshNode{} + for _, name := range names { + root := t.TempDir() + dir := filepath.Join(root, "saves") + if err := os.MkdirAll(dir, 0o777); err != nil { + t.Fatal(err) + } + st, err := store.Open(filepath.Join(root, "opensave.db")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { st.Close() }) + if err := st.EnsureDefaultSettings(root, filepath.Join(root, "backups")); err != nil { + t.Fatal(err) + } + if err := st.CreateGame(store.Game{ID: "game1", Name: "Game One", SavePath: dir, AutoSync: true, MaxSnapshots: 50}); err != nil { + t.Fatal(err) + } + n := &meshNode{name: name, dir: dir, st: st, blockBudget: -1} + n.eng = New(st, snapshot.New(st), &meshTransport{m: m, self: n}) + m.nodes[n.peer().ID] = n + byName[name] = n + } + for _, a := range byName { + for _, b := range byName { + if a != b { + p := b.peer() + if err := a.st.UpsertPeer(store.Peer{ID: p.ID, Name: p.Name, Address: p.Address, Port: p.Port, Status: "online"}); err != nil { + t.Fatal(err) + } + } + } + } + return m, byName +} + +func (m *mesh) setOffline(n *meshNode, off bool) { + m.mu.Lock() + m.offline[n.peer().ID] = off + m.mu.Unlock() +} + +func (m *mesh) deletes() int { + m.mu.Lock() + defer m.mu.Unlock() + return m.deletesSent +} + +// syncPair runs one sync of a with b, as a's engine would. +func syncPair(t *testing.T, a, b *meshNode) (Result, error) { + t.Helper() + return a.eng.SyncWithPeer(context.Background(), "game1", b.peer()) +} + +// change edits the save the way a game does, then tells the engine — the +// watcher's part. +func (n *meshNode) change(t *testing.T, edit func(dir string)) { + t.Helper() + edit(n.dir) + n.eng.NoteLocalChange("game1") +} + +func (n *meshNode) files(t *testing.T) map[string]string { + t.Helper() + m, err := delta.BuildManifest(n.dir) + if err != nil { + t.Fatal(err) + } + out := map[string]string{} + for p, f := range m.Files { + out[p] = f.Hash + } + return out +} + +func (n *meshNode) version(t *testing.T) VersionInfo { + t.Helper() + game, err := n.st.GetGame("game1") + if err != nil { + t.Fatal(err) + } + m, err := delta.BuildManifest(n.dir) + if err != nil { + t.Fatal(err) + } + return *n.eng.LocalVersion(game, m) +} + +func sameSave(a, b map[string]string) bool { + if len(a) != len(b) { + return false + } + for p, h := range a { + if b[p] != h { + return false + } + } + return true +} + +func describe(files map[string]string) string { + paths := make([]string, 0, len(files)) + for p := range files { + paths = append(paths, p) + } + sort.Strings(paths) + if len(paths) > 6 { + return fmt.Sprintf("%d files (%v …)", len(paths), paths[:6]) + } + return fmt.Sprintf("%v", paths) +} + +func writeMany(t *testing.T, dir, prefix string, n int) { + t.Helper() + for i := 0; i < n; i++ { + write(t, dir, fmt.Sprintf("%s/chunk_%03d.bin", prefix, i), fmt.Sprintf("%s-%d", prefix, i)) + } +} + +// settleAll syncs every reachable node with every other, a few rounds, the +// way devices that are all online settle. +func settleAll(t *testing.T, nodes ...*meshNode) { + t.Helper() + for round := 0; round < 4; round++ { + for _, a := range nodes { + for _, b := range nodes { + if a != b { + _, _ = syncPair(t, a, b) + } + } + } + } +} + +// --- scenarios --------------------------------------------------------------- + +// The case this exists for. Three devices share a save; the server changes it +// a lot. One laptop starts taking the new version and is cut off part-way; the +// server then goes offline. The two laptops must not sync with each other in +// the meantime — not the half-finished copy, and not its missing files read as +// deletions — and when the server is back, both end up with its save exactly. +func TestVersions_IncompleteDevicesWaitForTheNewestOne(t *testing.T) { + m, n := newMesh(t, "server", "laptop", "desk") + server, laptop, desk := n["server"], n["laptop"], n["desk"] + + writeMany(t, server.dir, "map", 40) + write(t, server.dir, "players.db", "v1") + settleAll(t, server, laptop, desk) + original := server.files(t) + for _, d := range []*meshNode{laptop, desk} { + if !sameSave(d.files(t), original) { + t.Fatalf("%s did not take the initial save: %s", d.name, describe(d.files(t))) + } + } + + // The desk is switched off for a while; the server plays and the laptop + // takes that. The desk is now two versions behind, the laptop one — + // which makes the laptop newer than the desk, the comparison a device + // part-way through a pull must never win. + m.setOffline(desk, true) + server.change(t, func(dir string) { write(t, dir, "players.db", "v1.5") }) + settleAll(t, server, laptop) + m.setOffline(desk, false) + + // The server plays on: new chunks, some removed, one changed. + server.change(t, func(dir string) { + writeMany(t, dir, "map2", 30) + for i := 0; i < 5; i++ { + _ = os.Remove(filepath.Join(dir, "map", fmt.Sprintf("chunk_%03d.bin", i))) + } + write(t, dir, "players.db", "v2") + }) + newest := server.files(t) + + // The laptop starts taking it and is cut off after 10 files. + server.blockBudget = 10 + if _, err := syncPair(t, laptop, server); err == nil { + t.Fatal("a pull cut off part-way reported success") + } + server.blockBudget = -1 + if !laptop.version(t).Outdated() { + t.Fatalf("the laptop does not know it is behind: %+v", laptop.version(t)) + } + m.setOffline(server, true) + + // Server gone. The laptop and desk meet, repeatedly, both ways. + deskBefore := desk.files(t) + laptopBefore := laptop.files(t) + for i := 0; i < 3; i++ { + for _, pair := range [][2]*meshNode{{desk, laptop}, {laptop, desk}} { + res, err := syncPair(t, pair[0], pair[1]) + if err != nil { + t.Fatalf("%s↔%s: %v", pair[0].name, pair[1].name, err) + } + if res.Status == "conflict" { + t.Fatalf("%s↔%s raised a conflict between two copies of the same history", pair[0].name, pair[1].name) + } + } + } + if !sameSave(desk.files(t), deskBefore) { + t.Errorf("the desk changed while the newest device was away: was %s, now %s", + describe(deskBefore), describe(desk.files(t))) + } + if !sameSave(laptop.files(t), laptopBefore) { + t.Errorf("the laptop's partial copy changed while the newest device was away: was %s, now %s", + describe(laptopBefore), describe(laptop.files(t))) + } + if !desk.version(t).Outdated() { + t.Error("the desk never learned from the laptop that a newer version exists") + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent to another device", m.deletes()) + } + + // The server returns; everyone ends with exactly its save. + m.setOffline(server, false) + settleAll(t, server, laptop, desk) + for _, d := range []*meshNode{server, laptop, desk} { + if !sameSave(d.files(t), newest) { + t.Errorf("%s does not hold the newest save: %s", d.name, describe(d.files(t))) + } + if v := d.version(t); v.Outdated() { + t.Errorf("%s still thinks it is behind: %+v", d.name, v) + } + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent to another device", m.deletes()) + } +} + +// An empty folder — a new device, or one whose copy never arrived — is never +// one side of a conflict, and never makes the device with the save lose files. +func TestVersions_AnEmptyDeviceTakesTheSaveWithoutAsking(t *testing.T) { + m, n := newMesh(t, "full", "empty") + full, empty := n["full"], n["empty"] + writeMany(t, full.dir, "map", 100) + want := full.files(t) + + // The device with the save syncs first: nothing of its own may go. + res, err := syncPair(t, full, empty) + if err != nil { + t.Fatal(err) + } + if res.Status == "conflict" { + t.Fatal("a save against an empty folder raised a conflict") + } + if !sameSave(full.files(t), want) { + t.Fatalf("the device with the save lost files to an empty one: %s", describe(full.files(t))) + } + if res, err := syncPair(t, empty, full); err != nil || res.Status == "conflict" { + t.Fatalf("the empty device's sync: %+v, %v", res, err) + } + if !sameSave(empty.files(t), want) { + t.Errorf("the empty device did not take the save: %s", describe(empty.files(t))) + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// A save that existed before versions did, identical on two devices, is one +// version at once: no conflict, nothing moved. +func TestVersions_IdenticalSavesFromBeforeVersionsAgree(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + writeMany(t, a.dir, "s", 5) + writeMany(t, b.dir, "s", 5) + res, err := syncPair(t, a, b) + if err != nil || res.Status != "in_sync" { + t.Fatalf("identical saves: %+v, %v", res, err) + } + if compareVersions(a.version(t).Vector, b.version(t).Vector) != versionEqual { + t.Errorf("identical saves got different versions: %v vs %v", a.version(t).Vector, b.version(t).Vector) + } +} + +// A deletion made in the game is part of the new version, and reaches the +// other device as that — whole — however many files it is. +func TestVersions_DeletionsTravelWithTheVersion(t *testing.T) { + m, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + writeMany(t, a.dir, "map", 20) + settleAll(t, a, b) + a.change(t, func(dir string) { + _ = os.RemoveAll(filepath.Join(dir, "map")) + write(t, dir, "fresh.sav", "new world") + }) + settleAll(t, a, b) + if !sameSave(b.files(t), a.files(t)) { + t.Errorf("b = %s, want %s", describe(b.files(t)), describe(a.files(t))) + } + if m.deletes() != 0 { + t.Errorf("the newer device sent %d deletion(s) itself; the older one takes them", m.deletes()) + } +} + +// Both devices changed the save while apart: that, and only that, is asked +// about. Resolution is mutual — keeping A's on A asks B in turn rather than +// overwriting it, and A is not asked again. Once B agrees, the answer spreads, +// and a third device that had changed nothing takes it without being asked. +func TestVersions_IndependentChangesAskEachSideOnceAndTheAnswerSpreads(t *testing.T) { + _, n := newMesh(t, "a", "b", "c") + a, b, c := n["a"], n["b"], n["c"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b, c) + + a.change(t, func(dir string) { write(t, dir, "slot.sav", "a-played") }) + b.change(t, func(dir string) { write(t, dir, "slot.sav", "b-played") }) + + res, err := syncPair(t, a, b) + if err != nil || res.Status != "conflict" { + t.Fatalf("independent changes: %+v, %v — want a conflict", res, err) + } + if _, err := a.eng.ResolveConflict(context.Background(), "game1", b.peer().ID, "keep-local"); err != nil { + t.Fatal(err) + } + bBefore := b.files(t) + settleAll(t, a, b, c) + if len(a.eng.ActiveConflicts()) != 0 { + t.Error("A was asked again after answering") + } + if len(b.eng.ActiveConflicts()) == 0 { + t.Fatal("B was never asked: A's answer must not decide B's save on its own") + } + if !sameSave(b.files(t), bBefore) { + t.Error("B's save changed before anyone there answered") + } + if _, err := b.eng.ResolveConflict(context.Background(), "game1", a.peer().ID, "keep-remote"); err != nil { + t.Fatal(err) + } + settleAll(t, a, b, c) + for _, d := range []*meshNode{b, c} { + if got := d.files(t)["slot.sav"]; got != a.files(t)["slot.sav"] { + t.Errorf("%s did not take the kept version", d.name) + } + if len(d.eng.ActiveConflicts()) != 0 { + t.Errorf("%s still has a conflict after both answered", d.name) + } + } +} + +// Both sides answer "keep mine" (or "keep both"): the later answer stands and +// both end on one save, instead of asking each other forever. +func TestVersions_TwoAnswersSettleOnTheLater(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b) + a.change(t, func(dir string) { write(t, dir, "slot.sav", "a-played") }) + b.change(t, func(dir string) { write(t, dir, "slot.sav", "b-played") }) + if res, _ := syncPair(t, a, b); res.Status != "conflict" { + t.Fatalf("A: %+v, want a conflict", res) + } + if res, _ := syncPair(t, b, a); res.Status != "conflict" { + t.Fatalf("B: %+v, want a conflict", res) + } + if _, err := a.eng.ResolveConflict(context.Background(), "game1", b.peer().ID, "keep-local"); err != nil { + t.Fatal(err) + } + time.Sleep(5 * time.Millisecond) + if _, err := b.eng.ResolveConflict(context.Background(), "game1", a.peer().ID, "keep-local"); err != nil { + t.Fatal(err) + } + settleAll(t, a, b) + if got := a.files(t)["slot.sav"]; got != b.files(t)["slot.sav"] { + t.Fatal("the two never settled on one save") + } + if len(a.eng.ActiveConflicts())+len(b.eng.ActiveConflicts()) != 0 { + t.Error("a conflict is still open after both answered") + } + if got, _ := os.ReadFile(filepath.Join(a.dir, "slot.sav")); string(got) != "b-played" { + t.Errorf("A holds %q; B answered last, so its save should stand", got) + } +} + +// A change the watcher never reported — made while the app was closed — still +// becomes a version once the two devices meet, instead of leaving them on one +// version with different files, each waiting for the other. +func TestVersions_AnUnnoticedChangeIsFoundWhenTheDevicesMeet(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b) + write(t, a.dir, "slot.sav", "played offline") // no NoteLocalChange + settleAll(t, b, a) + if got := b.files(t)["slot.sav"]; got != a.files(t)["slot.sav"] { + t.Errorf("the unnoticed change never reached b") + } +} + +// Files a sync writes are not a change made here: the pull-in-progress record +// keeps them from becoming a version, even across a restart. +func TestVersions_PulledFilesAreNotANewVersion(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + writeMany(t, a.dir, "map", 10) + settleAll(t, a, b) + before := b.version(t).Vector + + a.change(t, func(dir string) { writeMany(t, dir, "map3", 10) }) + a.blockBudget = 3 + _, _ = syncPair(t, b, a) + a.blockBudget = -1 + + // What a restart leaves: the record says a pull was running, and the + // watcher reports the half-written folder as a change. + rec, _ := b.st.GetGameVersion("game1") + rec.Pulling = true + _ = b.st.SaveGameVersion(rec) + b.eng.NoteLocalChange("game1") + if got := b.version(t).Vector; compareVersions(got, before) != versionEqual { + t.Fatalf("a half-finished pull became a version of its own: %v (was %v)", got, before) + } + settleAll(t, a, b) + if !sameSave(b.files(t), a.files(t)) { + t.Errorf("b did not finish taking a's version: %s", describe(b.files(t))) + } +} diff --git a/internal/store/migrations/0039_game_versions.sql b/internal/store/migrations/0039_game_versions.sql new file mode 100644 index 00000000..3c0c2ce1 --- /dev/null +++ b/internal/store/migrations/0039_game_versions.sql @@ -0,0 +1,42 @@ +-- Which version of a game's save this device holds, as a version vector, and +-- whether it knows of a newer one it has not finished taking yet (see +-- internal/p2p/syncengine/version.go). +-- +-- vector is a JSON object of writer key to counter. A writer key belongs to +-- one device for one game (key below), and its counter moves only when the +-- save really changes there — the game or the user wrote it, never a sync. +-- Comparing two vectors says, without guessing from files or clocks, whether +-- one device is simply behind the other or both changed independently. +-- +-- target is the newest version this device has heard of that is newer than +-- its own, '' when none. While it is set this device is outdated: it takes +-- the save only from a device holding that version, and is a source for +-- nobody — its files are not handed out, and what it lacks is not read as +-- deleted. +-- +-- counter is this device's own counter for the game. It never goes back, even +-- when the device adopts another's version, so the same key and number never +-- name two different saves. +-- +-- pulling is set while a pull towards target is under way, so the files it +-- writes are not mistaken for a new local version if the app stops half-way. +-- +-- hash is the save's content hash when vector was last set ('' unknown). Two +-- devices on one version holding different files means a change went +-- unnoticed on one of them; this says which. +-- +-- answered is the other device's version this one kept its own save over when +-- someone answered a conflict here ('' none), and answered_at when (Unix ms). +-- It travels with the version, so the other device is asked in turn rather +-- than overwritten, and this one is not asked again. +CREATE TABLE game_versions ( + game_id TEXT PRIMARY KEY REFERENCES games(id) ON DELETE CASCADE, + key TEXT NOT NULL, + counter INTEGER NOT NULL DEFAULT 0, + vector TEXT NOT NULL DEFAULT '{}', + target TEXT NOT NULL DEFAULT '', + pulling INTEGER NOT NULL DEFAULT 0, + hash TEXT NOT NULL DEFAULT '', + answered TEXT NOT NULL DEFAULT '', + answered_at INTEGER NOT NULL DEFAULT 0 +); diff --git a/internal/store/versions.go b/internal/store/versions.go new file mode 100644 index 00000000..3803db67 --- /dev/null +++ b/internal/store/versions.go @@ -0,0 +1,71 @@ +package store + +import ( + "crypto/rand" + "database/sql" + "encoding/hex" + "errors" + "fmt" +) + +// Save versions: which version of a game's save this device holds. See +// migrations/0037_game_versions.sql for the columns and +// internal/p2p/syncengine/version.go for the rules. The vectors are kept as +// the JSON the sync engine writes; this layer does not interpret them. + +// GameVersion is one game's version record on this device. +type GameVersion struct { + GameID string `db:"game_id"` + Key string `db:"key"` + Counter int64 `db:"counter"` + Vector string `db:"vector"` + Target string `db:"target"` + Pulling bool `db:"pulling"` + Hash string `db:"hash"` + // Answered is the JSON vector this device kept its save over in a + // conflict, AnsweredAt when. + Answered string `db:"answered"` + AnsweredAt int64 `db:"answered_at"` +} + +// GetGameVersion returns a game's version record, creating an empty one — +// with a fresh writer key — the first time it is asked for. A game that is not +// tracked gets a record that is not stored. +func (s *Store) GetGameVersion(gameID string) (GameVersion, error) { + var v GameVersion + err := s.db.Get(&v, `SELECT * FROM game_versions WHERE game_id = ?`, gameID) + if err == nil { + return v, nil + } + if !errors.Is(err, sql.ErrNoRows) { + return GameVersion{}, fmt.Errorf("get game version %s: %w", gameID, err) + } + var b [8]byte + if _, err := rand.Read(b[:]); err != nil { + return GameVersion{}, err + } + v = GameVersion{GameID: gameID, Key: hex.EncodeToString(b[:]), Vector: "{}"} + // OR IGNORE: two callers creating it at once keep the first key. + if _, err := s.db.Exec(`INSERT OR IGNORE INTO game_versions (game_id, key, vector) + SELECT ?, ?, '{}' WHERE EXISTS (SELECT 1 FROM games WHERE id = ?)`, + gameID, v.Key, gameID); err != nil { + return GameVersion{}, fmt.Errorf("create game version %s: %w", gameID, err) + } + if err := s.db.Get(&v, `SELECT * FROM game_versions WHERE game_id = ?`, gameID); err != nil && + !errors.Is(err, sql.ErrNoRows) { + return GameVersion{}, fmt.Errorf("get game version %s: %w", gameID, err) + } + return v, nil +} + +// SaveGameVersion writes a game's version record. The writer key is never +// changed by this: it is the record's identity. +func (s *Store) SaveGameVersion(v GameVersion) error { + _, err := s.db.Exec(`UPDATE game_versions SET counter = ?, vector = ?, target = ?, pulling = ?, hash = ?, + answered = ?, answered_at = ? WHERE game_id = ?`, + v.Counter, v.Vector, v.Target, v.Pulling, v.Hash, v.Answered, v.AnsweredAt, v.GameID) + if err != nil { + return fmt.Errorf("save game version %s: %w", v.GameID, err) + } + return nil +} From ad6271ac99e43d8f71a1e8785acadd286f9ce331 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Thu, 1 Oct 2026 22:56:36 +0200 Subject: [PATCH 15/31] sync: settle saves from before versions by their files, or by "Use this save everywhere" Saves that existed before version tracking get a starting version named after their content, and nothing recorded says which device's is newer. Between two of those, the old file-by-file comparison used to decide - and it either asked, or merged the two file by file into a save no game wrote. Now, between saves from before versions: - if the two last agreed on a state and one still holds exactly it, the other is newer; - otherwise if one is newer in every file that differs (newer, or missing on the other side) and the other holds no file written after its newest, that one is newer; - the older device takes the newer save whole, never file by file. The side that gives way was older in every differing file, so it loses nothing newer than what it takes. - anything else is asked about. "Use this save everywhere" (game header, POST /api/games/{id}/use-everywhere) is for states nobody can rank - half-arrived copies, mixtures. The device's version then covers every save from before versions (entry "g:*"), so every such device takes it whole, keeping a snapshot first. A change made on another device since versions existed is not overruled: that is a conflict. An open conflict no longer blocks a game for good once a version newer than both sides of it arrives: it was settled elsewhere, and is cleared. --- .../frontend/src/views/game/GameHeader.svelte | 16 ++ internal/api/routes.go | 23 +++ internal/p2p/syncengine/engine.go | 18 +- internal/p2p/syncengine/version.go | 185 +++++++++++++++++- internal/p2p/syncengine/version_test.go | 151 ++++++++++++++ 5 files changed, 384 insertions(+), 9 deletions(-) diff --git a/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte b/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte index 8eb05a2e..ceafac2a 100644 --- a/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte +++ b/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte @@ -13,6 +13,7 @@ import ArrowLeft from 'lucide-svelte/icons/arrow-left'; import Play from 'lucide-svelte/icons/play'; import RefreshCw from 'lucide-svelte/icons/refresh-cw'; + import Upload from 'lucide-svelte/icons/upload'; import FolderOpen from 'lucide-svelte/icons/folder-open'; import Pencil from 'lucide-svelte/icons/pencil'; import Star from 'lucide-svelte/icons/star'; @@ -43,6 +44,20 @@ const syncNow = () => run('Sync triggered', () => api.post(`/api/games/${game.id}/sync`)); const launchGame = () => run('Launching…', () => api.post(`/api/games/${game.id}/launch`)); + // For when the devices hold states nobody can rank — a copy that arrived + // only half-way, a mixture of two saves — and the person knows which device + // has the right one. Every other device takes this save whole, keeping a + // snapshot of its own first; one that changed the save on its own since is + // still asked. + async function useEverywhere() { + const ok = await askConfirm( + `Make this device's save of ${game.name} the one every other device uses? They each replace theirs with it — every file, including removing files this device does not have — and keep a snapshot of their own first. A device where the save was changed after this update is asked instead.`, + { title: 'Use this save everywhere?', confirmText: 'Use this save everywhere', danger: true } + ); + if (!ok) return; + await run(`${game.name}: this save goes to your other devices`, () => api.post(`/api/games/${game.id}/use-everywhere`)); + } + let editPath = false; let pathDraft = ''; @@ -138,6 +153,7 @@ {#if canLaunch(game)} {/if} + diff --git a/internal/api/routes.go b/internal/api/routes.go index 65a28d0c..fbac46fe 100644 --- a/internal/api/routes.go +++ b/internal/api/routes.go @@ -1,6 +1,7 @@ package api import ( + "context" "encoding/json" "errors" "fmt" @@ -49,6 +50,7 @@ func (s *Server) routes(r chi.Router) { r.Post("/api/games/{gameId}/snapshot", s.handleCreateSnapshot) r.Post("/api/games/{gameId}/rollback", s.handleRollback) + r.Post("/api/games/{gameId}/use-everywhere", s.handleUseSaveEverywhere) r.Get("/api/games/{gameId}/save-files", s.handleGameSaveFiles) r.Get("/api/games/{gameId}/snapshot/{snapshotId}/files", s.handleSnapshotFiles) r.Post("/api/games/{gameId}/snapshot/{snapshotId}/restore-file", s.handleRestoreFile) @@ -637,6 +639,27 @@ func (s *Server) handleRollback(w http.ResponseWriter, r *http.Request) { writeJSON(w, http.StatusOK, snap) } +// handleUseSaveEverywhere makes this device's save the one every other device +// takes (syncengine.UseSaveEverywhere), and starts a sync to hand it out. The +// sync runs in the background: a large save takes a while, and the answer the +// person needs is that it was accepted. A paused device hands it out when it +// resumes. +func (s *Server) handleUseSaveEverywhere(w http.ResponseWriter, r *http.Request) { + gameID := chi.URLParam(r, "gameId") + vec, err := s.Daemon.P2P.Sync.UseSaveEverywhere(gameID) + if err != nil { + writeError(w, http.StatusConflict, err.Error()) + return + } + s.Daemon.P2P.GoSync(func(ctx context.Context) { + ctx, cancel := context.WithTimeout(ctx, 30*time.Minute) + defer cancel() + _, _ = s.Daemon.P2P.SyncGame(ctx, gameID) + }) + s.BroadcastGamesUpdate() + writeJSON(w, http.StatusOK, map[string]any{"version": vec.String()}) +} + func (s *Server) handleCreateBranch(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") // copyCurrentSave defaults to true when the field is absent: a caller that diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index 2c10af0f..b7109514 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -563,13 +563,21 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re remoteData.Manifest = filterManifest(remoteData.Manifest, ignoreRules) } - // 3. Existing unresolved conflict blocks further syncing. + // 3. Existing unresolved conflict blocks further syncing — unless a + // version newer than both sides of it has turned up since (version.go): + // it was settled elsewhere, by an answer or "Use this save everywhere", + // and the question here no longer stands. e.mu.Lock() - if existing := e.activeConflicts[gameID]; existing != nil { - e.mu.Unlock() - return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, nil - } + existing := e.activeConflicts[gameID] e.mu.Unlock() + if existing != nil { + if remoteData.Version == nil || !game.AutoSync || !e.supersedes(gameID, *remoteData.Version) { + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, nil + } + e.Log("info", fmt.Sprintf("%q: %s holds a version newer than both sides of the open conflict — taking it instead of asking", + game.Name, peer.Name)) + e.clearConflict(gameID) + } // 3b. The versions decide, when both devices keep them (version.go): who // is behind takes the newer save whole, from a device that holds it, and diff --git a/internal/p2p/syncengine/version.go b/internal/p2p/syncengine/version.go index 684f0c09..362306c4 100644 --- a/internal/p2p/syncengine/version.go +++ b/internal/p2p/syncengine/version.go @@ -5,6 +5,7 @@ import ( "crypto/sha256" "encoding/hex" "encoding/json" + "errors" "fmt" "sort" "strings" @@ -56,17 +57,30 @@ const ( versionConcurrent ) +// genesisAll is the entry "Use this save everywhere" adds (UseSaveEverywhere): +// a version holding it counts as having every save from before versions +// existed behind it, whatever its content was named. +const genesisAll = "g:*" + +// holds reports whether v includes entry k at counter n. +func holds(v VersionVector, k string, n int64) bool { + if v[k] >= n { + return true + } + return k != genesisAll && strings.HasPrefix(k, "g:") && v[genesisAll] >= 1 +} + // compareVersions says how a relates to b. func compareVersions(a, b VersionVector) versionOrder { aBehind, bBehind := false, false for k, bv := range b { - if a[k] < bv { + if !holds(a, k, bv) { aBehind = true break } } for k, av := range a { - if b[k] < av { + if !holds(b, k, av) { bBehind = true break } @@ -601,8 +615,24 @@ func (e *Engine) syncByVersion(ctx context.Context, game store.Game, peer Peer, } if mine.Vector.genesisOnly() || theirs.Vector.genesisOnly() { // A save from before versions existed, never compared with this one - // since: the file comparison and the shared history decide, as before. - return Result{}, false, nil + // since: nothing recorded says which is newer. The files may, when + // they say it unmistakably (transitionNewer); then the older device + // takes the newer save whole, as any device behind does. Never file by + // file, which is how devices ended up holding mixtures of two saves + // that no game ever wrote. When the files do not say, the person is + // asked — or uses "Use this save everywhere" on the device that has + // the right one. + switch transitionNewer(local, remote.Manifest, e.Store.GetAgreedHash(gameID, peer.ID)) { + case 1: + return pushTo() + case -1: + e.Log("info", fmt.Sprintf("%q: %s's save is the newer one in every file that differs — taking it whole", + game.Name, peer.Name)) + return e.pullVersion(ctx, game, peer, local, remote, theirs.Vector) + } + e.Log("warn", fmt.Sprintf("%q differs from %s's and neither is clearly the newer — asking", game.Name, peer.Name)) + e.registerConflict(gameID, peer, local, remote) + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, true, nil } // Someone already answered: here, keeping this device's save over the @@ -825,3 +855,150 @@ func (e *Engine) forgetLogOnce(key string) { delete(e.loggedOnce, key) e.versionMu.Unlock() } + +// transitionNewer says which of two saves from before versions is the newer, +// when that is beyond doubt: 1 for local, -1 for remote, 0 when it is not. +// +// Beyond doubt means one of two things. The devices last agreed on a state +// and one of them still holds exactly it: that one has not changed since, so +// the other is newer. Or one save is newer in every file that differs — each +// is newer there or missing on the other side — and the other holds no file +// of its own written after the first's newest. That is what a save someone +// played on looks like next to one nobody touched. +// +// It is deliberately not "the newer file wins, file by file": that makes a +// mixture of two saves no game wrote. And not "the newest file wins", which a +// mixture passes as easily as a real save. Here the device that gives way was +// older in every file that differs, so it never loses anything newer than +// what it takes — at worst it takes a copy that arrived only half-way, and +// then the complete save, newer than both, replaces that too when it is +// seen. States newer each in different files fail the test both ways and are +// left to a person. +func transitionNewer(local, remote delta.Manifest, agreed string) int { + if agreed != "" { + lh, rh := local.ManifestHash(), remote.ManifestHash() + switch { + case lh == agreed && rh != agreed: + return -1 + case rh == agreed && lh != agreed: + return 1 + } + } + l, r := newerInEveryFile(local, remote), newerInEveryFile(remote, local) + switch { + case l && !r: + return 1 + case r && !l: + return -1 + } + return 0 +} + +// newerInEveryFile reports whether a is newer than b wherever they differ, +// and they do differ. +func newerInEveryFile(a, b delta.Manifest) bool { + differ := false + var aNewest delta.Milli + for p, af := range a.Files { + if af.MtimeMs > aNewest { + aNewest = af.MtimeMs + } + bf, ok := b.Files[p] + if !ok { + differ = true + continue + } + if af.Hash != bf.Hash { + differ = true + if af.MtimeMs <= bf.MtimeMs { + return false + } + } + } + for p, bf := range b.Files { + if _, ok := a.Files[p]; ok { + continue + } + differ = true + if bf.MtimeMs >= aNewest { + return false + } + } + return differ +} + +// supersedes reports whether theirs is newer than this device's version and +// than the version the open conflict is with: both sides of the question are +// behind it. +func (e *Engine) supersedes(gameID string, theirs VersionInfo) bool { + e.mu.Lock() + c := e.activeConflicts[gameID] + e.mu.Unlock() + if c == nil || c.remoteVersion == nil { + return false + } + e.versionMu.Lock() + gv, err := e.loadVersionLocked(gameID) + e.versionMu.Unlock() + if err != nil || len(gv.vec) == 0 { + return false + } + return compareVersions(gv.vec, theirs.Vector) == versionOlder && + compareVersions(c.remoteVersion.Vector, theirs.Vector) == versionOlder +} + +// UseSaveEverywhere makes this device's save of a game the one every other +// device takes: its version becomes newer than every save from before +// versions existed, on any device, so each of those takes this one whole, +// keeping a snapshot of its own first. For when several devices hold states +// nobody can rank — copies that arrived half-way, mixtures of two saves — +// and a person knows which device has the right one. +// +// A change made on another device since versions existed is not overruled: +// that device and this one have each changed the save independently, and +// the person there is asked, as for any conflict. +func (e *Engine) UseSaveEverywhere(gameID string) (VersionVector, error) { + game, err := e.Store.GetGame(gameID) + if err != nil { + return nil, err + } + if !game.AutoSync { + return nil, errors.New("automatic sync is off for this game, so it does not take part in versions") + } + m, err := delta.BuildManifest(game.SavePath) + if err != nil { + return nil, err + } + if len(m.Files) == 0 { + return nil, errors.New("this device holds no save files for this game") + } + e.versionMu.Lock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + e.versionMu.Unlock() + if err == nil { + err = errors.New("the game is not tracked") + } + return nil, err + } + gv.vec = mergeVersions(gv.vec, VersionVector{genesisAll: 1}) + gv.rec.Counter++ + gv.vec[gv.rec.Key] = gv.rec.Counter + gv.rec.Hash = e.versionHashOf(gameID, m) + gv.target, gv.answered, gv.rec.AnsweredAt = nil, nil, 0 + gv.rec.Pulling = false + saveErr := e.saveVersionLocked(gv) + vec := gv.vec.clone() + e.versionMu.Unlock() + if saveErr != nil { + return nil, saveErr + } + e.mu.Lock() + _, conflicted := e.activeConflicts[gameID] + e.mu.Unlock() + if conflicted { + e.clearConflict(gameID) + } + e.Log("info", fmt.Sprintf("%q: this device's save is to be used everywhere — version %s", game.Name, vec)) + return vec, nil +} diff --git a/internal/p2p/syncengine/version_test.go b/internal/p2p/syncengine/version_test.go index 6f2d7f12..18ae6171 100644 --- a/internal/p2p/syncengine/version_test.go +++ b/internal/p2p/syncengine/version_test.go @@ -35,6 +35,157 @@ func TestCompareVersions(t *testing.T) { } } +// "Use this save everywhere" covers every save from before versions, not +// changes made since. +func TestCompareVersions_UseEverywhereCoversEveryOldSave(t *testing.T) { + everywhere := VersionVector{"g:aaaa": 1, genesisAll: 1, "x": 1} + if got := compareVersions(VersionVector{"g:bbbb": 1}, everywhere); got != versionOlder { + t.Errorf("an old save of other content vs everywhere = %d, want older", got) + } + if got := compareVersions(VersionVector{}, everywhere); got != versionOlder { + t.Errorf("an empty folder vs everywhere = %d, want older", got) + } + if got := compareVersions(VersionVector{"g:bbbb": 1, "y": 1}, everywhere); got != versionConcurrent { + t.Errorf("a save changed since versions vs everywhere = %d, want concurrent", got) + } +} + +func setTimes(t *testing.T, dir, rel string, at time.Time) { + t.Helper() + if err := os.Chtimes(filepath.Join(dir, filepath.FromSlash(rel)), at, at); err != nil { + t.Fatal(err) + } +} + +// Two saves from before versions, one played on since they last matched: +// that one is newer in every file that differs, and the other takes it whole +// — no question, and no mixture. +func TestVersions_TransitionTakesTheClearlyNewerSave(t *testing.T) { + m, n := newMesh(t, "played", "old") + played, old := n["played"], n["old"] + t0 := time.Now().Add(-48 * time.Hour) + t1 := time.Now().Add(-1 * time.Hour) + for _, d := range []*meshNode{played, old} { + write(t, d.dir, "slot1.sav", "day one") + write(t, d.dir, "meta.dat", "m1") + write(t, d.dir, "stale.tmp", "gone later") + setTimes(t, d.dir, "slot1.sav", t0) + setTimes(t, d.dir, "meta.dat", t0) + setTimes(t, d.dir, "stale.tmp", t0) + } + // Played on, before the update: changed, added, removed. + write(t, played.dir, "slot1.sav", "day five") + write(t, played.dir, "slot2.sav", "new slot") + _ = os.Remove(filepath.Join(played.dir, "stale.tmp")) + setTimes(t, played.dir, "slot1.sav", t1) + setTimes(t, played.dir, "slot2.sav", t1) + + for _, pair := range [][2]*meshNode{{old, played}, {played, old}} { + res, err := syncPair(t, pair[0], pair[1]) + if err != nil || res.Status == "conflict" { + t.Fatalf("%s↔%s: %+v, %v", pair[0].name, pair[1].name, res, err) + } + } + settleAll(t, played, old) + if !sameSave(old.files(t), played.files(t)) { + t.Errorf("the old device did not take the played save whole: %s", describe(old.files(t))) + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// Newer each in a different file: nothing says which is the save, so the +// person is asked rather than either being taken — or a mixture made. +func TestVersions_TransitionAsksWhenNeitherIsClearlyNewer(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + t0 := time.Now().Add(-48 * time.Hour) + t1 := time.Now().Add(-1 * time.Hour) + write(t, a.dir, "one.sav", "a-new") + write(t, a.dir, "two.sav", "old") + write(t, b.dir, "one.sav", "old") + write(t, b.dir, "two.sav", "b-new") + setTimes(t, a.dir, "one.sav", t1) + setTimes(t, a.dir, "two.sav", t0) + setTimes(t, b.dir, "one.sav", t0) + setTimes(t, b.dir, "two.sav", t1) + aBefore, bBefore := a.files(t), b.files(t) + + res, err := syncPair(t, a, b) + if err != nil || res.Status != "conflict" { + t.Fatalf("mixed old saves: %+v, %v — want a conflict", res, err) + } + if !sameSave(a.files(t), aBefore) || !sameSave(b.files(t), bBefore) { + t.Error("a save changed before anyone answered") + } +} + +// "Use this save everywhere" on the device with the right save: every other +// device takes it whole — even one already waiting on a question about two +// mixed states — and nobody is asked. +func TestVersions_UseThisSaveEverywhere(t *testing.T) { + m, n := newMesh(t, "right", "mixed1", "mixed2") + right, mixed1, mixed2 := n["right"], n["mixed1"], n["mixed2"] + t0 := time.Now().Add(-48 * time.Hour) + t1 := time.Now().Add(-1 * time.Hour) + writeMany(t, right.dir, "map", 12) + write(t, right.dir, "players.db", "current") + // Two mixtures nobody can rank. + writeMany(t, mixed1.dir, "map", 6) + write(t, mixed1.dir, "players.db", "old-1") + write(t, mixed1.dir, "extra1.bin", "x") + writeMany(t, mixed2.dir, "map", 4) + write(t, mixed2.dir, "players.db", "old-2") + write(t, mixed2.dir, "extra2.bin", "y") + setTimes(t, mixed1.dir, "players.db", t1) + setTimes(t, mixed1.dir, "extra1.bin", t0) + setTimes(t, mixed2.dir, "players.db", t0) + // Newer than anything mixed1 holds, so neither is newer in every file. + setTimes(t, mixed2.dir, "extra2.bin", time.Now().Add(time.Hour)) + + if res, _ := syncPair(t, mixed1, mixed2); res.Status != "conflict" { + t.Fatalf("setup: two mixtures should conflict, got %+v", res) + } + if _, err := right.eng.UseSaveEverywhere("game1"); err != nil { + t.Fatal(err) + } + settleAll(t, right, mixed1, mixed2) + want := right.files(t) + for _, d := range []*meshNode{mixed1, mixed2} { + if !sameSave(d.files(t), want) { + t.Errorf("%s did not take the save used everywhere: %s", d.name, describe(d.files(t))) + } + if len(d.eng.ActiveConflicts()) != 0 { + t.Errorf("%s still has a conflict open", d.name) + } + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// A change made on another device after versions existed is not overruled by +// "Use this save everywhere": that is a conflict like any other. +func TestVersions_UseEverywhereDoesNotOverruleALaterChange(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b) + b.change(t, func(dir string) { write(t, dir, "slot.sav", "b-played") }) + if _, err := a.eng.UseSaveEverywhere("game1"); err != nil { + t.Fatal(err) + } + bBefore := b.files(t) + settleAll(t, a, b) + if !sameSave(b.files(t), bBefore) { + t.Error("b's change since the update was replaced without anyone being asked") + } + if len(a.eng.ActiveConflicts())+len(b.eng.ActiveConflicts()) == 0 { + t.Error("nobody was asked") + } +} + // --- a network of real engines ------------------------------------------- // meshNode is one device: its own store, save folder and engine. From c722f55890fe3df2320f1fe90cdff1d6863866b0 Mon Sep 17 00:00:00 2001 From: ChakraFusion <112588066+ChakraFusion@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:29:27 +0200 Subject: [PATCH 16/31] sync: devices on an OpenSave without save versions can no longer do damage The transition to save versions happens one device at a time, and a device still on an older build decides by the old file comparison: whatever its copy lacks it reads as deleted, so a copy that arrived half-way spread its gaps as deletions to the devices that had the whole save. That is how a restored save lost 6,927 files again to a device not yet updated. - Nothing is taken from a peer that does not keep save versions (status peer_outdated_app); it can still take from this device. - Deletions it asks for are refused, on the LAN and over the relay. Requests from builds that keep versions say so ("versioned": true). - The device list marks such a peer "needs update". - The conflict dialog offers "Use this device's save everywhere" for when neither side is the right save. Peers signal that they keep versions with "versions": true in the manifest answer (and LAN protocol revision 3). The relay path now also carries the save's version: until now, relay syncs never got one and always fell back to the old file comparison. --- .../src/components/ConflictModal.svelte | 40 ++++++++++++++++++- .../frontend/src/views/Devices.svelte | 5 +++ internal/api/server.go | 5 ++- internal/cliapp/syncsummary.go | 2 + internal/p2p/deletion_race_test.go | 26 ++++++++++-- internal/p2p/routes.go | 17 +++++++- internal/p2p/syncengine/engine.go | 18 ++++++++- internal/p2p/syncengine/transport.go | 15 +++++++ internal/p2p/syncengine/version.go | 37 +++++++++++++++++ internal/p2p/syncengine/version_test.go | 35 +++++++++++++++- internal/p2p/transport_lan.go | 4 +- internal/p2p/transport_wan.go | 2 +- internal/p2p/wanclient_handlers.go | 19 +++++++++ 13 files changed, 215 insertions(+), 10 deletions(-) diff --git a/cmd/opensave-app/frontend/src/components/ConflictModal.svelte b/cmd/opensave-app/frontend/src/components/ConflictModal.svelte index b16e4151..f24a7baa 100644 --- a/cmd/opensave-app/frontend/src/components/ConflictModal.svelte +++ b/cmd/opensave-app/frontend/src/components/ConflictModal.svelte @@ -1,6 +1,6 @@