Skip to content
Merged
182 changes: 182 additions & 0 deletions bench/warmup_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,182 @@
package bench_test

import (
"fmt"
"strings"
"testing"

"github.com/stretchr/testify/require"

ascache "github.com/sshaplygin/as-cache"
"github.com/sshaplygin/as-cache/bench"
)

// switchOnceTo asks for a switch to target on its first selection and for no
// change on every later one, so the switch lands at a known request. Bandit
// calls are serialised by the cache, so done needs no lock.
type switchOnceTo struct {
target ascache.PolicyType
done bool
}

func (b *switchOnceTo) RecordStats(ascache.ShadowStats) {}

func (b *switchOnceTo) SelectPolicy() ascache.PolicyType {
if b.done {
return ascache.Undefined
}
b.done = true

return b.target
}

// warmupWindows are the reported request ranges, as offsets from the switch.
var warmupWindows = [][2]int{{0, 1000}, {1000, 5000}, {5000, 20000}, {20000, 100000}}

type warmupRow struct {
name string
hits []int
}

// TestSwitchWarmupCost measures what a switch costs in the requests right
// after it, per migration strategy.
//
// A switch is only worth making if the policy switched to earns back what the
// switch loses. The loss is concentrated in the requests immediately after it,
// where a cold policy starts empty, a warm one starts with the outgoing
// policy's contents, and a gradual one fills as keys are asked for. An average
// over a whole run hides that shape entirely, so hits are reported in windows
// measured from the switch.
//
// The switch is scripted rather than chosen, so every strategy switches at the
// same request, from LRU to LFU, on the evidence suite's zipf workload and
// cache size. Two references bracket the result: LFU replayed on its own over
// the whole trace, which is an LFU that never had to warm up, and LRU replayed
// on its own, which is not switching at all.
func TestSwitchWarmupCost(t *testing.T) {
if testing.Short() {
t.Skip("evidence run; use make evidence")
}

const (
capacity = 500
switchAt = 100000
)
w := bench.Zipf(2*switchAt, 20000, 1.1, 1)

lruBuilder, lfuBuilder := fixedPolicy(t, "LRU"), fixedPolicy(t, "LFU")

rows := make([]warmupRow, 0, 8)
for _, ref := range []struct {
name string
builder bench.PolicyBuilder
}{
{"LFU all along (no warm-up)", lfuBuilder},
{"LRU, never switched", lruBuilder},
} {
policy, err := ref.builder.Build(capacity)
require.NoError(t, err)
rows = append(rows, warmupRow{ref.name, windowHits(policy, w, switchAt)})
}

for _, cfg := range []struct {
name string
strategy ascache.MigrationStrategy
maxRequests int64
}{
{"cold", ascache.MigrationCold, 0},
{"warm", ascache.MigrationWarm, 0},
{"gradual", ascache.MigrationGradual, 0},
{"gradual, capped at 10 Gets", ascache.MigrationGradual, 10},
{"gradual, capped at 100 Gets", ascache.MigrationGradual, 100},
{"gradual, capped at 1000 Gets", ascache.MigrationGradual, 1000},
} {
lru, err := lruBuilder.Build(capacity)
require.NoError(t, err)
lfu, err := lfuBuilder.Build(capacity)
require.NoError(t, err)

cache, err := ascache.NewAdaptiveCache(
[]ascache.Policy[string, int]{lru, lfu},
&switchOnceTo{target: ascache.LFU},
&ascache.Settings{
// The Get completing request switchAt ends the first epoch, so
// request index switchAt is the first one the new policy serves.
EpochRequests: switchAt,
EvictPartialCapacityFilling: true,
MigrationStrategy: cfg.strategy,
MigrationMaxRequests: cfg.maxRequests,
},
)
require.NoError(t, err)

hits := windowHits(cache, w, switchAt)
require.Equal(t, ascache.LFU, cache.ActivePolicy(), "%s: the scripted switch must have happened", cfg.name)
require.NoError(t, cache.Close())

rows = append(rows, warmupRow{cfg.name, hits})
}

t.Logf("\nhit rate after a switch from LRU to LFU at request %d (%s, %d requests, cache %d)\n%s",
switchAt, w.Name, len(w.Keys), capacity, warmupTable(rows))
}

func fixedPolicy(t *testing.T, name string) bench.PolicyBuilder {
t.Helper()

for _, builder := range bench.FixedPolicies() {
if builder.Name == name {
return builder
}
}
require.FailNow(t, "no fixed policy named "+name)

return bench.PolicyBuilder{}
}

// windowHits replays w through c as a read-through cache, filling on every miss
// as bench.Replay does, and counts the hits falling in each of warmupWindows.
func windowHits(c bench.Cache, w bench.Workload, switchAt int) []int {
hits := make([]int, len(warmupWindows))

for i, key := range w.Keys {
_, ok := c.Get(key)
if !ok {
c.Add(key, i)

continue
}
if i < switchAt {
continue
}

offset := i - switchAt
for j, window := range warmupWindows {
if offset >= window[0] && offset < window[1] {
hits[j]++
}
}
}

return hits
}

func warmupTable(rows []warmupRow) string {
var b strings.Builder

b.WriteString("| Configuration |")
for _, window := range warmupWindows {
fmt.Fprintf(&b, " %d-%d |", window[0], window[1])
}
b.WriteString("\n| --- |" + strings.Repeat(" --- |", len(warmupWindows)) + "\n")

for _, row := range rows {
fmt.Fprintf(&b, "| %s |", row.name)
for j, window := range warmupWindows {
fmt.Fprintf(&b, " %.2f%% |", 100*float64(row.hits[j])/float64(window[1]-window[0]))
}
b.WriteString("\n")
}

return b.String()
}
26 changes: 24 additions & 2 deletions cache.go
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,10 @@ type AdaptiveCache[K comparable, V any] struct {
migrateFrom PolicyType
migrationKeys []K
migrationRealKeys map[K]struct{}
// migrationRequests counts Gets served while the current gradual window
// is open, against Settings.MigrationMaxRequests. It needs no atomic:
// every Get in a window already holds the write lock.
migrationRequests int64

// --- Control Plane ---

Expand Down Expand Up @@ -110,8 +114,9 @@ type AdaptiveCache[K comparable, V any] struct {
// epochTicker is nil when the cache ends its epochs on request count
// alone, since time.NewTicker rejects a non-positive duration.
epochTicker *time.Ticker
// epochRequests counts Get calls since the last request-driven epoch. It
// is mutated on the read path, so it must be atomic.
// epochRequests counts every Get since construction; an epoch runs on each
// multiple of Settings.EpochRequests (see countRequest). It is mutated on
// the read path, so it must be atomic.
epochRequests atomic.Int64
settings *Settings

Expand All @@ -124,6 +129,13 @@ type AdaptiveCache[K comparable, V any] struct {
// recordActiveSample counts the active policy's result for a key that is part
// of the measured sample. Unsampled keys are served normally but not counted,
// so the active arm's evidence covers the same substream as every shadow's.
//
// On the read-lock path it runs after the lock is released, so it is not
// atomic with the Get it records. An epoch collecting in that gap reports the
// sample one epoch late; a switch landing in it credits the sample to the
// policy just made active. switchLocked clears the counters, but it cannot
// reach a Get already past its lookup, so what remains is bounded by the Gets
// in flight at that instant, not by how long the bandit took to decide.
func (c *AdaptiveCache[K, V]) recordActiveSample(sampled, hit bool) {
if !sampled {
return
Expand Down Expand Up @@ -185,6 +197,16 @@ func (c *AdaptiveCache[K, V]) get(key K) (V, bool) {
val, found := c.policies[c.activePolicy].Get(key)
c.recordActiveSample(sampled, found)

// A window still open after this Get counts it against the cap. Closing
// demotes the source, which is safe here: the value above came from the
// active policy, and the source is not the active policy.
if c.migrating {
c.migrationRequests++
if limit := c.settings.MigrationMaxRequests; limit > 0 && c.migrationRequests >= limit {
c.closeMigrationLocked()
}
}

return val, found
}

Expand Down
20 changes: 19 additions & 1 deletion docs/configuration.md
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,10 @@ type Settings struct {
// Default: MigrationCold.
MigrationStrategy MigrationStrategy

// MigrationMaxRequests caps a MigrationGradual window at N Get calls.
// Zero sets no cap. See "Migration Strategies".
MigrationMaxRequests int64

// ObserveOnly measures every arm without ever switching.
// See docs/advisor-mode.md.
ObserveOnly bool
Expand All @@ -44,7 +48,21 @@ type Settings struct {
| --- | --- | --- |
| `MigrationCold` (default) | New active policy starts empty | Simple; causes a temporary miss spike |
| `MigrationWarm` | All key/value pairs copied at switch time | No miss spike; O(n) work at switch |
| `MigrationGradual` | Keys promoted on Get; one key drained per Add | Spreads migration cost; window closes at the next epoch at the latest |
| `MigrationGradual` | Keys promoted on Get; one key drained per Add | Spreads migration cost; every `Get` takes the write lock while the window is open, which closes at the next epoch at the latest |

A gradual window serialises reads for as long as it is open. On a long epoch
that can be most of the epoch, so `MigrationMaxRequests` caps it at a number of
`Get` calls: when the cap is reached the window closes and the old policy is
demoted, and any key not promoted by then is gone — a later `Get` for it is a
miss, the same as it would have been under `MigrationCold`. Only `Get` counts,
because `Get` is what takes the write lock; `Add` drains a key per call and
shortens the window anyway. Zero, the default, sets no cap.

Measured on zipf with a switch from LRU to LFU, a cap of 100 cost 6.5 points of
hit rate over the first 1,000 requests after the switch, a cap of 10 made gradual
behave like cold, and a cap of 1,000 was never reached because read-through
traffic had already drained the window. [Evidence](evidence.md#what-does-a-switch-cost-right-after-it)
has the table, including what each strategy costs.

## Reducing shadow overhead

Expand Down
40 changes: 40 additions & 0 deletions docs/evidence.md
Original file line number Diff line number Diff line change
Expand Up @@ -407,3 +407,43 @@ Higher rates cost more and buy no better ranking here, so 0.05 is a reasonable
default. Raise it if your keyspace is small enough that 5% of it is only a
handful of keys -- `MinShadowCapacity` guards the degenerate end by raising the
effective rate rather than letting a miniature shrink into noise.

## What does a switch cost right after it?

Every figure above is a whole-run average, which hides where a switch's cost
falls: in the requests immediately after it. `TestSwitchWarmupCost` scripts one
switch, from LRU to LFU at request 100,000 of the zipf workload above (200,000
requests, cache 500), and reports hit rate in windows measured from the switch.
The switch is forced rather than chosen, so every strategy switches at the same
request. The run is deterministic: two runs produce identical tables.

| Configuration | 0-1000 | 1000-5000 | 5000-20000 | 20000-100000 |
| --- | --- | --- | --- | --- |
| LFU all along (no warm-up) | 77.00% | 74.28% | 74.59% | 73.90% |
| LRU, never switched | 70.50% | 67.90% | 67.58% | 66.82% |
| cold | 59.50% | 69.45% | 72.93% | 73.52% |
| warm | 70.50% | 70.03% | 72.91% | 73.55% |
| gradual | 70.40% | 70.12% | 72.88% | 73.52% |
| gradual, capped at 10 Gets | 59.90% | 69.47% | 72.93% | 73.52% |
| gradual, capped at 100 Gets | 63.90% | 69.58% | 72.99% | 73.53% |
| gradual, capped at 1000 Gets | 70.40% | 70.12% | 72.88% | 73.52% |

- **Cold pays for the switch up front.** Its first 1,000 requests serve 59.50%,
11.0 points below not switching at all and 17.5 below an LFU that never had to
warm up. By the next window it is already ahead of not switching.
- **Warm and gradual show no dip.** Both serve the first 1,000 requests within
0.1 points of LRU's own rate, because the entries LRU held are there to be hit.
- **None of them becomes the LFU that was there all along.** From 20,000 to
100,000 requests after the switch every strategy sits at 73.52-73.55%, against
73.90%. Moving the contents does not move the access history an LFU running
from the start would have built, and the gap is what that history was worth
here.
- **The window cap costs only when it bites.** At 10 Gets the gradual window
closes almost immediately and the first 1,000 requests look like cold
(59.90%); at 100 they lose 6.5 points against uncapped gradual. At 1,000 the
row is identical to uncapped: under read-through traffic every miss is an
`Add`, every `Add` drains a pending key, and the window emptied on its own
somewhere between 100 and 1,000 Gets.

Reproduce with `cd bench && go test -run TestSwitchWarmupCost -v .`, or
`make evidence`.
2 changes: 1 addition & 1 deletion docs/policies.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ cache, err := ascache.NewAdaptiveCache(
| LFU | `policies.NewLFU` | this repository's O(1) LFU; strong on stationary popularity, weak when it shifts |
| 2Q | `policies.NewTwoQueue` | scan-resistant; a scan cannot flush the working set |
| Random | `policies.NewRandomPolicy` | no bookkeeping; the control arm worth beating |
| TTL | `policies.NewTTL` | expiry as well as recency |
| TTL | `policies.NewTTL` | expiry as well as recency; expiry runs on the wall clock, so its hit rate depends on how fast traffic arrives, and a replay is reproducible only while the TTL is far longer than the run |
| ARC | `policies/arc.NewPolicy` | separate module — see below |
| W-TinyLFU | `policies/tinylfu.NewPolicy` | separate module; the strongest baseline |
| S3-FIFO | `policies/fifo.NewS3FIFOPolicy` | separate module; three FIFO queues, and deterministic |
Expand Down
17 changes: 11 additions & 6 deletions epoch.go
Original file line number Diff line number Diff line change
Expand Up @@ -28,20 +28,25 @@ func (c *AdaptiveCache[K, V]) runAdaptiveSelect() {
// the call that completes it. Caller must hold no lock: runEpoch takes the
// write lock.
//
// Exactly one caller per epoch sees the count equal the limit, so exactly one
// epoch runs however many goroutines are in Get. The limit is subtracted
// rather than the counter reset, so requests arriving mid-crossing still
// count towards the next epoch.
// The counter only ever increases, and an epoch runs on the call whose
// increment returns a multiple of the limit. Add hands every caller a distinct
// value, so each multiple is seen by exactly one caller: one epoch per limit
// requests however many goroutines are in Get, and a request arriving
// mid-crossing counts toward the next epoch.
//
// Do not reintroduce "compare with the limit, then subtract it". Callers that
// increment between one caller's comparison and its subtraction push the count
// past the limit unobserved, nothing subtracts again, and request-driven
// epochs stop for good.
func (c *AdaptiveCache[K, V]) countRequest() {
limit := c.settings.EpochRequests
if limit <= 0 {
return
}

if c.epochRequests.Add(1) != limit {
if c.epochRequests.Add(1)%limit != 0 {
return
}
c.epochRequests.Add(-limit)

c.runEpoch()
}
Expand Down
Loading
Loading