From d51abfbbd2235e49aae3049dda5daf852a486b2f Mon Sep 17 00:00:00 2001 From: Alexander Neubeck Date: Thu, 24 Sep 2026 08:26:18 -0700 Subject: [PATCH 1/2] Add cycle-consistent virtual permutations and measured comparison Keep the existing list-consistent iterator unchanged. Add allocation-free slot lookup, invariant tests, deterministic distribution diagnostics, and paired Criterion measurements with documented tradeoffs. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- crates/consistent-choose-k/README.md | 21 + .../consistent-choose-k/benchmarks/Cargo.toml | 6 + .../benchmarks/replica_comparison.rs | 223 +++++++ .../summarize_replica_comparison.py | 58 ++ .../docs/permutation-design.md | 9 +- .../docs/virtual-permutation-performance.md | 280 +++++++++ .../docs/virtual-permutation.md | 233 ++++++++ .../examples/permutation_diagnostics.rs | 99 +++ crates/consistent-choose-k/src/lib.rs | 2 + .../src/virtual_permutation.rs | 563 ++++++++++++++++++ 10 files changed, 1493 insertions(+), 1 deletion(-) create mode 100644 crates/consistent-choose-k/benchmarks/replica_comparison.rs create mode 100644 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py create mode 100644 crates/consistent-choose-k/docs/virtual-permutation-performance.md create mode 100644 crates/consistent-choose-k/docs/virtual-permutation.md create mode 100644 crates/consistent-choose-k/examples/permutation_diagnostics.rs create mode 100644 crates/consistent-choose-k/src/virtual_permutation.rs diff --git a/crates/consistent-choose-k/README.md b/crates/consistent-choose-k/README.md index 45574ab..1049a38 100644 --- a/crates/consistent-choose-k/README.md +++ b/crates/consistent-choose-k/README.md @@ -37,6 +37,27 @@ Why replication matters - Distributes read/write load across multiple owners, reducing hotspots. - Enables fast recovery and higher tail-latency resilience. +## Two permutation APIs, different membership semantics + +The existing `ConsistentPermutation` preserves **survivor list order** when +nodes are appended or removed from the end of `0..n`. The additional, +experimental `VirtualPermutation` instead preserves **replica slots**: on a +single-node append, at most one old slot changes, to the new node; on removal, +only a surviving slot that named that node changes. It does not preserve list +restriction and is not a drop-in replacement for the existing iterator or +its failover policies. Both yield distinct nodes and stable `k` prefixes. + +`VirtualPermutation::new(n, seed)` supports `1..=u64::MAX`, allocates no state, +and offers both an iterator and absolute `replica_at(slot)` lookup. Expected +`O(k)` enumeration follows under ideal independent uniform permutations with +constant-cost forward/inverse evaluation, **not** as a worst-case guarantee. +The implemented noncryptographic, 64-bit seeded Feistel family approximates +that randomness model; exact uniformity and independence are not claimed. + +See the [algorithm, proof assumptions and API guide](docs/virtual-permutation.md) +and the [reproducible comparison with the existing algorithm](docs/virtual-permutation-performance.md). +The existing APIs and their mappings remain unchanged. + ## Applications beyond replication The `ConsistentChooseK` iterator produces a per-key ranking of all `n` nodes in priority order — consistently and with zero memory overhead. This ranking is a strict superset of simple replication and enables drop-in replacements for several well-known algorithms that traditionally require maintaining expensive data structures such as hash rings. diff --git a/crates/consistent-choose-k/benchmarks/Cargo.toml b/crates/consistent-choose-k/benchmarks/Cargo.toml index 5f671fb..a206bce 100644 --- a/crates/consistent-choose-k/benchmarks/Cargo.toml +++ b/crates/consistent-choose-k/benchmarks/Cargo.toml @@ -9,6 +9,12 @@ path = "performance.rs" harness = false test = false +[[bench]] +name = "replica_comparison" +path = "replica_comparison.rs" +harness = false +test = false + [dependencies] consistent-choose-k = { path = "../" } diff --git a/crates/consistent-choose-k/benchmarks/replica_comparison.rs b/crates/consistent-choose-k/benchmarks/replica_comparison.rs new file mode 100644 index 0000000..ac9b5af --- /dev/null +++ b/crates/consistent-choose-k/benchmarks/replica_comparison.rs @@ -0,0 +1,223 @@ +//! Paired workloads only in the existing iterator's supported domain. +//! See ../docs/virtual-permutation-performance.md for methodology and results. + +use std::{ + hash::{DefaultHasher, Hash, Hasher}, + hint::black_box, + time::Duration, +}; + +use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use criterion::{criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, Throughput}; +use rand::{rngs::StdRng, RngExt, SeedableRng}; + +const WORKLOAD_SEED: u64 = 0x7065_726d_7574_6531; +const KEY_COUNT: usize = 128; +const NODES: &[u32] = &[ + 1, + 3, + 7, + 8, + 9, + 15, + 16, + 17, + 31, + 32, + 33, + 255, + 256, + 257, + 1000, + 1023, + 1024, + 1025, + 65535, + 65536, + 65537, + 1_000_000, + (1 << 30) - 1, + 1 << 30, +]; + +fn hash_key(key: u64) -> u64 { + let mut hasher = DefaultHasher::new(); + key.hash(&mut hasher); + hasher.finish() +} + +fn keys() -> Vec { + StdRng::seed_from_u64(WORKLOAD_SEED) + .random_iter() + .take(KEY_COUNT) + .collect() +} + +fn counts(n: u32) -> Vec { + let mut counts = vec![1, 2, 3, 8, 16]; + if n <= 1024 { + counts.extend([n as usize / 4, n as usize]); + } + counts.retain(|&k| k > 0 && k <= n as usize); + counts.sort_unstable(); + counts.dedup(); + counts +} + +fn consume(iter: impl Iterator>, k: usize) { + let sum = iter + .take(k) + .fold(0u64, |sum, node| sum.wrapping_add(node.into())); + black_box(sum); +} + +fn end_to_end(c: &mut Criterion) { + let keys = keys(); + let seeds: Vec<_> = keys.iter().copied().map(hash_key).collect(); + for mode in ["fresh", "seeded"] { + let mut group = c.benchmark_group(format!("replicas/{mode}")); + // Both execute exactly one complete query per key, including + // construction, streaming consumption and state destruction. + group.throughput(Throughput::Elements(KEY_COUNT as u64)); + for &n in NODES { + for k in counts(n) { + let input = if mode == "fresh" { &keys } else { &seeds }; + group.bench_function(BenchmarkId::new("layered", format!("n{n}_k{k}")), |b| { + b.iter(|| { + for &key in black_box(input) { + let seed = if mode == "fresh" { hash_key(key) } else { key }; + consume(ConsistentPermutation::new(black_box(n), seed), black_box(k)); + } + }) + }); + group.bench_function(BenchmarkId::new("virtual", format!("n{n}_k{k}")), |b| { + b.iter(|| { + for &key in black_box(input) { + let seed = if mode == "fresh" { hash_key(key) } else { key }; + consume( + VirtualPermutation::new(u64::from(black_box(n)), seed), + black_box(k), + ); + } + }) + }); + } + } + group.finish(); + } +} + +fn cost_components(c: &mut Criterion) { + let keys = keys(); + let seeds: Vec<_> = keys.iter().copied().map(hash_key).collect(); + let mut setup = c.benchmark_group("replicas/setup"); + setup.throughput(Throughput::Elements(KEY_COUNT as u64)); + setup.bench_function("hash_u64", |b| { + b.iter(|| { + for &key in black_box(&keys) { + black_box(hash_key(key)); + } + }) + }); + for &n in &[17, 257, 1000, 65537, 1 << 30] { + setup.bench_function(BenchmarkId::new("layered", n), |b| { + b.iter(|| { + for &seed in black_box(&seeds) { + black_box(ConsistentPermutation::new(black_box(n), seed)); + } + }) + }); + setup.bench_function(BenchmarkId::new("virtual", n), |b| { + b.iter(|| { + for &seed in black_box(&seeds) { + black_box(VirtualPermutation::new(u64::from(black_box(n)), seed)); + } + }) + }); + } + setup.finish(); + + for mode in ["stream_only", "collect", "slot"] { + let mut group = c.benchmark_group(format!("replicas/{mode}")); + group.throughput(Throughput::Elements(KEY_COUNT as u64)); + for &n in &[17, 257, 1000, 65537, 1 << 30] { + for k in counts(n) { + group.bench_function(BenchmarkId::new("layered", format!("n{n}_k{k}")), |b| { + match mode { + "stream_only" => b.iter_batched_ref( + || { + seeds + .iter() + .map(|&seed| ConsistentPermutation::new(n, seed)) + .collect::>() + }, + |iterators| { + for iter in black_box(iterators) { + consume(iter, black_box(k)); + } + }, + BatchSize::SmallInput, + ), + _ => b.iter(|| { + for &seed in black_box(&seeds) { + let mut iter = ConsistentPermutation::new(black_box(n), seed); + if mode == "collect" { + // Same output width and allocation policy for both. + let mut out = Vec::with_capacity(black_box(k)); + out.extend(iter.take(k).map(u64::from)); + black_box(out); + } else { + // No random-access API in the baseline: nth must replay. + black_box(iter.nth(black_box(k - 1))); + } + } + }), + } + }); + group.bench_function(BenchmarkId::new("virtual", format!("n{n}_k{k}")), |b| { + match mode { + "stream_only" => b.iter_batched_ref( + || { + seeds + .iter() + .map(|&seed| VirtualPermutation::new(u64::from(n), seed)) + .collect::>() + }, + |iterators| { + for iter in black_box(iterators) { + consume(iter, black_box(k)); + } + }, + BatchSize::SmallInput, + ), + _ => b.iter(|| { + for &seed in black_box(&seeds) { + let iter = VirtualPermutation::new(u64::from(black_box(n)), seed); + if mode == "collect" { + let mut out = Vec::with_capacity(black_box(k)); + out.extend(iter.take(k)); + black_box(out); + } else { + black_box(iter.replica_at(black_box(k as u64 - 1))); + } + } + }), + } + }); + } + } + group.finish(); + } +} + +criterion_group! { + name = benches; + config = Criterion::default() + .sample_size(20) + .warm_up_time(Duration::from_millis(100)) + .measurement_time(Duration::from_millis(300)) + .nresamples(1000) + .without_plots(); + targets = end_to_end, cost_components +} +criterion_main!(benches); diff --git a/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py b/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py new file mode 100644 index 0000000..0fa9f65 --- /dev/null +++ b/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py @@ -0,0 +1,58 @@ +#!/usr/bin/env python3 +"""Convert Criterion's batch estimates to ns/query and ns/replica CSV. + +Usage: python3 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py \ + target/criterion > comparison.csv +""" + +import csv +import json +from pathlib import Path +import re +import sys + + +def summarize(root): + rows = [] + for metadata in root.glob("**/new/benchmark.json"): + benchmark = json.loads(metadata.read_text()) + group = benchmark["group_id"] + if not group.startswith("replicas/"): + continue + mode = group.removeprefix("replicas/") + algorithm = benchmark["function_id"] + value = benchmark.get("value_str") or "" + match = re.fullmatch(r"n(\d+)_k(\d+)", value) + n, k = map(int, match.groups()) if match else (int(value or 0), 0) + estimates = json.loads((metadata.parent / "estimates.json").read_text()) + mean = estimates["mean"] + interval = mean["confidence_interval"] + divisor = benchmark["throughput"]["Elements"] + rows.append( + ( + mode, + algorithm, + n, + k, + mean["point_estimate"] / divisor, + interval["lower_bound"] / divisor, + interval["upper_bound"] / divisor, + estimates["std_dev"]["point_estimate"] / divisor, + mean["point_estimate"] / divisor / (k if k and mode != "slot" else 1), + ) + ) + if not rows: + raise SystemExit(f"No replica_comparison results found in {root}") + writer = csv.writer(sys.stdout) + writer.writerow( + ["mode", "algorithm", "n", "k", "ns_query", "ci95_low", "ci95_high", + "stddev_ns", "ns_replica"] + ) + for row in sorted(rows): + writer.writerow([*row[:4], *(f"{value:.3f}" for value in row[4:])]) + + +if __name__ == "__main__": + if len(sys.argv) != 2: + raise SystemExit(__doc__) + summarize(Path(sys.argv[1])) diff --git a/crates/consistent-choose-k/docs/permutation-design.md b/crates/consistent-choose-k/docs/permutation-design.md index 6ff55dd..9fd144c 100644 --- a/crates/consistent-choose-k/docs/permutation-design.md +++ b/crates/consistent-choose-k/docs/permutation-design.md @@ -4,6 +4,14 @@ This document explains the design of [`ConsistentPermutation`], the per-layer Feistel permutation iterator that this crate uses to drive its `n`-consistent ranking. +This is **survivor-list consistency**, not the cycle-projection/replica-slot +consistency of the additional [`VirtualPermutation`](virtual-permutation.md). +The algorithms are not interchangeable membership policies. See their +[paired performance comparison](virtual-permutation-performance.md). +Uniformity arguments below model the per-layer bijections as independent +uniform permutations; the actual finite-key, noncryptographic Feistel family +is a practical approximation, not an exact uniform sample from all permutations. + Given a 64-bit `key` and a universe size `n`, the iterator produces a uniformly distributed permutation of `[0, n)` as a streaming iterator satisfying: @@ -450,4 +458,3 @@ Observations matching the theory: pre-build option is `O(k log k)`), which is why the speedup ratio widens with `k`: at `k = 1 000, n = 1 000` the permutation implementation is ~50× faster. - diff --git a/crates/consistent-choose-k/docs/virtual-permutation-performance.md b/crates/consistent-choose-k/docs/virtual-permutation-performance.md new file mode 100644 index 0000000..11ebc35 --- /dev/null +++ b/crates/consistent-choose-k/docs/virtual-permutation-performance.md @@ -0,0 +1,280 @@ +# Replica-selection comparison + +The new `VirtualPermutation` trades slower sequential enumeration for +allocation-free state and direct replica-slot evaluation. It is **not a faster +drop-in replacement** for the existing `ConsistentPermutation`. + +The existing algorithm preserves survivor **list order**; the new one preserves +each old **slot** except a slot replaced by an appended node (or previously +naming a removed node). Both provide distinct, `k`-prefix-stable selections, +but these membership policies are not interchangeable. The +[design note](virtual-permutation.md) explains the construction, ideal-model +proof, noncryptographic approximation and expected-versus-worst-case distinction. + +## Reproduce + +From the repository root: + +```sh +cargo test -p consistent-choose-k +cargo bench -p consistent-choose-k-benchmarks --bench replica_comparison -- --noplot +python3 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py \ + target/criterion > comparison.csv +cargo run --release -p consistent-choose-k --example permutation_diagnostics +cargo run --release -p consistent-choose-k --example permutation_diagnostics -- --held-out +cargo test --release -p consistent-choose-k operation_count_diagnostics \ + -- --ignored --nocapture +``` + +Run these **sequentially**, not alongside other CPU-intensive tasks. Criterion +filters can restrict a rerun, for example +`-- 'replicas/fresh/.*/n1000_k3$' --noplot`. +Raw sample timings and estimates remain under `target/criterion`; the script +emits mean ns/query, bootstrap 95% confidence interval, sample standard deviation +in ns/query, and mean ns/replica. Criterion's console times are for a **batch of +128 queries**, so divide by 128, not by `k`, to obtain ns/query. Divide again by +`k` for ns/replica (a slot query returns just one result). + +### Recorded environment + +Measured on September 24, 2026, on an Apple M4 Max, native +`aarch64-apple-darwin`, macOS 27.0 build 26A428: + +- `rustc 1.92.0 (ded5c06cf 2025-12-08)`, LLVM 21.1.3. +- Apple clang 21.0.0 (`clang-2100.1.1.101`). +- Repository bench profile: optimized with debug info; configured + `-C target-feature=+neon`; no extra LTO, native-CPU flags or PGO. +- Criterion 0.8.2, rand 0.10.3; 20 samples, 100 ms warmup, 300 ms target + measurement time per case, 1,000 bootstrap resamples, no plots. +- The default selected Xcode could not link because its license was + unaccepted. All successful commands used the independently installed + Command Line Tools via + `DEVELOPER_DIR=/Library/Developer/CommandLineTools`. + No license was accepted and no system setting was changed. + +These are bounded microbenchmarks on a shared host, without CPU pinning, +frequency control or isolation. Intervals describe repeated timing samples of +one fixed key corpus, **not** uncertainty across all keys, machines or compiler +versions. Algorithms ran sequentially, existing first in each pair. Small +differences should not be interpreted as portable wins. + +### Workload and accounting + +The fixture is 128 `u64` keys generated by `StdRng` with seed +`0x7065726d75746531`. The fresh-key mode hashes each key with `DefaultHasher` +inside every query, identically for both algorithms. Width/round preparation +and all inverse walks in the new evaluator are included, as is the existing +iterator's counter allocation. These are fresh **queries** of a repeated +deterministic corpus, not new unpredictable keys on every benchmark iteration. +`StdRng` and `DefaultHasher` are not cross-version reproducibility contracts; +use the recorded toolchain and dependency versions to reproduce the exact +corpus and mapping. + +| Group | Timed work | +| --- | --- | +| `fresh` (primary comparison) | Key hashing, construction, streaming checksum of `k` values, destruction | +| `seeded` | Same query, reusing prehashed seeds; no free algorithm-specific cache | +| `setup` | Hash alone, or construction and destruction from a prehashed seed | +| `stream_only` | Consume `k` values from separately prepared iterators; construction and destruction excluded equally with `iter_batched_ref` | +| `collect` | Prehashed construction, allocate/fill/drop `Vec` with the same capacity `k`, destroy iterator | +| `slot` | Prehashed construction plus the result at `k-1`; existing `.nth(k-1)` replays, new `replica_at(k-1)` does not | + +The last row is a comparison of ways to answer a single-slot query, **not** a +claim that the existing API has a constant-time random-access operation. +Streaming-only is a diagnostic component benchmark, not a substitute for the +fresh-query comparison: prepared state has different cache footprints. +Inputs pass through `black_box`, and outputs are consumed. Both collection +paths use `u64` output elements, even though the old API returns `u32`. + +The full paired matrix uses +`n = 1,3,7,8,9,15,16,17,31,32,33,255,256,257,1000,1023,1024,1025,65535,65536,65537,1000000,2^30-1,2^30`. +For each, it uses valid `k` from `1,2,3,8,16`; at `n <= 1024`, it also uses +`floor(n/4)` and `n`, removing zeros and duplicates. There are 131 `(n,k)` +pairs in each of the fresh and seeded groups. Component groups use +`n = 17,257,1000,65537,2^30`. All paired timings stay within the old +implementation's supported domain. No wider-domain speedup is inferred. + +## Fresh-query results + +Representative mean **ns/query**, with bootstrap 95% confidence intervals in +brackets. Ratio is new/existing time: **above one means a regression**. + +| n | k | Existing layered | New virtual | Ratio | +| ---: | ---: | ---: | ---: | ---: | +| 1 | 1 | 58.04 [57.76, 58.27] | 7.38 [7.08, 7.82] | 0.13x | +| 8 | 1 | 51.31 [50.97, 51.63] | 114.80 [114.46, 115.16] | 2.24x | +| 8 | 8 | 227.41 [223.70, 230.99] | 1,005.85 [1,000.85, 1,014.19] | 4.42x | +| 17 | 3 | 110.06 [108.94, 111.07] | 622.33 [619.42, 625.40] | 5.65x | +| 33 | 33 | 600.19 [593.91, 604.57] | 7,988.45 [7,913.42, 8,104.45] | 13.31x | +| 255 | 3 | 38.10 [37.82, 38.39] | 142.31 [141.60, 143.18] | 3.74x | +| 256 | 3 | 39.69 [39.27, 40.12] | 143.02 [142.31, 143.75] | 3.60x | +| 257 | 3 | 63.41 [62.81, 63.97] | 305.48 [302.11, 310.19] | 4.82x | +| 1,000 | 1 | 22.39 [21.38, 24.14] | 49.53 [47.21, 52.43] | 2.21x | +| 1,000 | 3 | 35.58 [35.43, 35.72] | 104.90 [102.88, 107.36] | 2.95x | +| 1,000 | 16 | 103.18 [102.78, 103.58] | 536.52 [519.93, 561.57] | 5.20x | +| 1,000 | 250 | 2,042.64 [2,013.97, 2,070.34] | 14,996.65 [14,959.80, 15,025.20] | 7.34x | +| 1,000 | 1,000 | 9,735.29 [9,668.93, 9,808.83] | 85,990.56 [84,682.97, 87,389.54] | 8.83x | +| 1,023 | 3 | 36.29 [35.55, 37.30] | 106.26 [99.49, 118.27] | 2.93x | +| 1,024 | 3 | 35.79 [35.60, 35.99] | 97.85 [96.85, 98.99] | 2.73x | +| 1,025 | 3 | 64.33 [64.06, 64.61] | 266.90 [265.96, 268.03] | 4.15x | +| 65,535 | 3 | 35.19 [35.02, 35.39] | 88.23 [87.26, 89.16] | 2.51x | +| 65,536 | 3 | 36.40 [36.06, 36.77] | 90.23 [89.02, 91.43] | 2.48x | +| 65,537 | 3 | 66.23 [65.47, 66.87] | 268.53 [265.08, 272.36] | 4.05x | +| 1,000,000 | 3 | 36.14 [35.67, 36.65] | 96.65 [95.81, 97.70] | 2.67x | +| 2^30 - 1 | 3 | 38.49 [38.01, 38.91] | 87.74 [87.45, 88.21] | 2.28x | +| 2^30 | 3 | 38.86 [38.60, 39.12] | 89.96 [89.23, 90.68] | 2.31x | + +For example, `n=1000,k=3` is 11.86 versus 34.97 **ns/replica**; +`k=1000` is 9.74 versus 85.99 ns/replica. A full permutation has more +upper-half input slots, which need inverse walks; a short initial prefix often +avoids that work. Consequently mean cost per replica is not constant across +all `k`, even though the ideal-model expectation is bounded uniformly. + +Across the entire primary matrix the measured ratio ranges from 0.13x +(`n=1,k=1`, a degenerate constant-output case) to 13.31x +(`n=33,k=33`). The new implementation was slower in all 130 nontrivial cases. +The existing code's inexpensive four-round upper-layer +primitive and shared streaming counters generally beat repeated evaluation of +an 8/16/24-round, fully mixed forward/inverse primitive. Avoiding a small state +allocation does not make up for that extra arithmetic and traversal. +This compares the actual implementations, not algorithms normalized to an +identical primitive: both traversal strategy and mixing cost contribute. +All reported timings use the final strengthened schedule, not the rejected +eight-round version discussed below. + +## Separating setup, streaming and direct-slot costs + +Mean ns/query [95% confidence interval]. All rows below are prehashed; in the +`slot` rows, `k` means "query slot `k-1`", **not** consume `k` replicas. + +| Mode | n | k | Existing layered | New virtual | +| --- | ---: | ---: | ---: | ---: | +| Constructor + drop | 1,000 | - | 9.73 [9.62, 9.83] | 1.11 [1.10, 1.12] | +| Constructor + drop | 2^30 | - | 9.77 [9.72, 9.84] | 1.08 [1.08, 1.09] | +| Seeded stream query | 1,000 | 3 | 32.45 [32.21, 32.66] | 103.99 [102.70, 105.32] | +| Seeded stream query | 1,000 | 1,000 | 9,737.78 [9,680.29, 9,789.97] | 85,750.68 [85,646.50, 85,868.42] | +| Stream only | 1,000 | 3 | 14.62 [14.45, 14.82] | 91.78 [91.32, 92.34] | +| Stream only | 1,000 | 1,000 | 9,681.61 [9,618.72, 9,750.56] | 86,659.57 [86,278.25, 87,113.38] | +| Collect | 1,000 | 3 | 43.94 [43.02, 45.48] | 113.49 [112.83, 114.09] | +| Collect | 1,000 | 1,000 | 9,895.86 [9,786.30, 10,060.80] | 90,027.62 [88,127.94, 92,190.79] | +| Slot | 1,000 | 3 | 29.93 [29.79, 30.07] | 30.83 [30.69, 30.97] | +| Slot | 1,000 | 16 | 94.70 [93.68, 95.85] | 28.88 [28.76, 28.98] | +| Slot | 1,000 | 250 | 2,088.37 [2,042.18, 2,135.62] | 50.60 [50.48, 50.77] | +| Slot | 1,000 | 1,000 | 9,498.73 [9,419.35, 9,599.28] | 72.18 [72.02, 72.36] | +| Slot | 65,537 | 3 | 60.03 [59.29, 60.85] | 79.55 [78.88, 80.39] | + +Hashing a `u64` alone measured 7.48 [7.46, 7.50] ns. Component costs should +not be added or subtracted as exact identities: compiler optimization, +instruction overlap and cache state differ between the groups. +For `n=1000,k=3`, the seeded-query sample standard deviations were 0.54 ns +(existing) and 2.95 ns (new); collecting 1,000 replicas had standard deviations +315.22 ns and 4,646.27 ns respectively. All 721 benchmark estimates, including +their standard deviations, can be regenerated with the CSV script. + +Direct access becomes useful for later slots: querying slot 999 at `n=1000` +was about 132x faster than replaying the old iterator. Early slots need not +benefit, as the slot-2 rows show. This does not change the sequential-enumeration +regression, and the saved allocation is not being silently excluded from the +primary comparison. + +## State and allocations + +On this target, `size_of::()` is 40 bytes plus one +allocation for `4 * max(1, ceil(log2(n)/2))` bytes of counters: 20 bytes at +`n=1000`, 36 at `n=65537`, and 60 at `n=2^30`, excluding allocator metadata. +`size_of::()` is 24 bytes and it has no heap allocation. +It uses an additional constant-sized parameter block and scalars on the stack +during evaluation, not a recursion stack or a per-`n` cache. + +These allocation counts follow the source, rather than a custom allocator +installed in the timed benchmark. The diagnostic example prints the actual +struct sizes. The collection benchmark adds one capacity-`k` `Vec` +allocation to **both** algorithms, so it has two total allocations for the +existing iterator and one for the new one. Optional output storage is `8k` +bytes in both cases. + +## Distribution diagnostics and the rejected first version + +The deterministic example evaluates 200,000 hashed seeds per method for +`n = 3,4,5,7,8,9,16,17,32,64`. It reports each of the first up to eight slot +marginals, ordered pairs `(slot 0, slot 1)` and `(slot 0, last sampled slot)`, +unordered first-three subsets, and consecutive-key primary pairs. Expected +cells exclude repeated nodes for within-key pairs and include them for +cross-key pairs. All nontrivial cells are printed even when sparse (for +example, the `n=64` choose-three diagnostic has only 4.80 expected per cell). +These are exploratory diagnostics, not random p-value CI gates or a proof of +independence; the histograms overlap and are not independent tests. + +The primary corpus hashes `0x7065726d75746531 XOR i`, `i=0..199999`, with +`DefaultHasher`. A disjoint, deterministic held-out corpus uses +`0x686f6c646f757431 XOR i`. The held-out corpus was first evaluated **after** +fixing the stronger schedule; no additional tuning followed its results. +It is independent input data for diagnosis, not a claim that a deterministic +hash function supplies mathematically independent random variables. + +The initial **eight-round-at-every-width** implementation was rejected: +at `n=8`, its ordered `(0,1)` pair statistic was 1753.19 on 55 degrees of +freedom even though its primary marginal statistic was 4.24 on 7 degrees of +freedom. Clean marginals were insufficient. The final schedule uses 24 rounds +at widths 2--4, 16 at 5--7 and 8 at 8--64, fixed solely by width. This +strengthens mixing instead of reducing rounds to improve performance. + +| Metric, n=8 | Existing primary / held-out chi-square | Final virtual primary / held-out chi-square | Degrees of freedom | +| --- | ---: | ---: | ---: | +| Primary marginal | 4.79 / 4.66 | 8.63 / 1.22 | 7 | +| Slot 7 marginal | 11.50 / 5.73 | 6.44 / 1.26 | 7 | +| Ordered slots (0,1) | 65.97 / 43.73 | 54.15 / 44.41 | 55 | +| Ordered slots (0,7) | 66.92 / 44.74 | 60.47 / 45.43 | 55 | +| Unordered first-three subset | 53.47 / 59.83 | 47.66 / 40.41 | 55 | +| Consecutive-key primaries | 50.26 / 61.98 | 68.31 / 44.11 | 63 | + +At `n=32`, the final virtual first-three subset statistics are 4863.03 and +4873.25 on 4959 degrees of freedom, versus 5068.47 and 5024.53 for the existing +implementation. Its ordered `(0,1)` statistics are 1033.88 and 920.17 on 991 +degrees of freedom, versus 1032.74 and 985.50 for the existing implementation. + +No similarly large deviations appeared in the final virtual family's measured +marginals, ordered pairs, subsets or consecutive-key pairs on either corpus. +That observation does not establish exact uniformity, bound unseen-key bias, +or validate cryptographic properties. + +The **unchanged existing implementation** also has detectable deviations in +these broader diagnostics: at `n=9`, slot 4 has chi-square 253.52 on 8 degrees +of freedom in the primary corpus and 167.39 in the held-out corpus; slots 3 +and 5 also deviate. This is a limitation of the measured baseline, not a +reason to alter its behavior in this PR or to declare the new algorithm +uniform. Timing and statistical evidence answer different questions. + +## Untimed operation counts and tails + +The ignored diagnostic test wraps the **same evaluator and primitive** with +forward/inverse counters, outside any timed benchmark. For each `(n,slot)`, +it uses 10,000 deterministic SplitMix64-mixed seeds from integers `0..9999` +(the test source fixes the exact mixer and offset). Percentiles are nearest- +rank percentiles of total forward plus inverse calls; one call can contain +8, 16 or 24 Feistel rounds depending on width. + +| n | slot | Mean forward | Mean inverse | Mean visited levels | p50 calls | p99 calls | Observed max | +| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| 256 | 0 | 1.9808 | 0 | 1.9808 | 1 | 7 | 8 | +| 257 | 0 | 3.9806 | 1.0051 | 2.9715 | 4 | 16 | 28 | +| 257 | 256 | 3.9660 | 3.9270 | 2.9786 | 7 | 22 | 41 | +| 65,536 | 0 | 1.9922 | 0 | 1.9922 | 1 | 7 | 15 | +| 65,537 | 0 | 3.9822 | 0.9900 | 2.9922 | 4 | 16 | 30 | +| 65,537 | 65,536 | 4.0104 | 4.0244 | 3.0065 | 7 | 23 | 36 | +| 2^30 | 2^29 | 1.9907 | 1.4876 | 1.9907 | 1 | 16 | 49 | +| 2^63 + 1 | 2^63 | 3.9795 | 4.0169 | 2.9878 | 7 | 23 | 40 | +| u64::MAX | 0 | 2.0194 | 0 | 2.0194 | 2 | 7 | 16 | + +The last two rows are **unpaired, wider-domain diagnostics**, not performance +comparisons with the old API. Near a half-full top domain, traversal work +increases substantially; constant expected work does not mean flat latency. +The 8.0348-call sample mean at `n=65537,slot=65536` is not a contradiction of +the ideal-model bound: it is a finite sample of a finite-key approximation, +not the ideal ensemble expectation. + +Observed maxima are not guarantees. A deterministic adversarial-oracle unit +test uses ascending cycles and needs over 4,096 calls for `n=2049,slot=2048`; +the evaluator still returns the correct result without a retry cap. Long +cycles and linear-in-domain worst cases remain possible. Neither the +diagnostics nor the timing confidence intervals establish a latency bound. diff --git a/crates/consistent-choose-k/docs/virtual-permutation.md b/crates/consistent-choose-k/docs/virtual-permutation.md new file mode 100644 index 0000000..2d05a37 --- /dev/null +++ b/crates/consistent-choose-k/docs/virtual-permutation.md @@ -0,0 +1,233 @@ +# VirtualPermutation: cycle-consistent replica slots + +`VirtualPermutation` is an additional experimental algorithm, not a replacement +for `ConsistentPermutation`. Both produce deterministic, distinct candidates, +and every `k`-selection is a prefix of the same per-key list. Their membership +semantics are **different**: + +| Property | Existing `ConsistentPermutation` | New `VirtualPermutation` | +| --- | --- | --- | +| Append one node | Insert it into the ranking; later replica slots can shift | At most one old replica slot changes, to the new node | +| Remove the last node | Remove it from the ranking; later slots can shift | Only a surviving slot that named the removed node changes | +| Survivor list order | Preserved | Not promised | +| Projection between sizes | Delete entries from the output list | Delete labels from the permutation's cycles | +| Direct slot lookup | Replay iterator through that slot | `replica_at(slot)`, without replay | +| Supported `n` | `1..=2^30` (`u32`) | `1..=u64::MAX` | +| Mutable state | Per-layer `Vec` counters | Three `u64` fields; no heap allocation | + +For example, the permutation written as an output list `[2, 0, 1]` is the +cycle `0 -> 2 -> 1 -> 0`. Cycle-deleting node 2 produces `[1, 0]`, +**not** the survivor list `[0, 1]`. The changed old slot is 0, the slot that +named the removed node. This difference matters for failover and bounded-load +policies that depend on survivor priority; do not substitute the new API into +`ConsistentNodeMap` or existing consumers without considering their semantics. + +Membership is consecutive IDs `0..n`. Additions append IDs and removals remove +a suffix; arbitrary holes, weights and physical-node remapping are out of scope. +The one-slot bound is per single-node change, not for an entire suffix at once. + +## API and randomness model + +```rust +use consistent_choose_k::VirtualPermutation; +use std::hash::{DefaultHasher, Hash, Hasher}; + +let mut hasher = DefaultHasher::new(); +"object-key".hash(&mut hasher); +let seed = hasher.finish(); +let permutation = VirtualPermutation::new(1_000, seed); +let third_replica = permutation.replica_at(2); +let replicas: Vec = permutation.take(3).collect(); +assert_eq!(third_replica, replicas[2]); +``` + +Like the existing iterator, the constructor accepts an already well-mixed +64-bit seed, not an arbitrary application key. Use the same seed for all `n` +and `k`. `DefaultHasher` above is convenient for local examples and matches the +benchmark convention; Rust does not promise its mapping is stable across +versions. Distributed deployments need a specified, versioned key hash and +identical algorithm versions on every participant. + +`new(0, seed)` and `replica_at(slot >= n)` panic, following the existing +constructor's assertion convention. `replica_at` is absolute, independent of +the cursor. The iterator is cloneable and fused, reports its remaining size, +and implements `nth` without replay. `.take(0)` is empty, `.take(n)` is the full +permutation, and `.take(k)` for `k > n` stops at exhaustion, as usual for Rust +iterators. There is no separate `k` constructor argument. + +Under **independent uniform ideal permutations for each key and bit width**, +the construction below gives a uniform permutation for every fixed `n`: +each ordered `k`-tuple of distinct nodes is equally likely, and different keys' +preference permutations are independent. Within a key, replicas are sampled +without replacement, not independently. + +The actual primitive is an alternating-XOR Feistel network with SplitMix64's +avalanche finalizer as its round mixer. It uses 24 rounds for widths 2 through +4, 16 for widths 5 through 7, and 8 for widths 8 through 64. Tiny half-domains +need extra rounds: the initial eight-round version had a strong ordered-pair +bias at width three despite clean marginals. This fixed policy depends **only** +on bit width, never active `n`, requested `k`, or observed outputs. Unequal +halves support odd widths; the one-bit case is a keyed XOR. +Widths are domain-separated through a +mixed seed; round keys use distinct Weyl offsets. The inverse undoes the same +updates in reverse order. Geometry never requires a shift by 64: the largest +half-width is 32 and the evaluator's largest half boundary is `1 << 63`. +The conceptual full width-64 domain has size `2^64`, but that cardinality is +never materialized in a `u64`. + +This finite, 64-bit seeded family is **noncryptographic and only a practical +pseudorandom approximation**, not independently sampled ideal permutations, +not a proven secure PRP, and not exactly uniform over all `n!` permutations. +Domain separation and avalanche do not prove independence. Seed collisions +give identical permutations. Even conventional XOR-Feistel families have +structural restrictions (for example, even permutation parity when both +halves have at least two bits). The mixing schedule is fixed rather than +weakened to make a benchmark win. Do not use this as encryption or with +adversarial keys requiring a cryptographic guarantee. + +The existing `layer_apply` is deliberately unchanged: it only supports even +widths through 30, has no inverse, and has its own key/round schedule. Extending +or replacing it would change existing mappings. The new private primitive +therefore lives with the new evaluator. + +## Dyadic lift and cycle deletion + +The following is a derived construction and evaluator, not an implementation +of a published constant-time replica-selection algorithm. + +Let `h = 2^(b-1)`, let `A = [0,h)` be the old labels, and let `Q = P(seed,b)` +be an ordinary permutation of `[0,2h)`. Let `R` be its cycle projection onto +`A`: follow `Q` until the next label in `A`. Starting with `F_1(0) = 0`, define + +```text +F_(2h) = extend(F_h composed with inverse(R), fixing upper labels) composed with Q +``` + +Every `Q`-chain starting at old label `a` ends at old label `R(a)`. +The left composition changes only that chain's final destination, from +`R(a)` to `F_h(a)`. Upper edges and upper-only cycles are unchanged. Thus +cycle-projecting `F_(2h)` onto `A` gives `F_h`. For `h < n < 2h`, define `F_n` +by deleting all labels at least `n` from `F_(2h)`'s cycles. Cycle projections +compose, including across power-of-two boundaries. + +**Uniformity in the ideal model.** Conditional on any fixed `Q`, independent +uniform `F_h` makes `F_h composed with inverse(R)` uniform on the old-label +permutation group. This factor is consequently independent of `Q`; composing +it with uniform `Q` makes `F_(2h)` uniform. Cycle projection preserves +uniformity: each permutation of `n-1` labels has exactly `n` extensions +(insert the new label after any old label, or as a singleton cycle). +Induction establishes uniformity at every size. This proof uses ideal +uniformity, not the diagnostics of the implemented finite-key family. + +**Consistency.** Deleting label `n` from `F_(n+1)` only redirects its +predecessor to its successor, or removes its singleton cycle. For each +surviving input `r < n`, `F_(n+1)(r)` is therefore either `F_n(r)` or `n`. +At most one old input changes. A fixed prefix of input slots inherits this +property, while bijectivity supplies distinct outputs and prefix stability. +Enumerate *distinct inputs* `0,1,...,k-1`; following one output cycle instead +would not enumerate a full permutation. + +## Evaluator + +```text +replica_at(seed, n, r): + require 0 <= r < n + b = bit_length(n - 1) + x = r + while b > 0: + half = 1 << (b - 1) + y = P(seed, b, x) + if y >= n: + x = y + continue + if y >= half: + return y + while x >= half: + x = P_inverse(seed, b, x) + n = half + b -= 1 + return 0 +``` + +Walking inactive upper labels can use `Q` rather than the recursively defined +`F`, because edges whose outputs are upper labels were unchanged by the lift. +On reaching a lower output `y`, walking backward from its predecessor `x` +finds the old input `a = inverse(R)(y)`. The next level evaluates `F_h(a)`. +This proves the evaluator agrees with the explicit lift. + +Walks terminate because they follow a finite bijection's cycle. A forward walk +starts in the retained set, so it cannot remain forever in an inactive-only +cycle. A backward walk is entered only after reaching a lower label, so that +cycle necessarily meets the lower half. There are no retry caps or mapping- +changing fallbacks. + +## Expected work, not a worst-case guarantee + +Count each forward or inverse permutation evaluation as one constant-cost +operation. The following bounds are derived for **ideal independent uniform** +permutations and a fixed input, not asserted as a published theorem or a +guarantee for every seed of the implemented mixer. + +At a full dyadic level, descent probability is exactly one half. For an +upper input conditional on descent, the mean inverse-walk length is +`2h/(h+1)`: predecessors are sampled without replacement from `2h-1` labels, +`h` of them lower. The terminal lower input is uniform. If `T_s(a)` is mean +cost at full size `s`, and `U_s` its average over inputs, then + +```text +T_(2h)(a) = 1 + T_h(a)/2 if a < h +T_(2h)(a) = 1 + h/(h+1) + U_h/2 if a >= h +U_(2h) = 1 + h/(2(h+1)) + U_h/2 +T_1 = U_1 = 0 +``` + +Induction gives `U_s < 3` and `T_s(a) < 3.5`. The initial partial level is +different: its descent probability `p = h/n` can approach one. Its forward +length has mean `ell = (2h+1)/(n+1) < 2`; the retained endpoint is uniform +and independent of that length. On descent, the inverse walk retraces those +`ell-1` inactive edges. Starting from an upper input also requires finding a +lower predecessor; conditional on forward length `L`, this takes +`(2h-L+1)/(h+1)` further inverse calls on average. Thus + +```text +E C_n(r) = ell + p * (ell - 1 + T_h(r)) if r < h +E C_n(r) = ell + p * (ell - 1 + (2h+1-ell)/(h+1) + U_h) if r >= h +``` + +In particular, mean total work is **less than eight primitive calls per +slot**, uniformly in `n,r` in the ideal model. The upper-input bound approaches +eight at `n=h+1, r=h` as `h` grows. All later levels are full, so descent is +geometric after the initial level. The entering lower input depends on upper +permutations, but is independent of lower-width permutations; no independence +between walks, or between replica costs, is assumed. Linearity of expectation +gives expected `O(k)` enumeration. + +Long cycles still permit linear-in-domain walks for an unlucky permutation. +This is **not worst-case `O(k)`**, nor an adversarial-latency guarantee. State +is `O(1)` machine words, excluding optional returned output, with a +constant-sized width parameter block and no recursive stack, ring, permutation +array, or duplicate set. See [measurements and diagnostics](virtual-permutation-performance.md) +for observed forward/inverse counts and tails on the actual mixer. + +## References and verification + +The mathematical antecedent is a *virtual permutation*, using **cycle** +projection: + +- Neretin, [Virtual permutations and polymorphisms, section 1.2](https://arxiv.org/html/2202.12978v1#S1): + cycle deletion and its equivariance. +- Bourgade, Najnudel and Nikeghbali, + [A unitary extension of virtual permutations, section 1](https://arxiv.org/html/1102.2633v1#S1): + cycle projections, the Chinese restaurant construction, and the uniform + family as the Ewens parameter-one case. + +These references do **not** supply this evaluator, Feistel schedule, or its +performance bound. + +Tests exercise all word widths and extreme values, exhaustive small PRP +round-trips, full small-domain permutations, `k` prefixes, append/delete +consistency including dyadic boundaries and `u64::MAX`, and an independent +table-based lift/projection oracle. Exhaustive ideal permutations check equal +lift multiplicities at size four and all extensions of one fixed size-four +permutation through sizes five to eight. These are regression checks, not +substitutes for the ideal-model proof or evidence of cryptographic security. diff --git a/crates/consistent-choose-k/examples/permutation_diagnostics.rs b/crates/consistent-choose-k/examples/permutation_diagnostics.rs new file mode 100644 index 0000000..20ec0cf --- /dev/null +++ b/crates/consistent-choose-k/examples/permutation_diagnostics.rs @@ -0,0 +1,99 @@ +//! Deterministic distribution diagnostics, not probabilistic CI gates. +//! cargo run --release -p consistent-choose-k --example permutation_diagnostics + +use std::hash::{DefaultHasher, Hash, Hasher}; + +use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; + +const SAMPLES: u64 = 200_000; +fn key_seed(key: u64, workload_seed: u64) -> u64 { + let mut hasher = DefaultHasher::new(); + (workload_seed ^ key).hash(&mut hasher); + hasher.finish() +} + +fn report(algorithm: &str, n: usize, metric: &str, cells: &[u64]) { + let expected = cells.iter().sum::() as f64 / cells.len() as f64; + let chi2: f64 = cells + .iter() + .map(|&v| (v as f64 - expected).powi(2) / expected) + .sum(); + println!( + "{algorithm},{n},{metric},{},{expected:.3},{chi2:.3},{},{}", + cells.len() - 1, + cells.iter().min().expect("nonempty histogram"), + cells.iter().max().expect("nonempty histogram"), + ); +} + +fn main() { + let mut args = std::env::args().skip(1); + let workload_seed = match args.next().as_deref() { + None => 0x7065_726d_7574_6531, + Some("--held-out") => 0x686f_6c64_6f75_7431, + Some(_) => panic!("usage: permutation_diagnostics [--held-out]"), + }; + assert!( + args.next().is_none(), + "usage: permutation_diagnostics [--held-out]" + ); + println!("# samples={SAMPLES}, key_seed={workload_seed:#x}"); + println!( + "# iterator_bytes: layered={}, virtual={}; layered also owns heap counters", + std::mem::size_of::(), + std::mem::size_of::(), + ); + println!("algorithm,n,metric,df,expected_per_cell,chi2,min,max"); + for n in [3, 4, 5, 7, 8, 9, 16, 17, 32, 64] { + for algorithm in ["layered", "virtual"] { + let slots = n.min(8); + let mut marginal = vec![vec![0u64; n]; slots]; + let mut pairs = vec![0; n * n]; + let mut distant_pairs = vec![0; n * n]; + let mut triples = vec![0; n * (n - 1) * (n - 2) / 6]; + let mut consecutive_keys = vec![0; n * n]; + let mut previous_primary = None; + for key in 0..SAMPLES { + let seed = key_seed(key, workload_seed); + let values: Vec = if algorithm == "layered" { + ConsistentPermutation::new(n as u32, seed) + .take(slots) + .map(|v| v as usize) + .collect() + } else { + VirtualPermutation::new(n as u64, seed) + .take(slots) + .map(|v| v as usize) + .collect() + }; + for (histogram, &v) in marginal.iter_mut().zip(&values) { + histogram[v] += 1; + } + pairs[values[0] * n + values[1]] += 1; + distant_pairs[values[0] * n + values[slots - 1]] += 1; + let mut triple = [values[0], values[1], values[2]]; + triple.sort_unstable(); + let [a, b, c] = triple; + triples[c * (c - 1) * (c - 2) / 6 + b * (b - 1) / 2 + a] += 1; + if let Some(previous) = previous_primary { + consecutive_keys[previous * n + values[0]] += 1; + } + previous_primary = Some(values[0]); + } + for (slot, cells) in marginal.iter().enumerate() { + report(algorithm, n, &format!("slot_{slot}"), cells); + } + for (metric, cells) in [("ordered_0_1", pairs), ("ordered_0_last", distant_pairs)] { + assert!((0..n).all(|v| cells[v * n + v] == 0)); + let off_diagonal: Vec<_> = cells + .into_iter() + .enumerate() + .filter_map(|(i, count)| (i / n != i % n).then_some(count)) + .collect(); + report(algorithm, n, metric, &off_diagonal); + } + report(algorithm, n, "choose_3", &triples); + report(algorithm, n, "consecutive_key_primaries", &consecutive_keys); + } + } +} diff --git a/crates/consistent-choose-k/src/lib.rs b/crates/consistent-choose-k/src/lib.rs index 25b7e0f..4278f83 100644 --- a/crates/consistent-choose-k/src/lib.rs +++ b/crates/consistent-choose-k/src/lib.rs @@ -3,6 +3,7 @@ mod consistent_hash; mod consistent_permutation; mod consistent_reservoir; mod node_map; +mod virtual_permutation; pub use choose_k::ConsistentChooseKHasher; pub use consistent_hash::{ ConsistentHashIterator, ConsistentHashRevIterator, ConsistentHasher, HashSeqBuilder, @@ -11,3 +12,4 @@ pub use consistent_hash::{ pub use consistent_permutation::ConsistentPermutation; pub use consistent_reservoir::ConsistentReservoir; pub use node_map::ConsistentNodeMap; +pub use virtual_permutation::VirtualPermutation; diff --git a/crates/consistent-choose-k/src/virtual_permutation.rs b/crates/consistent-choose-k/src/virtual_permutation.rs new file mode 100644 index 0000000..76be660 --- /dev/null +++ b/crates/consistent-choose-k/src/virtual_permutation.rs @@ -0,0 +1,563 @@ +//! Cycle-consistent permutations, as opposed to the survivor-list consistency +//! of `ConsistentPermutation`. See `docs/virtual-permutation.md`. + +use std::iter::FusedIterator; + +const WEYL: u64 = 0x9E37_79B9_7F4A_7C15; + +#[inline] +fn mix(mut x: u64) -> u64 { + x = (x ^ (x >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + x = (x ^ (x >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + x ^ (x >> 31) +} + +trait Permutation { + fn forward(&self, x: u64) -> u64; + fn inverse(&self, x: u64) -> u64; +} + +/// Alternating Feistel half updates, including unequal halves for odd widths. +/// This is a noncryptographic family, not a uniform or secure PRP. +struct WordPermutation { + key: u64, + bits: u32, + right_bits: u32, + left_mask: u64, + right_mask: u64, +} + +impl WordPermutation { + #[inline] + fn new(seed: u64, bits: u32) -> Self { + debug_assert!((1..=64).contains(&bits)); + let left_bits = bits / 2; + let right_bits = bits - left_bits; + Self { + key: mix(seed ^ WEYL.wrapping_mul(u64::from(bits))), + bits, + right_bits, + left_mask: (1u64 << left_bits) - 1, + right_mask: (1u64 << right_bits) - 1, + } + } + + #[inline] + fn round(&self, x: u64, round: u64) -> u64 { + mix(x ^ self.key.wrapping_add(WEYL.wrapping_mul(round + 1))) + } + + #[inline] + fn transform_rounds(&self, x: u64) -> u64 { + let mut left = x >> self.right_bits; + let mut right = x & self.right_mask; + for i in 0..PAIRS { + let pair = if INVERSE { PAIRS - 1 - i } else { i }; + if INVERSE { + right ^= self.round(left, 2 * pair + 1) & self.right_mask; + left ^= self.round(right, 2 * pair) & self.left_mask; + } else { + left ^= self.round(right, 2 * pair) & self.left_mask; + right ^= self.round(left, 2 * pair + 1) & self.right_mask; + } + } + (left << self.right_bits) | right + } + + #[inline] + fn transform(&self, x: u64) -> u64 { + // Tiny half-domains need more mixing: eight rounds had strong + // ordered-pair bias at width three despite clean marginals. + match self.bits { + 1 => x ^ (self.key & 1), + 2..=4 => self.transform_rounds::<12, INVERSE>(x), + 5..=7 => self.transform_rounds::<8, INVERSE>(x), + _ => self.transform_rounds::<4, INVERSE>(x), + } + } +} + +impl Permutation for WordPermutation { + #[inline] + fn forward(&self, x: u64) -> u64 { + self.transform::(x) + } + + #[inline] + fn inverse(&self, x: u64) -> u64 { + self.transform::(x) + } +} + +#[inline] +fn evaluate(mut n: u64, mut x: u64, mut layer: impl FnMut(u32) -> P) -> u64 { + let mut bits = u64::BITS - (n - 1).leading_zeros(); + while bits > 0 { + let half = 1u64 << (bits - 1); + let permutation = layer(bits); + loop { + let y = permutation.forward(x); + if y >= n { + x = y; + continue; + } + if y >= half { + return y; + } + // Find the old input at the start of this Q-chain, not its old + // output y: this applies inverse(cycle_projection(Q)) before + // evaluating the smaller consistent permutation. + while x >= half { + x = permutation.inverse(x); + } + break; + } + n = half; + bits -= 1; + } + 0 +} + +/// An allocation-free, per-key permutation of `0..n` with stable replica slots. +/// +/// For fixed `(n, seed)`, taking `k` items gives `k` distinct nodes and is a +/// prefix of every longer selection. Appending node `n` changes at most one +/// existing slot, and that slot changes to `n`; removing the last node changes +/// only its slot (among slots that still exist). Membership must be a prefix +/// of consecutive IDs. +/// +/// **Not a drop-in replacement for [`crate::ConsistentPermutation`]:** this +/// preserves cycle projections, not survivor list order. Removing a node from +/// the output list does not generally produce the smaller permutation. +/// +/// Uniform, independent permutations per key and bit width give uniform +/// selections and expected `O(k)` evaluation with constant-cost forward/inverse +/// primitives. The implemented 64-bit seeded, 8/16/24-round Feistel family is +/// only a practical noncryptographic approximation to that ideal. Neither +/// exact uniformity, cryptographic security, nor worst-case `O(k)` is claimed. +/// State is constant size; no rings, permutation tables, or duplicate sets +/// are constructed. Walks have no artificial retry limit. +/// +/// ``` +/// use consistent_choose_k::VirtualPermutation; +/// +/// let seed = 0x1234_5678_9abc_def0; // normally a well-mixed hash of the key +/// let selection = VirtualPermutation::new(100, seed); +/// let third = selection.replica_at(2); +/// assert_eq!(selection.clone().nth(2), Some(third)); +/// let replicas: Vec<_> = selection.take(3).collect(); +/// assert_eq!(replicas.len(), 3); +/// ``` +#[derive(Clone, Debug)] +pub struct VirtualPermutation { + seed: u64, + n: u64, + next: u64, +} + +impl VirtualPermutation { + /// Construct a permutation for `1..=u64::MAX` nodes. + /// + /// As with [`crate::ConsistentPermutation::new`], supply a well-mixed + /// 64-bit hash of the key. Width/round domain separation is internal and + /// independent of `n` and of the requested replica count. + /// + /// # Panics + /// + /// Panics if `n == 0`. + pub fn new(n: u64, seed: u64) -> Self { + assert!(n > 0, "n must be at least 1"); + Self { seed, n, next: 0 } + } + + /// Universe size, independent of the iterator's current position. + pub fn n(&self) -> u64 { + self.n + } + + /// Evaluate an absolute zero-based replica slot without advancing the + /// iterator. Expected constant work under ideal independent permutations; + /// no worst-case constant-time guarantee. + /// + /// # Panics + /// + /// Panics if `slot >= self.n()`. + pub fn replica_at(&self, slot: u64) -> u64 { + assert!(slot < self.n, "replica slot must be less than n"); + evaluate(self.n, slot, |bits| WordPermutation::new(self.seed, bits)) + } +} + +impl Iterator for VirtualPermutation { + type Item = u64; + + fn next(&mut self) -> Option { + if self.next == self.n { + return None; + } + let value = self.replica_at(self.next); + self.next += 1; + Some(value) + } + + fn size_hint(&self) -> (usize, Option) { + match usize::try_from(self.n - self.next) { + Ok(remaining) => (remaining, Some(remaining)), + Err(_) => (usize::MAX, None), + } + } + + fn nth(&mut self, n: usize) -> Option { + self.next = self.next.saturating_add(n as u64).min(self.n); + self.next() + } +} + +impl FusedIterator for VirtualPermutation {} + +#[cfg(test)] +mod tests { + use super::*; + + fn seed(i: u64) -> u64 { + mix(i.wrapping_add(WEYL)) + } + + #[test] + fn word_round_trips_all_widths() { + for bits in 1..=64 { + let mask = u64::MAX >> (64 - bits); + for key in [0, 1, u64::MAX, seed(42)] { + let p = WordPermutation::new(key, bits); + for x in [0, 1, mask / 2, mask / 2 + 1, mask] + .into_iter() + .chain((0..128).map(|i| seed(i) & mask)) + { + let y = p.forward(x); + assert_eq!(y & !mask, 0); + assert_eq!(p.inverse(y), x, "bits={bits} key={key} x={x}"); + assert_eq!(p.forward(p.inverse(x)), x); + } + } + } + } + + #[test] + fn word_exhaustive_small_domains() { + for bits in 1..=10 { + for key in 0..16 { + let p = WordPermutation::new(seed(key), bits); + let mut outputs: Vec<_> = (0..1 << bits) + .map(|x| { + let y = p.forward(x); + assert_eq!(p.inverse(y), x); + y + }) + .collect(); + outputs.sort_unstable(); + assert_eq!(outputs, (0..1 << bits).collect::>()); + } + } + } + + #[test] + fn full_permutations_prefixes_and_membership() { + for key in 0..32 { + let key = seed(key); + let mut previous = vec![]; + for n in 1..=257 { + let permutation = VirtualPermutation::new(n, key); + let values: Vec<_> = permutation.clone().collect(); + let mut sorted = values.clone(); + sorted.sort_unstable(); + assert_eq!(sorted, (0..n).collect::>()); + for k in [0, 1, 2, 3, 8, n / 2, n] { + if k <= n { + assert_eq!( + permutation.clone().take(k as usize).collect::>(), + values[..k as usize] + ); + } + } + let mut changed = 0; + for (r, old) in previous.iter().enumerate() { + assert_eq!(permutation.replica_at(r as u64), values[r]); + if *old != values[r] { + assert_eq!(values[r], n - 1); + changed += 1; + } + // Cycle-deleting the new node recovers every old slot. + let projected = if values[r] == n - 1 { + values[n as usize - 1] + } else { + values[r] + }; + assert_eq!(*old, projected); + } + assert!(changed <= 1); + previous = values; + } + } + } + + #[test] + fn large_domains_and_power_boundaries() { + for bits in 1..64 { + let half = 1u64 << bits; + for n in [half - 1, half, half + 1, u64::MAX - 1] { + for key in [0, 1, u64::MAX, seed(123)] { + let p = VirtualPermutation::new(n, key); + let larger = VirtualPermutation::new(n + 1, key); + for r in [0, n / 2, n - 1] { + let old = p.replica_at(r); + let new = larger.replica_at(r); + assert!(old < n && new <= n); + assert!(new == old || new == n); + assert_eq!(old, if new == n { larger.replica_at(n) } else { new }); + } + } + } + } + let mut p = VirtualPermutation::new(u64::MAX, seed(9)); + p.next = u64::MAX - 1; + assert!(p.next().is_some()); + assert_eq!(p.next(), None); + assert_eq!(p.next(), None); + } + + #[test] + fn iterator_boundaries() { + let mut p = VirtualPermutation::new(1, 0); + assert_eq!(p.n(), 1); + assert_eq!(p.size_hint(), (1, Some(1))); + assert_eq!(p.clone().take(0).count(), 0); + assert_eq!(p.next(), Some(0)); + assert_eq!(p.size_hint(), (0, Some(0))); + assert_eq!(p.next(), None); + assert_eq!(p.replica_at(0), 0); + assert_eq!(p.nth(usize::MAX), None); + let mut p = VirtualPermutation::new(100, seed(4)); + assert_eq!(p.nth(12), Some(p.replica_at(12))); + assert_eq!(p.next(), Some(p.replica_at(13))); + assert_eq!(p.nth(usize::MAX), None); + assert_eq!(p.next(), None); + } + + #[test] + #[should_panic(expected = "n must be at least 1")] + fn invalid_empty_domain() { + VirtualPermutation::new(0, 0); + } + + #[test] + #[should_panic(expected = "replica slot must be less than n")] + fn invalid_slot() { + VirtualPermutation::new(10, 0).replica_at(10); + } + + #[test] + fn long_cycles_are_not_truncated() { + use std::cell::Cell; + + struct Rotation<'a> { + mask: u64, + calls: &'a Cell, + } + impl Permutation for Rotation<'_> { + fn forward(&self, x: u64) -> u64 { + self.calls.set(self.calls.get() + 1); + x.wrapping_add(1) & self.mask + } + fn inverse(&self, x: u64) -> u64 { + self.calls.set(self.calls.get() + 1); + x.wrapping_sub(1) & self.mask + } + } + // Every dyadic lift of these ascending cycles is the same ascending + // cycle. A last-slot query just above a half boundary takes long walks. + let calls = Cell::new(0); + let result = evaluate(2049, 2048, |bits| Rotation { + mask: (1 << bits) - 1, + calls: &calls, + }); + assert_eq!(result, 0); + assert!(calls.get() > 4096); + } + + // Deliberately explicit tables only in the oracle, never in the evaluator. + fn project(p: &[u64], n: usize) -> Vec { + (0..n) + .map(|x| { + let mut y = p[x]; + while y >= n as u64 { + y = p[y as usize]; + } + y + }) + .collect() + } + + fn lift(lower: &[u64], q: &[u64]) -> Vec { + let h = lower.len(); + let r = project(q, h); + let mut inverse = vec![0; h]; + for (x, &y) in r.iter().enumerate() { + inverse[y as usize] = x; + } + q.iter() + .map(|&y| { + if y < h as u64 { + lower[inverse[y as usize]] + } else { + y + } + }) + .collect() + } + + #[test] + fn matches_explicit_lift_and_cycle_projection() { + for key in 0..32 { + let key = seed(key); + let mut full = vec![0]; + for bits in 1..=8 { + let p = WordPermutation::new(key, bits); + let q: Vec<_> = (0..1 << bits).map(|x| p.forward(x)).collect(); + full = lift(&full, &q); + for n in (full.len() / 2 + 1)..=full.len() { + let actual: Vec<_> = VirtualPermutation::new(n as u64, key).collect(); + assert_eq!(actual, project(&full, n), "bits={bits} n={n}"); + } + } + } + } + + fn permutations(n: usize) -> Vec> { + fn visit(values: &mut [u64], start: usize, out: &mut Vec>) { + if start == values.len() { + out.push(values.to_vec()); + } else { + for i in start..values.len() { + values.swap(start, i); + visit(values, start + 1, out); + values.swap(start, i); + } + } + } + let mut out = vec![]; + visit(&mut (0..n as u64).collect::>(), 0, &mut out); + out + } + + #[test] + fn ideal_uniform_lift_fibers() { + use std::collections::BTreeMap; + + // Every ideal Q_1,Q_2 combination: F_4 has equal multiplicities. + let mut counts = BTreeMap::new(); + for lower in permutations(2) { + for q in permutations(4) { + *counts.entry(lift(&lower, &q)).or_insert(0) += 1; + } + } + assert_eq!(counts.len(), 24); + assert!(counts.values().all(|&count| count == 2)); + + // Fix F_4; every extension through each intermediate n occurs 4! times + // at n=8, and equally often for n=5,6,7 after cycle projection. + let lower = vec![2, 0, 3, 1]; + let mut counts: Vec, usize>> = (5..=8).map(|_| BTreeMap::new()).collect(); + for q in permutations(8) { + let full = lift(&lower, &q); + assert_eq!(project(&full, 4), lower); + for (i, n) in (5..=8).enumerate() { + *counts[i].entry(project(&full, n)).or_insert(0) += 1; + } + } + for (i, count) in counts.iter().enumerate() { + let expected_distinct: usize = (5..=i + 5).product(); + assert_eq!(count.len(), expected_distinct); + assert!(count.values().all(|&v| v == 40320 / expected_distinct)); + } + } + + #[test] + #[ignore = "deterministic operation-count report; not a timing or statistical CI gate"] + fn operation_count_diagnostics() { + use std::{cell::Cell, rc::Rc}; + + struct Counted { + permutation: WordPermutation, + calls: Rc>, + } + impl Permutation for Counted { + fn forward(&self, x: u64) -> u64 { + let mut calls = self.calls.get(); + calls[0] += 1; + self.calls.set(calls); + self.permutation.forward(x) + } + fn inverse(&self, x: u64) -> u64 { + let mut calls = self.calls.get(); + calls[1] += 1; + self.calls.set(calls); + self.permutation.inverse(x) + } + } + + println!("n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls"); + for n in [ + 1, + 7, + 8, + 9, + 255, + 256, + 257, + 1023, + 1024, + 1025, + 65535, + 65536, + 65537, + (1 << 30) - 1, + 1 << 30, + (1 << 63) + 1, + u64::MAX, + ] { + for slot in [0, n / 2, n - 1] { + let calls = Rc::new(Cell::new([0; 3])); + let mut totals = [0u64; 3]; + let mut samples = vec![]; + for key in 0..10_000 { + calls.set([0; 3]); + let result = evaluate(n, slot, |bits| { + let mut counts = calls.get(); + counts[2] += 1; + calls.set(counts); + Counted { + permutation: WordPermutation::new(seed(key), bits), + calls: Rc::clone(&calls), + } + }); + assert!(result < n); + let counts = calls.get(); + for i in 0..3 { + totals[i] += counts[i]; + } + samples.push(counts[0] + counts[1]); + } + samples.sort_unstable(); + println!( + "{n},{slot},{:.4},{:.4},{:.4},{},{},{}", + totals[0] as f64 / 10_000.0, + totals[1] as f64 / 10_000.0, + totals[2] as f64 / 10_000.0, + samples[4999], + samples[9899], + samples[9999], + ); + } + } + } +} From 316ecdadb614cc6fc1ade91702b6ee58280ee150 Mon Sep 17 00:00:00 2001 From: Alexander Neubeck Date: Thu, 24 Sep 2026 09:35:48 -0700 Subject: [PATCH 2/2] Compare cycle projection with the existing balanced Feistel Add an explicitly experimental two-bit variant using the exact existing permutation and its inverse. Preserve existing mappings and the stronger variant, and rerun three-way timing, held-out statistics, and traversal diagnostics. Document both recovered throughput and repeatable small-domain bias. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- crates/consistent-choose-k/README.md | 11 +- .../benchmarks/replica_comparison.rs | 50 ++- .../docs/virtual-permutation-performance.md | 166 ++++++++- .../docs/virtual-permutation.md | 45 +++ .../examples/permutation_diagnostics.rs | 14 +- .../src/consistent_permutation.rs | 69 +++- crates/consistent-choose-k/src/lib.rs | 2 +- .../src/virtual_permutation.rs | 328 +++++++++++++++--- 8 files changed, 629 insertions(+), 56 deletions(-) diff --git a/crates/consistent-choose-k/README.md b/crates/consistent-choose-k/README.md index 1049a38..3cef1e5 100644 --- a/crates/consistent-choose-k/README.md +++ b/crates/consistent-choose-k/README.md @@ -37,7 +37,7 @@ Why replication matters - Distributes read/write load across multiple owners, reducing hotspots. - Enables fast recovery and higher tail-latency resilience. -## Two permutation APIs, different membership semantics +## Permutation APIs and membership semantics The existing `ConsistentPermutation` preserves **survivor list order** when nodes are appended or removed from the end of `0..n`. The additional, @@ -58,6 +58,15 @@ See the [algorithm, proof assumptions and API guide](docs/virtual-permutation.md and the [reproducible comparison with the existing algorithm](docs/virtual-permutation-performance.md). The existing APIs and their mappings remain unchanged. +`BalancedVirtualPermutation` is a matched-network experiment: it gives the +same cycle/slot semantics as `VirtualPermutation`, but uses **exactly** the +existing `ConsistentPermutation` Feistel, with even widths and two bits per +lift, over `1..=2^30`. Its state is also allocation-free. The +[three-way comparison](docs/virtual-permutation-performance.md#matched-network-follow-up) +repeats performance and primary/held-out statistical diagnostics. This variant +has repeatable small-domain distribution bias and is not the default or a +statistically equivalent replacement for the stronger mixer. + ## Applications beyond replication The `ConsistentChooseK` iterator produces a per-key ranking of all `n` nodes in priority order — consistently and with zero memory overhead. This ranking is a strict superset of simple replication and enables drop-in replacements for several well-known algorithms that traditionally require maintaining expensive data structures such as hash rings. diff --git a/crates/consistent-choose-k/benchmarks/replica_comparison.rs b/crates/consistent-choose-k/benchmarks/replica_comparison.rs index ac9b5af..60ac48e 100644 --- a/crates/consistent-choose-k/benchmarks/replica_comparison.rs +++ b/crates/consistent-choose-k/benchmarks/replica_comparison.rs @@ -7,7 +7,7 @@ use std::{ time::Duration, }; -use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use consistent_choose_k::{BalancedVirtualPermutation, ConsistentPermutation, VirtualPermutation}; use criterion::{criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, Throughput}; use rand::{rngs::StdRng, RngExt, SeedableRng}; @@ -101,6 +101,17 @@ fn end_to_end(c: &mut Criterion) { } }) }); + group.bench_function(BenchmarkId::new("balanced", format!("n{n}_k{k}")), |b| { + b.iter(|| { + for &key in black_box(input) { + let seed = if mode == "fresh" { hash_key(key) } else { key }; + consume( + BalancedVirtualPermutation::new(black_box(n), seed), + black_box(k), + ); + } + }) + }); } } group.finish(); @@ -134,6 +145,13 @@ fn cost_components(c: &mut Criterion) { } }) }); + setup.bench_function(BenchmarkId::new("balanced", n), |b| { + b.iter(|| { + for &seed in black_box(&seeds) { + black_box(BalancedVirtualPermutation::new(black_box(n), seed)); + } + }) + }); } setup.finish(); @@ -204,6 +222,36 @@ fn cost_components(c: &mut Criterion) { }), } }); + group.bench_function(BenchmarkId::new("balanced", format!("n{n}_k{k}")), |b| { + match mode { + "stream_only" => b.iter_batched_ref( + || { + seeds + .iter() + .map(|&seed| BalancedVirtualPermutation::new(n, seed)) + .collect::>() + }, + |iterators| { + for iter in black_box(iterators) { + consume(iter, black_box(k)); + } + }, + BatchSize::SmallInput, + ), + _ => b.iter(|| { + for &seed in black_box(&seeds) { + let iter = BalancedVirtualPermutation::new(black_box(n), seed); + if mode == "collect" { + let mut out = Vec::with_capacity(black_box(k)); + out.extend(iter.take(k)); + black_box(out); + } else { + black_box(iter.replica_at(black_box(k as u64 - 1))); + } + } + }), + } + }); } } group.finish(); diff --git a/crates/consistent-choose-k/docs/virtual-permutation-performance.md b/crates/consistent-choose-k/docs/virtual-permutation-performance.md index 11ebc35..e16731d 100644 --- a/crates/consistent-choose-k/docs/virtual-permutation-performance.md +++ b/crates/consistent-choose-k/docs/virtual-permutation-performance.md @@ -1,9 +1,16 @@ # Replica-selection comparison -The new `VirtualPermutation` trades slower sequential enumeration for +The original `VirtualPermutation` trades slower sequential enumeration for allocation-free state and direct replica-slot evaluation. It is **not a faster drop-in replacement** for the existing `ConsistentPermutation`. +The [matched-network follow-up](#matched-network-follow-up) adds +`BalancedVirtualPermutation`, which uses exactly the existing Feistel on +two-bit layers. All three methods are remeasured together; repeatable +small-domain bias in the matched variant is reported rather than hidden. +The original two-way tables below are historical measurements at commit +`d51abfb`, not fresh measurements of the follow-up. + The existing algorithm preserves survivor **list order**; the new one preserves each old **slot** except a slot replaced by an appended node (or previously naming a removed node). Both provide distinct, `k`-prefix-stable selections, @@ -11,6 +18,10 @@ but these membership policies are not interchangeable. The [design note](virtual-permutation.md) explains the construction, ideal-model proof, noncryptographic approximation and expected-versus-worst-case distinction. +Runner names are `layered` for the existing streaming iterator, `virtual` +for the stronger one-bit cycle construction, and `balanced` for the +matched-Feistel two-bit cycle construction. + ## Reproduce From the repository root: @@ -62,7 +73,7 @@ differences should not be interpreted as portable wins. The fixture is 128 `u64` keys generated by `StdRng` with seed `0x7065726d75746531`. The fresh-key mode hashes each key with `DefaultHasher` -inside every query, identically for both algorithms. Width/round preparation +inside every query, identically for all methods. Width/round preparation and all inverse walks in the new evaluator are included, as is the existing iterator's counter allocation. These are fresh **queries** of a repeated deterministic corpus, not new unpredictable keys on every benchmark iteration. @@ -94,7 +105,7 @@ pairs in each of the fresh and seeded groups. Component groups use `n = 17,257,1000,65537,2^30`. All paired timings stay within the old implementation's supported domain. No wider-domain speedup is inferred. -## Fresh-query results +## Original fresh-query results Representative mean **ns/query**, with bootstrap 95% confidence intervals in brackets. Ratio is new/existing time: **above one means a regression**. @@ -139,7 +150,7 @@ an 8/16/24-round, fully mixed forward/inverse primitive. Avoiding a small state allocation does not make up for that extra arithmetic and traversal. This compares the actual implementations, not algorithms normalized to an identical primitive: both traversal strategy and mixing cost contribute. -All reported timings use the final strengthened schedule, not the rejected +All timings in this original comparison use the final strengthened schedule, not the rejected eight-round version discussed below. ## Separating setup, streaming and direct-slot costs @@ -168,8 +179,9 @@ not be added or subtracted as exact identities: compiler optimization, instruction overlap and cache state differ between the groups. For `n=1000,k=3`, the seeded-query sample standard deviations were 0.54 ns (existing) and 2.95 ns (new); collecting 1,000 replicas had standard deviations -315.22 ns and 4,646.27 ns respectively. All 721 benchmark estimates, including -their standard deviations, can be regenerated with the CSV script. +315.22 ns and 4,646.27 ns respectively. The original run had 721 benchmark +estimates. The current three-method runner emits 1,081 estimates, including +standard deviations, with the same CSV script. Direct access becomes useful for later slots: querying slot 999 at `n=1000` was about 132x faster than replaying the old iterator. Early slots need not @@ -278,3 +290,145 @@ test uses ascending cycles and needs over 4,096 calls for `n=2049,slot=2048`; the evaluator still returns the correct result without a retry cap. Long cycles and linear-in-domain worst cases remain possible. Neither the diagnostics nor the timing confidence intervals establish a latency bound. + +## Matched-network follow-up + +At the user's request, `BalancedVirtualPermutation` now uses exactly the +existing `layer_apply` network: same seed, mixer, balanced halves, round +counts, rotation and Weyl key schedule. The cycle construction descends by +two bits, as the existing iterator does. A new inverse shares the same round +function and reverses that exact key schedule. It does not introduce a +different mixer, extra seed hash, or independently tuned round count. + +The same machine, compiler, flags, key corpus, 131-case matrix and Criterion +settings were used again, sequentially in `layered`, `virtual`, `balanced` +order per case. There are 1,081 estimates across all groups. All three +implementations are rerun; comparisons below use this run, not subtraction of +timings from the earlier run. The shared-host and fixed-corpus limitations +still apply. State is 24 bytes with no allocation for either cycle iterator, +versus 40 bytes plus the heap counters for the existing iterator. + +### Fresh queries: matching the primitive removes much of the overhead + +Mean ns/query [bootstrap 95% interval]. The final ratio is +**matched/existing**; below one favors the matched variant. + +| n | k | Existing | Stronger one-bit | Matched two-bit | Matched/existing | +| ---: | ---: | ---: | ---: | ---: | ---: | +| 8 | 1 | 45.27 [44.67, 45.87] | 110.60 [110.18, 111.13] | 61.42 [60.35, 62.62] | 1.36x | +| 8 | 8 | 205.76 [203.68, 207.92] | 1,030.44 [1,021.06, 1,041.23] | 608.10 [592.46, 623.97] | 2.96x | +| 16 | 3 | 57.51 [57.16, 57.82] | 334.54 [333.57, 335.44] | 67.91 [66.89, 69.14] | 1.18x | +| 17 | 3 | 105.10 [104.22, 105.84] | 619.91 [618.09, 621.79] | 261.59 [258.54, 266.26] | 2.49x | +| 33 | 33 | 637.60 [604.96, 684.35] | 8,604.99 [8,085.82, 9,388.97] | 2,155.89 [2,115.56, 2,192.34] | 3.38x | +| 255 | 3 | 35.09 [34.77, 35.39] | 140.78 [140.32, 141.18] | 30.76 [30.58, 30.97] | 0.88x | +| 256 | 3 | 35.99 [35.64, 36.29] | 139.85 [139.28, 140.62] | 31.11 [30.80, 31.46] | 0.86x | +| 257 | 3 | 61.67 [61.43, 61.91] | 304.03 [302.58, 305.93] | 198.41 [195.88, 201.23] | 3.22x | +| 257 | 8 | 155.53 [151.73, 159.47] | 872.01 [846.66, 902.67] | 536.47 [521.46, 564.32] | 3.45x | +| 1,000 | 1 | 19.75 [19.70, 19.81] | 43.94 [43.47, 44.47] | 16.26 [16.03, 16.48] | 0.82x | +| 1,000 | 3 | 32.83 [32.63, 33.02] | 101.72 [101.21, 102.22] | 27.72 [27.58, 27.86] | 0.84x | +| 1,000 | 16 | 98.73 [98.11, 99.45] | 539.84 [504.74, 605.05] | 137.11 [135.35, 138.99] | 1.39x | +| 1,000 | 250 | 2,059.36 [2,047.90, 2,070.14] | 14,914.99 [14,779.03, 15,043.08] | 3,374.48 [3,336.90, 3,412.61] | 1.64x | +| 1,000 | 1,000 | 9,808.45 [9,658.26, 10,006.25] | 83,713.94 [83,369.99, 84,114.21] | 22,289.60 [22,158.85, 22,422.35] | 2.27x | +| 1,024 | 3 | 33.67 [33.45, 33.88] | 98.44 [97.52, 99.27] | 29.54 [28.99, 30.18] | 0.88x | +| 1,025 | 3 | 65.79 [65.44, 66.12] | 267.49 [266.66, 268.47] | 182.40 [180.91, 184.61] | 2.77x | +| 65,536 | 3 | 33.12 [32.94, 33.28] | 83.63 [83.24, 83.98] | 28.43 [28.16, 28.67] | 0.86x | +| 65,537 | 3 | 65.87 [65.58, 66.16] | 262.49 [262.17, 262.80] | 171.81 [169.97, 174.83] | 2.61x | +| 1,000,000 | 3 | 35.39 [34.87, 36.04] | 95.33 [94.27, 96.33] | 27.43 [27.22, 27.61] | 0.77x | +| 2^30 | 3 | 34.24 [33.94, 34.51] | 90.35 [89.09, 91.52] | 26.34 [26.18, 26.47] | 0.77x | + +At `n=1000,k=3`, matching the network and stride reduces the cycle +construction from 101.72 to 27.72 ns/query (3.67x faster), making it about +16% faster than the existing iterator on this workload. That is 9.24 +ns/replica versus 10.94 existing and 33.91 stronger-one-bit. For a full +1,000-node permutation it is still 2.27x slower than existing (22.29 versus +9.81 ns/replica), although much faster than the stronger variant's 83.71. + +The matched mean is below the existing mean in 31 of 131 cases, including the +degenerate `n=1` case; small differences are not blanket claims of significance. +Its largest observed mean ratio is 3.45x at `n=257,k=8`. Just above powers of +four, the top domain is nearly three-quarters inactive, and cycle walking plus +inverse traversal is still expensive. This experiment changes primitive +**and stride together**; it does not isolate the machine cost of odd halves +alone. It demonstrates that the earlier slowdown was not an unavoidable cost +of cycle consistency, but it does not show that the cycle method always wins. + +The component measurements help distinguish allocation from streaming work. +Below are mean ns/query from the same run, with prehashed seeds; slot queries +return **one** value, while the other `k` rows consume or collect `k` values. + +| Workload, n=1000 | Existing | Stronger one-bit | Matched two-bit | +| --- | ---: | ---: | ---: | +| Constructor + drop | 9.71 | 1.09 | 1.17 | +| Seeded query, k=3 | 31.25 | 103.66 | 27.08 | +| Stream only, k=3 | 14.45 | 89.51 | 20.95 | +| Stream only, k=1000 | 9,541.23 | 86,465.65 | 22,863.95 | +| Collect, k=3 | 42.90 | 112.43 | 36.66 | +| Collect, k=1000 | 9,727.46 | 87,758.08 | 22,936.49 | +| Query slot 2 | 30.03 | 31.54 | 7.89 | +| Query slot 999 | 9,604.66 | 77.43 | 12.69 | + +For `k=3` stream-only, the existing/matched 95% intervals are +[14.25, 14.66] / [20.80, 21.11] ns, and sample standard deviations +0.47 / 0.39 ns. For the seeded complete query, the intervals are +[30.98, 31.47] / [26.92, 27.29] ns. The fresh-query win is therefore consistent +with avoiding allocation, **not** evidence that the cycle traversal itself is +cheaper. These components still cannot be added as exact identities because +their compilation and cache conditions differ. + +The matched direct slot-999 query has interval [12.53, 12.88] ns and sample +standard deviation 0.42 ns. The existing iterator must replay to reach that +slot; this is an API/workload advantage, not a claim of comparable sequential +enumeration speed. + +### Statistical quality is not equivalent + +Both deterministic 200,000-key corpora were rerun without tuning the matched +network. The existing and stronger variants' reported diagnostic rows are +unchanged from the prior run. The matched variant has repeatable deviations: + +| Metric | Degrees of freedom | Matched primary chi-square | Matched held-out chi-square | +| --- | ---: | ---: | ---: | +| n=5, slot 4 | 4 | 82.80 | 84.94 | +| n=5, ordered slots (0,4) | 19 | 98.15 | 98.63 | +| n=8, primary marginal | 7 | 6.21 | 7.26 | +| n=8, ordered slots (0,1) | 55 | 112.66 | 131.31 | +| n=8, first-three subset | 55 | 133.99 | 146.63 | +| n=9, first-three subset | 83 | 148.11 | 141.19 | + +For the `n=5,slot=4` marginal, the most frequent node occurs 41,613 and 41,612 +times against 40,000 expected in each corpus, about a 4% excess. At `n=8`, +the stronger variant's first-three subset statistics remain 47.66 / 40.41, +and the existing iterator's are 53.47 / 59.83, versus the matched variant's +133.99 / 146.63 (all 55 degrees of freedom). + +The same primitive need not have the same statistical behavior after list +projection versus cycle projection. These results do not distinguish +within-width weaknesses from cross-width seed correlations, nor do they +prove exact uniformity for any method. They do show that replacing the +stronger network is **not statistically neutral**. The matched variant is +retained as an explicitly experimental comparison, not silently substituted +for `VirtualPermutation` or promoted as meeting the ideal randomness model. + +### Traversal counts explain remaining boundary costs + +The operation-count diagnostic now prints both cycle variants, using the same +10,000 seeds per `(n,slot)` as before. Means count forward plus inverse calls, +not Feistel rounds or nanoseconds: + +| n | slot | Stronger mean calls | Matched mean calls | Stronger p99 | Matched p99 | Matched observed max | +| ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| 256 | 0 | 1.9808 | 1.3247 | 7 | 4 | 4 | +| 257 | 0 | 4.9857 | 8.2873 | 16 | 34 | 59 | +| 257 | 256 | 7.8930 | 13.2029 | 22 | 41 | 134 | +| 65,536 | 0 | 1.9922 | 1.3330 | 7 | 4 | 8 | +| 65,537 | 0 | 4.9722 | 8.3462 | 16 | 34 | 68 | +| 65,537 | 65,536 | 8.0348 | 13.4067 | 23 | 41 | 73 | + +At full powers of four, primary lookup uses fewer levels; a regression test +also confirms that the matched and existing algorithms give the **same +primary value** there. Just above those boundaries, however, the two-bit +construction must traverse a much larger inactive upper region and then +often retrace it. Its cheaper primitive can still yield faster timings than +the stronger variant despite more primitive calls. These measured tails are +not worst-case guarantees, and the one-bit variant's ideal less-than-eight- +call bound does not apply to the two-bit variant. diff --git a/crates/consistent-choose-k/docs/virtual-permutation.md b/crates/consistent-choose-k/docs/virtual-permutation.md index 2d05a37..ae5b797 100644 --- a/crates/consistent-choose-k/docs/virtual-permutation.md +++ b/crates/consistent-choose-k/docs/virtual-permutation.md @@ -15,6 +15,14 @@ semantics are **different**: | Supported `n` | `1..=2^30` (`u32`) | `1..=u64::MAX` | | Mutable state | Per-layer `Vec` counters | Three `u64` fields; no heap allocation | +An additional **matched-network experiment**, `BalancedVirtualPermutation`, +uses the old iterator's exact Feistel network inside the new cycle-projection +construction. It supports `1..=2^30`, returns `u64` values, and also has 24-byte +allocation-free state. Its `new(n: u32, seed)`, `n`, `replica_at`, `nth` and +iterator operations have the cycle/slot semantics, not survivor-list semantics. +The [matched comparison](virtual-permutation-performance.md#matched-network-follow-up) +records both its performance and its repeatable small-domain statistical bias. + For example, the permutation written as an output list `[2, 0, 1]` is the cycle `0 -> 2 -> 1 -> 0`. Cycle-deleting node 2 produces `[1, 0]`, **not** the survivor list `[0, 1]`. The changed old slot is 0, the slot that @@ -209,6 +217,43 @@ constant-sized width parameter block and no recursive stack, ring, permutation array, or duplicate set. See [measurements and diagnostics](virtual-permutation-performance.md) for observed forward/inverse counts and tails on the actual mixer. +## Matched even-width, two-bit variant + +The cycle-lift identity is not restricted to doubling. For the matched variant +let the full domain have size `4h`, with old set `[0,h)`. Reconnect old +destinations with the same `extend(F_h composed with inverse(R))` operation. +The chain argument and ideal-model uniformity proof above work unchanged. +Start at the smallest **even** bit width covering `n`, use boundary +`1 << (bits - 2)`, and descend by two bits. Intermediate sizes still use cycle +deletion, not output-list deletion. + +`BalancedVirtualPermutation` directly calls the existing `layer_apply` for +every forward evaluation. It uses the same master seed, round function, +12/10/6/4 round schedule at widths 2/4/6/8-and-above, and rotated-key/Weyl +schedule. There is no additional per-width seed hash. The new `layer_inverse` +undoes that exact mapping and key schedule, sharing the extracted round +function. Known-answer vectors, exhaustive small domains, all supported +widths, and extreme keys/inputs check the inverse and unchanged forward +mapping. + +Full-level descent now has probability one quarter in the ideal model, +instead of one half. The initial partial level can be less than half full, +however, so its forward and inverse walks can be longer. Expected work is +still bounded per slot under independent ideal permutations; the specific +less-than-eight-call bound above is for the one-bit variant and is **not** +claimed for the two-bit variant. Inverse evaluation also has real work to +recover the final round key before reversing the rounds; the benchmarks +charge that cost and do not cache it for free. + +This changes **both** layer stride and primitive compared with +`VirtualPermutation`; timing differences are not a pure attribution to odd +versus even halves alone. It does make the primitive and stride identical to +the existing streaming iterator. Neither implementation samples independent +uniform permutations at different widths: the shared 64-bit seed and the +finite Feistel family remain approximations. The matched variant's observed +bias is a substantive limitation, not explained away by the ideal proof. +The stronger one-bit variant remains available unchanged. + ## References and verification The mathematical antecedent is a *virtual permutation*, using **cycle** diff --git a/crates/consistent-choose-k/examples/permutation_diagnostics.rs b/crates/consistent-choose-k/examples/permutation_diagnostics.rs index 20ec0cf..ad1e7d6 100644 --- a/crates/consistent-choose-k/examples/permutation_diagnostics.rs +++ b/crates/consistent-choose-k/examples/permutation_diagnostics.rs @@ -3,7 +3,7 @@ use std::hash::{DefaultHasher, Hash, Hasher}; -use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use consistent_choose_k::{BalancedVirtualPermutation, ConsistentPermutation, VirtualPermutation}; const SAMPLES: u64 = 200_000; fn key_seed(key: u64, workload_seed: u64) -> u64 { @@ -39,13 +39,14 @@ fn main() { ); println!("# samples={SAMPLES}, key_seed={workload_seed:#x}"); println!( - "# iterator_bytes: layered={}, virtual={}; layered also owns heap counters", + "# iterator_bytes: layered={}, virtual={}, balanced={}; layered also owns heap counters", std::mem::size_of::(), std::mem::size_of::(), + std::mem::size_of::(), ); println!("algorithm,n,metric,df,expected_per_cell,chi2,min,max"); for n in [3, 4, 5, 7, 8, 9, 16, 17, 32, 64] { - for algorithm in ["layered", "virtual"] { + for algorithm in ["layered", "virtual", "balanced"] { let slots = n.min(8); let mut marginal = vec![vec![0u64; n]; slots]; let mut pairs = vec![0; n * n]; @@ -60,11 +61,16 @@ fn main() { .take(slots) .map(|v| v as usize) .collect() - } else { + } else if algorithm == "virtual" { VirtualPermutation::new(n as u64, seed) .take(slots) .map(|v| v as usize) .collect() + } else { + BalancedVirtualPermutation::new(n as u32, seed) + .take(slots) + .map(|v| v as usize) + .collect() }; for (histogram, &v) in marginal.iter_mut().zip(&values) { histogram[v] += 1; diff --git a/crates/consistent-choose-k/src/consistent_permutation.rs b/crates/consistent-choose-k/src/consistent_permutation.rs index 4d715c8..fb1d3a5 100644 --- a/crates/consistent-choose-k/src/consistent_permutation.rs +++ b/crates/consistent-choose-k/src/consistent_permutation.rs @@ -75,6 +75,14 @@ fn splitmix64(seed: u64) -> u64 { z ^ (z >> 31) } +#[inline] +fn layer_round(r: u32, key: u64, half_mask: u32) -> u32 { + let k_xor = key & 0xFFFF_FFFF; + let k_mul = (key >> 32) | 1; + let mixed = ((r as u64) ^ k_xor).wrapping_mul(k_mul); + (mixed as u32).wrapping_add((mixed >> 32) as u32) & half_mask +} + /// Apply the per-layer Feistel bijection on `[0, 2^n_bits)`. /// `master_key` must be well-avalanched: two near-identical keys /// will yield two highly correlated permutations. @@ -104,10 +112,7 @@ pub(crate) fn layer_apply(n_bits: u32, master_key: u64, x: u32) -> u32 { // high half (forced odd) is the multiplier. Higher bits stay // set on purpose — they contribute via the multiplicative // mix. - let k_xor = k & 0xFFFF_FFFF; - let k_mul = (k >> 32) | 1; - let mixed = ((r as u64) ^ k_xor).wrapping_mul(k_mul); - let f = (mixed as u32).wrapping_add((mixed >> 32) as u32) & half_mask; + let f = layer_round(r, k, half_mask); let new_l = r; let new_r = l ^ f; l = new_l; @@ -122,6 +127,30 @@ pub(crate) fn layer_apply(n_bits: u32, master_key: u64, x: u32) -> u32 { ((l << half_bits) | r) & n_mask } +/// Invert exactly the existing layer permutation, including its key schedule. +#[inline] +pub(crate) fn layer_inverse(n_bits: u32, master_key: u64, x: u32) -> u32 { + debug_assert!((2..=30).contains(&n_bits) && n_bits.is_multiple_of(2)); + let rounds = rounds_for_n_bits(n_bits); + let half_bits = n_bits / 2; + let half_mask = (1u32 << half_bits) - 1; + debug_assert!(x < 1u32 << n_bits, "input out of range"); + let mut l = x >> half_bits; + let mut r = x & half_mask; + let mut key = master_key; + for _ in 1..rounds { + key = key.rotate_right(n_bits).wrapping_add(0x9E37_79B9_7F4A_7C15); + } + for _ in 0..rounds { + let old_r = l; + let old_l = r ^ layer_round(l, key, half_mask); + l = old_l; + r = old_r; + key = key.wrapping_sub(0x9E37_79B9_7F4A_7C15).rotate_left(n_bits); + } + (l << half_bits) | r +} + /// `n`-consistent permutation iterator over `0..n` driven by one /// bijection per level (see module docs). pub struct ConsistentPermutation { @@ -237,6 +266,38 @@ mod tests { use super::*; + #[test] + fn layer_mapping_known_answers_and_inverse() { + for (bits, key, x, y) in [ + (2, 0, 0, 0), + (4, 0x1234_5678_9abc_def0, 9, 2), + (6, 1, 63, 62), + (8, 0x1234_5678_9abc_def0, 200, 100), + (16, u64::MAX, 65535, 0x3a83), + (30, 0x1234_5678_9abc_def0, 0x3fff_ffff, 0x1949_4328), + ] { + assert_eq!(layer_apply(bits, key, x), y); + assert_eq!(layer_inverse(bits, key, y), x); + } + for bits in (2..=30).step_by(2) { + let mask = (1u32 << bits) - 1; + for key in [0, 1, u64::MAX, splitmix64(42)] { + let inputs: Vec<_> = if bits <= 10 { + (0..=mask).collect() + } else { + [0, 1, mask / 2, mask / 2 + 1, mask] + .into_iter() + .chain((0..128).map(|i| splitmix64(i) as u32 & mask)) + .collect() + }; + for x in inputs { + assert_eq!(layer_inverse(bits, key, layer_apply(bits, key, x)), x); + assert_eq!(layer_apply(bits, key, layer_inverse(bits, key, x)), x); + } + } + } + } + /// Across many seeds, `layer_apply(_, key, 0)` must not always /// collapse to the same value (regression test for the "F(0, k) = /// 0" symmetry bug). diff --git a/crates/consistent-choose-k/src/lib.rs b/crates/consistent-choose-k/src/lib.rs index 4278f83..1b2e45a 100644 --- a/crates/consistent-choose-k/src/lib.rs +++ b/crates/consistent-choose-k/src/lib.rs @@ -12,4 +12,4 @@ pub use consistent_hash::{ pub use consistent_permutation::ConsistentPermutation; pub use consistent_reservoir::ConsistentReservoir; pub use node_map::ConsistentNodeMap; -pub use virtual_permutation::VirtualPermutation; +pub use virtual_permutation::{BalancedVirtualPermutation, VirtualPermutation}; diff --git a/crates/consistent-choose-k/src/virtual_permutation.rs b/crates/consistent-choose-k/src/virtual_permutation.rs index 76be660..d55bd3b 100644 --- a/crates/consistent-choose-k/src/virtual_permutation.rs +++ b/crates/consistent-choose-k/src/virtual_permutation.rs @@ -3,6 +3,8 @@ use std::iter::FusedIterator; +use crate::consistent_permutation::{layer_apply, layer_inverse}; + const WEYL: u64 = 0x9E37_79B9_7F4A_7C15; #[inline] @@ -90,10 +92,21 @@ impl Permutation for WordPermutation { } #[inline] -fn evaluate(mut n: u64, mut x: u64, mut layer: impl FnMut(u32) -> P) -> u64 { +fn evaluate(n: u64, x: u64, layer: impl FnMut(u32) -> P) -> u64 { + evaluate_with_stride::<1, P>(n, x, layer) +} + +#[inline] +fn evaluate_with_stride( + mut n: u64, + mut x: u64, + mut layer: impl FnMut(u32) -> P, +) -> u64 { + debug_assert!(STEP == 1 || STEP == 2); let mut bits = u64::BITS - (n - 1).leading_zeros(); + bits = bits.div_ceil(STEP) * STEP; while bits > 0 { - let half = 1u64 << (bits - 1); + let boundary = 1u64 << (bits - STEP); let permutation = layer(bits); loop { let y = permutation.forward(x); @@ -101,19 +114,19 @@ fn evaluate(mut n: u64, mut x: u64, mut layer: impl FnMut(u32) - x = y; continue; } - if y >= half { + if y >= boundary { return y; } // Find the old input at the start of this Q-chain, not its old // output y: this applies inverse(cycle_projection(Q)) before // evaluating the smaller consistent permutation. - while x >= half { + while x >= boundary { x = permutation.inverse(x); } break; } - n = half; - bits -= 1; + n = boundary; + bits -= STEP; } 0 } @@ -215,6 +228,110 @@ impl Iterator for VirtualPermutation { impl FusedIterator for VirtualPermutation {} +struct ExistingPermutation { + bits: u32, + seed: u64, +} + +impl Permutation for ExistingPermutation { + #[inline] + fn forward(&self, x: u64) -> u64 { + u64::from(layer_apply(self.bits, self.seed, x as u32)) + } + + #[inline] + fn inverse(&self, x: u64) -> u64 { + u64::from(layer_inverse(self.bits, self.seed, x as u32)) + } +} + +/// Experimental cycle-consistent replica slots using exactly the Feistel +/// network of [`crate::ConsistentPermutation`], including its round counts and +/// key schedule, with two bits per lift. +/// +/// This has the slot-consistency semantics of [`VirtualPermutation`], **not** +/// the survivor-list semantics of `ConsistentPermutation`. The supplied +/// well-mixed seed is passed unchanged to the existing network; no additional +/// width-domain seed mixer is inserted. Consequently its statistical quality +/// must be assessed separately from the independently keyed ideal model. +/// It is noncryptographic and does not promise exact uniformity. +/// +/// **Statistical caution:** the matched-network diagnostics show repeatable +/// small-domain bias. This variant is provided for comparison, not as a +/// statistically equivalent substitute for `VirtualPermutation`. +/// +/// State is allocation-free. The supported domain matches the existing +/// network: `1..=2^30` nodes. Iteration returns `u64`, as `VirtualPermutation` +/// does, and direct slot evaluation does not replay earlier slots. +/// +/// ``` +/// use consistent_choose_k::BalancedVirtualPermutation; +/// +/// let permutation = BalancedVirtualPermutation::new(100, 0x1234_5678_9abc_def0); +/// assert_eq!(permutation.clone().nth(2), Some(permutation.replica_at(2))); +/// assert_eq!(permutation.take(3).count(), 3); +/// ``` +#[derive(Clone, Debug)] +pub struct BalancedVirtualPermutation { + inner: VirtualPermutation, +} + +impl BalancedVirtualPermutation { + /// Construct an iterator with the existing Feistel network and seed. + /// + /// # Panics + /// + /// Panics unless `1 <= n <= 2^30`. + pub fn new(n: u32, seed: u64) -> Self { + assert!(n <= 1u32 << 30, "n must be at most 2^30"); + Self { + inner: VirtualPermutation::new(u64::from(n), seed), + } + } + + /// Universe size, independent of the iterator's position. + pub fn n(&self) -> u64 { + self.inner.n() + } + + /// Evaluate an absolute slot without advancing the iterator. + /// + /// # Panics + /// + /// Panics if `slot >= self.n()`. + pub fn replica_at(&self, slot: u64) -> u64 { + assert!(slot < self.inner.n, "replica slot must be less than n"); + evaluate_with_stride::<2, _>(self.inner.n, slot, |bits| ExistingPermutation { + bits, + seed: self.inner.seed, + }) + } +} + +impl Iterator for BalancedVirtualPermutation { + type Item = u64; + + fn next(&mut self) -> Option { + if self.inner.next == self.inner.n { + return None; + } + let value = self.replica_at(self.inner.next); + self.inner.next += 1; + Some(value) + } + + fn size_hint(&self) -> (usize, Option) { + self.inner.size_hint() + } + + fn nth(&mut self, n: usize) -> Option { + self.inner.next = self.inner.next.saturating_add(n as u64).min(self.inner.n); + self.next() + } +} + +impl FusedIterator for BalancedVirtualPermutation {} + #[cfg(test)] mod tests { use super::*; @@ -432,6 +549,115 @@ mod tests { } } + #[test] + fn balanced_permutations_match_explicit_quarter_lifts() { + for key in 0..32 { + let key = seed(key); + let mut full = vec![0]; + for bits in (2..=8).step_by(2) { + let q: Vec<_> = (0..1 << bits) + .map(|x| u64::from(layer_apply(bits, key, x))) + .collect(); + full = lift(&full, &q); + for n in (full.len() / 4 + 1)..=full.len() { + let iter = BalancedVirtualPermutation::new(n as u32, key); + let actual: Vec<_> = iter.clone().collect(); + assert_eq!(actual, project(&full, n)); + let mut sorted = actual.clone(); + sorted.sort_unstable(); + assert_eq!(sorted, (0..n as u64).collect::>()); + for k in [0, 1, n / 2, n] { + assert_eq!(iter.clone().take(k).collect::>(), actual[..k]); + } + for (slot, &value) in actual.iter().enumerate() { + assert_eq!(iter.replica_at(slot as u64), value); + } + let smaller: Vec<_> = + BalancedVirtualPermutation::new(n as u32 - 1, key).collect(); + assert_eq!(project(&actual, n - 1), smaller); + let changed: Vec<_> = smaller + .iter() + .zip(&actual) + .filter(|(old, new)| old != new) + .collect(); + assert!(changed.len() <= 1); + assert!(changed.iter().all(|(_, new)| **new == n as u64 - 1)); + } + } + } + } + + #[test] + fn balanced_large_domains_and_iterator_boundaries() { + for bits in 1..30 { + for n in [(1u32 << bits) - 1, 1 << bits, (1 << bits) + 1] { + for key in [0, 1, u64::MAX, seed(42)] { + let p = BalancedVirtualPermutation::new(n, key); + let larger = BalancedVirtualPermutation::new(n + 1, key); + for slot in [0, u64::from(n / 2), u64::from(n - 1)] { + let old = p.replica_at(slot); + let new = larger.replica_at(slot); + assert!(old < u64::from(n)); + assert!(new == old || new == u64::from(n)); + assert_eq!( + old, + if new == u64::from(n) { + larger.replica_at(new) + } else { + new + } + ); + } + } + } + } + let mut p = BalancedVirtualPermutation::new(1, 0); + assert_eq!(p.n(), 1); + assert_eq!(p.size_hint(), (1, Some(1))); + assert_eq!(p.next(), Some(0)); + assert_eq!(p.size_hint(), (0, Some(0))); + assert_eq!(p.next(), None); + assert_eq!(p.nth(usize::MAX), None); + let mut p = BalancedVirtualPermutation::new(1 << 30, seed(9)); + assert_eq!(p.nth((1 << 30) - 1), Some(p.replica_at((1 << 30) - 1))); + assert_eq!(p.next(), None); + } + + #[test] + fn balanced_primary_matches_existing_at_full_powers_of_four() { + for bits in (0..=30).step_by(2) { + for key in 0..128 { + let key = seed(key); + let n = 1u32 << bits; + let expected = crate::ConsistentPermutation::new(n, key) + .next() + .expect("nonempty domain"); + assert_eq!( + BalancedVirtualPermutation::new(n, key).replica_at(0), + u64::from(expected) + ); + } + } + } + + #[test] + #[should_panic(expected = "n must be at least 1")] + fn balanced_invalid_empty_domain() { + BalancedVirtualPermutation::new(0, 0); + } + + #[test] + #[should_panic(expected = "n must be at most 2^30")] + fn balanced_invalid_large_domain() { + BalancedVirtualPermutation::new((1 << 30) + 1, 0); + } + + #[test] + #[should_panic(expected = "replica slot must be less than n")] + fn balanced_invalid_slot() { + BalancedVirtualPermutation::new(1, 0).replica_at(1); + } + fn permutations(n: usize) -> Vec> { fn visit(values: &mut [u64], start: usize, out: &mut Vec>) { if start == values.len() { @@ -486,11 +712,22 @@ mod tests { fn operation_count_diagnostics() { use std::{cell::Cell, rc::Rc}; - struct Counted { - permutation: WordPermutation, + struct Counted

{ + permutation: P, calls: Rc>, } - impl Permutation for Counted { + impl

Counted

{ + fn new(permutation: P, calls: &Rc>) -> Self { + let mut counts = calls.get(); + counts[2] += 1; + calls.set(counts); + Self { + permutation, + calls: Rc::clone(calls), + } + } + } + impl Permutation for Counted

{ fn forward(&self, x: u64) -> u64 { let mut calls = self.calls.get(); calls[0] += 1; @@ -505,7 +742,9 @@ mod tests { } } - println!("n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls"); + println!( + "algorithm,n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls" + ); for n in [ 1, 7, @@ -525,38 +764,49 @@ mod tests { (1 << 63) + 1, u64::MAX, ] { - for slot in [0, n / 2, n - 1] { - let calls = Rc::new(Cell::new([0; 3])); - let mut totals = [0u64; 3]; - let mut samples = vec![]; - for key in 0..10_000 { - calls.set([0; 3]); - let result = evaluate(n, slot, |bits| { - let mut counts = calls.get(); - counts[2] += 1; - calls.set(counts); - Counted { - permutation: WordPermutation::new(seed(key), bits), - calls: Rc::clone(&calls), + for algorithm in ["virtual", "balanced"] { + if algorithm == "balanced" && n > 1 << 30 { + continue; + } + for slot in [0, n / 2, n - 1] { + let calls = Rc::new(Cell::new([0; 3])); + let mut totals = [0u64; 3]; + let mut samples = vec![]; + for key in 0..10_000 { + calls.set([0; 3]); + let result = if algorithm == "virtual" { + evaluate(n, slot, |bits| { + Counted::new(WordPermutation::new(seed(key), bits), &calls) + }) + } else { + evaluate_with_stride::<2, _>(n, slot, |bits| { + Counted::new( + ExistingPermutation { + bits, + seed: seed(key), + }, + &calls, + ) + }) + }; + assert!(result < n); + let counts = calls.get(); + for i in 0..3 { + totals[i] += counts[i]; } - }); - assert!(result < n); - let counts = calls.get(); - for i in 0..3 { - totals[i] += counts[i]; + samples.push(counts[0] + counts[1]); } - samples.push(counts[0] + counts[1]); + samples.sort_unstable(); + println!( + "{algorithm},{n},{slot},{:.4},{:.4},{:.4},{},{},{}", + totals[0] as f64 / 10_000.0, + totals[1] as f64 / 10_000.0, + totals[2] as f64 / 10_000.0, + samples[4999], + samples[9899], + samples[9999], + ); } - samples.sort_unstable(); - println!( - "{n},{slot},{:.4},{:.4},{:.4},{},{},{}", - totals[0] as f64 / 10_000.0, - totals[1] as f64 / 10_000.0, - totals[2] as f64 / 10_000.0, - samples[4999], - samples[9899], - samples[9999], - ); } } }