From d51abfbbd2235e49aae3049dda5daf852a486b2f Mon Sep 17 00:00:00 2001 From: Alexander Neubeck Date: Thu, 24 Sep 2026 08:26:18 -0700 Subject: [PATCH 1/3] Add cycle-consistent virtual permutations and measured comparison Keep the existing list-consistent iterator unchanged. Add allocation-free slot lookup, invariant tests, deterministic distribution diagnostics, and paired Criterion measurements with documented tradeoffs. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- crates/consistent-choose-k/README.md | 21 + .../consistent-choose-k/benchmarks/Cargo.toml | 6 + .../benchmarks/replica_comparison.rs | 223 +++++++ .../summarize_replica_comparison.py | 58 ++ .../docs/permutation-design.md | 9 +- .../docs/virtual-permutation-performance.md | 280 +++++++++ .../docs/virtual-permutation.md | 233 ++++++++ .../examples/permutation_diagnostics.rs | 99 +++ crates/consistent-choose-k/src/lib.rs | 2 + .../src/virtual_permutation.rs | 563 ++++++++++++++++++ 10 files changed, 1493 insertions(+), 1 deletion(-) create mode 100644 crates/consistent-choose-k/benchmarks/replica_comparison.rs create mode 100644 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py create mode 100644 crates/consistent-choose-k/docs/virtual-permutation-performance.md create mode 100644 crates/consistent-choose-k/docs/virtual-permutation.md create mode 100644 crates/consistent-choose-k/examples/permutation_diagnostics.rs create mode 100644 crates/consistent-choose-k/src/virtual_permutation.rs diff --git a/crates/consistent-choose-k/README.md b/crates/consistent-choose-k/README.md index 45574ab..1049a38 100644 --- a/crates/consistent-choose-k/README.md +++ b/crates/consistent-choose-k/README.md @@ -37,6 +37,27 @@ Why replication matters - Distributes read/write load across multiple owners, reducing hotspots. - Enables fast recovery and higher tail-latency resilience. +## Two permutation APIs, different membership semantics + +The existing `ConsistentPermutation` preserves **survivor list order** when +nodes are appended or removed from the end of `0..n`. The additional, +experimental `VirtualPermutation` instead preserves **replica slots**: on a +single-node append, at most one old slot changes, to the new node; on removal, +only a surviving slot that named that node changes. It does not preserve list +restriction and is not a drop-in replacement for the existing iterator or +its failover policies. Both yield distinct nodes and stable `k` prefixes. + +`VirtualPermutation::new(n, seed)` supports `1..=u64::MAX`, allocates no state, +and offers both an iterator and absolute `replica_at(slot)` lookup. Expected +`O(k)` enumeration follows under ideal independent uniform permutations with +constant-cost forward/inverse evaluation, **not** as a worst-case guarantee. +The implemented noncryptographic, 64-bit seeded Feistel family approximates +that randomness model; exact uniformity and independence are not claimed. + +See the [algorithm, proof assumptions and API guide](docs/virtual-permutation.md) +and the [reproducible comparison with the existing algorithm](docs/virtual-permutation-performance.md). +The existing APIs and their mappings remain unchanged. + ## Applications beyond replication The `ConsistentChooseK` iterator produces a per-key ranking of all `n` nodes in priority order — consistently and with zero memory overhead. This ranking is a strict superset of simple replication and enables drop-in replacements for several well-known algorithms that traditionally require maintaining expensive data structures such as hash rings. diff --git a/crates/consistent-choose-k/benchmarks/Cargo.toml b/crates/consistent-choose-k/benchmarks/Cargo.toml index 5f671fb..a206bce 100644 --- a/crates/consistent-choose-k/benchmarks/Cargo.toml +++ b/crates/consistent-choose-k/benchmarks/Cargo.toml @@ -9,6 +9,12 @@ path = "performance.rs" harness = false test = false +[[bench]] +name = "replica_comparison" +path = "replica_comparison.rs" +harness = false +test = false + [dependencies] consistent-choose-k = { path = "../" } diff --git a/crates/consistent-choose-k/benchmarks/replica_comparison.rs b/crates/consistent-choose-k/benchmarks/replica_comparison.rs new file mode 100644 index 0000000..ac9b5af --- /dev/null +++ b/crates/consistent-choose-k/benchmarks/replica_comparison.rs @@ -0,0 +1,223 @@ +//! Paired workloads only in the existing iterator's supported domain. +//! See ../docs/virtual-permutation-performance.md for methodology and results. + +use std::{ + hash::{DefaultHasher, Hash, Hasher}, + hint::black_box, + time::Duration, +}; + +use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use criterion::{criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, Throughput}; +use rand::{rngs::StdRng, RngExt, SeedableRng}; + +const WORKLOAD_SEED: u64 = 0x7065_726d_7574_6531; +const KEY_COUNT: usize = 128; +const NODES: &[u32] = &[ + 1, + 3, + 7, + 8, + 9, + 15, + 16, + 17, + 31, + 32, + 33, + 255, + 256, + 257, + 1000, + 1023, + 1024, + 1025, + 65535, + 65536, + 65537, + 1_000_000, + (1 << 30) - 1, + 1 << 30, +]; + +fn hash_key(key: u64) -> u64 { + let mut hasher = DefaultHasher::new(); + key.hash(&mut hasher); + hasher.finish() +} + +fn keys() -> Vec { + StdRng::seed_from_u64(WORKLOAD_SEED) + .random_iter() + .take(KEY_COUNT) + .collect() +} + +fn counts(n: u32) -> Vec { + let mut counts = vec![1, 2, 3, 8, 16]; + if n <= 1024 { + counts.extend([n as usize / 4, n as usize]); + } + counts.retain(|&k| k > 0 && k <= n as usize); + counts.sort_unstable(); + counts.dedup(); + counts +} + +fn consume(iter: impl Iterator>, k: usize) { + let sum = iter + .take(k) + .fold(0u64, |sum, node| sum.wrapping_add(node.into())); + black_box(sum); +} + +fn end_to_end(c: &mut Criterion) { + let keys = keys(); + let seeds: Vec<_> = keys.iter().copied().map(hash_key).collect(); + for mode in ["fresh", "seeded"] { + let mut group = c.benchmark_group(format!("replicas/{mode}")); + // Both execute exactly one complete query per key, including + // construction, streaming consumption and state destruction. + group.throughput(Throughput::Elements(KEY_COUNT as u64)); + for &n in NODES { + for k in counts(n) { + let input = if mode == "fresh" { &keys } else { &seeds }; + group.bench_function(BenchmarkId::new("layered", format!("n{n}_k{k}")), |b| { + b.iter(|| { + for &key in black_box(input) { + let seed = if mode == "fresh" { hash_key(key) } else { key }; + consume(ConsistentPermutation::new(black_box(n), seed), black_box(k)); + } + }) + }); + group.bench_function(BenchmarkId::new("virtual", format!("n{n}_k{k}")), |b| { + b.iter(|| { + for &key in black_box(input) { + let seed = if mode == "fresh" { hash_key(key) } else { key }; + consume( + VirtualPermutation::new(u64::from(black_box(n)), seed), + black_box(k), + ); + } + }) + }); + } + } + group.finish(); + } +} + +fn cost_components(c: &mut Criterion) { + let keys = keys(); + let seeds: Vec<_> = keys.iter().copied().map(hash_key).collect(); + let mut setup = c.benchmark_group("replicas/setup"); + setup.throughput(Throughput::Elements(KEY_COUNT as u64)); + setup.bench_function("hash_u64", |b| { + b.iter(|| { + for &key in black_box(&keys) { + black_box(hash_key(key)); + } + }) + }); + for &n in &[17, 257, 1000, 65537, 1 << 30] { + setup.bench_function(BenchmarkId::new("layered", n), |b| { + b.iter(|| { + for &seed in black_box(&seeds) { + black_box(ConsistentPermutation::new(black_box(n), seed)); + } + }) + }); + setup.bench_function(BenchmarkId::new("virtual", n), |b| { + b.iter(|| { + for &seed in black_box(&seeds) { + black_box(VirtualPermutation::new(u64::from(black_box(n)), seed)); + } + }) + }); + } + setup.finish(); + + for mode in ["stream_only", "collect", "slot"] { + let mut group = c.benchmark_group(format!("replicas/{mode}")); + group.throughput(Throughput::Elements(KEY_COUNT as u64)); + for &n in &[17, 257, 1000, 65537, 1 << 30] { + for k in counts(n) { + group.bench_function(BenchmarkId::new("layered", format!("n{n}_k{k}")), |b| { + match mode { + "stream_only" => b.iter_batched_ref( + || { + seeds + .iter() + .map(|&seed| ConsistentPermutation::new(n, seed)) + .collect::>() + }, + |iterators| { + for iter in black_box(iterators) { + consume(iter, black_box(k)); + } + }, + BatchSize::SmallInput, + ), + _ => b.iter(|| { + for &seed in black_box(&seeds) { + let mut iter = ConsistentPermutation::new(black_box(n), seed); + if mode == "collect" { + // Same output width and allocation policy for both. + let mut out = Vec::with_capacity(black_box(k)); + out.extend(iter.take(k).map(u64::from)); + black_box(out); + } else { + // No random-access API in the baseline: nth must replay. + black_box(iter.nth(black_box(k - 1))); + } + } + }), + } + }); + group.bench_function(BenchmarkId::new("virtual", format!("n{n}_k{k}")), |b| { + match mode { + "stream_only" => b.iter_batched_ref( + || { + seeds + .iter() + .map(|&seed| VirtualPermutation::new(u64::from(n), seed)) + .collect::>() + }, + |iterators| { + for iter in black_box(iterators) { + consume(iter, black_box(k)); + } + }, + BatchSize::SmallInput, + ), + _ => b.iter(|| { + for &seed in black_box(&seeds) { + let iter = VirtualPermutation::new(u64::from(black_box(n)), seed); + if mode == "collect" { + let mut out = Vec::with_capacity(black_box(k)); + out.extend(iter.take(k)); + black_box(out); + } else { + black_box(iter.replica_at(black_box(k as u64 - 1))); + } + } + }), + } + }); + } + } + group.finish(); + } +} + +criterion_group! { + name = benches; + config = Criterion::default() + .sample_size(20) + .warm_up_time(Duration::from_millis(100)) + .measurement_time(Duration::from_millis(300)) + .nresamples(1000) + .without_plots(); + targets = end_to_end, cost_components +} +criterion_main!(benches); diff --git a/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py b/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py new file mode 100644 index 0000000..0fa9f65 --- /dev/null +++ b/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py @@ -0,0 +1,58 @@ +#!/usr/bin/env python3 +"""Convert Criterion's batch estimates to ns/query and ns/replica CSV. + +Usage: python3 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py \ + target/criterion > comparison.csv +""" + +import csv +import json +from pathlib import Path +import re +import sys + + +def summarize(root): + rows = [] + for metadata in root.glob("**/new/benchmark.json"): + benchmark = json.loads(metadata.read_text()) + group = benchmark["group_id"] + if not group.startswith("replicas/"): + continue + mode = group.removeprefix("replicas/") + algorithm = benchmark["function_id"] + value = benchmark.get("value_str") or "" + match = re.fullmatch(r"n(\d+)_k(\d+)", value) + n, k = map(int, match.groups()) if match else (int(value or 0), 0) + estimates = json.loads((metadata.parent / "estimates.json").read_text()) + mean = estimates["mean"] + interval = mean["confidence_interval"] + divisor = benchmark["throughput"]["Elements"] + rows.append( + ( + mode, + algorithm, + n, + k, + mean["point_estimate"] / divisor, + interval["lower_bound"] / divisor, + interval["upper_bound"] / divisor, + estimates["std_dev"]["point_estimate"] / divisor, + mean["point_estimate"] / divisor / (k if k and mode != "slot" else 1), + ) + ) + if not rows: + raise SystemExit(f"No replica_comparison results found in {root}") + writer = csv.writer(sys.stdout) + writer.writerow( + ["mode", "algorithm", "n", "k", "ns_query", "ci95_low", "ci95_high", + "stddev_ns", "ns_replica"] + ) + for row in sorted(rows): + writer.writerow([*row[:4], *(f"{value:.3f}" for value in row[4:])]) + + +if __name__ == "__main__": + if len(sys.argv) != 2: + raise SystemExit(__doc__) + summarize(Path(sys.argv[1])) diff --git a/crates/consistent-choose-k/docs/permutation-design.md b/crates/consistent-choose-k/docs/permutation-design.md index 6ff55dd..9fd144c 100644 --- a/crates/consistent-choose-k/docs/permutation-design.md +++ b/crates/consistent-choose-k/docs/permutation-design.md @@ -4,6 +4,14 @@ This document explains the design of [`ConsistentPermutation`], the per-layer Feistel permutation iterator that this crate uses to drive its `n`-consistent ranking. +This is **survivor-list consistency**, not the cycle-projection/replica-slot +consistency of the additional [`VirtualPermutation`](virtual-permutation.md). +The algorithms are not interchangeable membership policies. See their +[paired performance comparison](virtual-permutation-performance.md). +Uniformity arguments below model the per-layer bijections as independent +uniform permutations; the actual finite-key, noncryptographic Feistel family +is a practical approximation, not an exact uniform sample from all permutations. + Given a 64-bit `key` and a universe size `n`, the iterator produces a uniformly distributed permutation of `[0, n)` as a streaming iterator satisfying: @@ -450,4 +458,3 @@ Observations matching the theory: pre-build option is `O(k log k)`), which is why the speedup ratio widens with `k`: at `k = 1 000, n = 1 000` the permutation implementation is ~50× faster. - diff --git a/crates/consistent-choose-k/docs/virtual-permutation-performance.md b/crates/consistent-choose-k/docs/virtual-permutation-performance.md new file mode 100644 index 0000000..11ebc35 --- /dev/null +++ b/crates/consistent-choose-k/docs/virtual-permutation-performance.md @@ -0,0 +1,280 @@ +# Replica-selection comparison + +The new `VirtualPermutation` trades slower sequential enumeration for +allocation-free state and direct replica-slot evaluation. It is **not a faster +drop-in replacement** for the existing `ConsistentPermutation`. + +The existing algorithm preserves survivor **list order**; the new one preserves +each old **slot** except a slot replaced by an appended node (or previously +naming a removed node). Both provide distinct, `k`-prefix-stable selections, +but these membership policies are not interchangeable. The +[design note](virtual-permutation.md) explains the construction, ideal-model +proof, noncryptographic approximation and expected-versus-worst-case distinction. + +## Reproduce + +From the repository root: + +```sh +cargo test -p consistent-choose-k +cargo bench -p consistent-choose-k-benchmarks --bench replica_comparison -- --noplot +python3 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py \ + target/criterion > comparison.csv +cargo run --release -p consistent-choose-k --example permutation_diagnostics +cargo run --release -p consistent-choose-k --example permutation_diagnostics -- --held-out +cargo test --release -p consistent-choose-k operation_count_diagnostics \ + -- --ignored --nocapture +``` + +Run these **sequentially**, not alongside other CPU-intensive tasks. Criterion +filters can restrict a rerun, for example +`-- 'replicas/fresh/.*/n1000_k3$' --noplot`. +Raw sample timings and estimates remain under `target/criterion`; the script +emits mean ns/query, bootstrap 95% confidence interval, sample standard deviation +in ns/query, and mean ns/replica. Criterion's console times are for a **batch of +128 queries**, so divide by 128, not by `k`, to obtain ns/query. Divide again by +`k` for ns/replica (a slot query returns just one result). + +### Recorded environment + +Measured on September 24, 2026, on an Apple M4 Max, native +`aarch64-apple-darwin`, macOS 27.0 build 26A428: + +- `rustc 1.92.0 (ded5c06cf 2025-12-08)`, LLVM 21.1.3. +- Apple clang 21.0.0 (`clang-2100.1.1.101`). +- Repository bench profile: optimized with debug info; configured + `-C target-feature=+neon`; no extra LTO, native-CPU flags or PGO. +- Criterion 0.8.2, rand 0.10.3; 20 samples, 100 ms warmup, 300 ms target + measurement time per case, 1,000 bootstrap resamples, no plots. +- The default selected Xcode could not link because its license was + unaccepted. All successful commands used the independently installed + Command Line Tools via + `DEVELOPER_DIR=/Library/Developer/CommandLineTools`. + No license was accepted and no system setting was changed. + +These are bounded microbenchmarks on a shared host, without CPU pinning, +frequency control or isolation. Intervals describe repeated timing samples of +one fixed key corpus, **not** uncertainty across all keys, machines or compiler +versions. Algorithms ran sequentially, existing first in each pair. Small +differences should not be interpreted as portable wins. + +### Workload and accounting + +The fixture is 128 `u64` keys generated by `StdRng` with seed +`0x7065726d75746531`. The fresh-key mode hashes each key with `DefaultHasher` +inside every query, identically for both algorithms. Width/round preparation +and all inverse walks in the new evaluator are included, as is the existing +iterator's counter allocation. These are fresh **queries** of a repeated +deterministic corpus, not new unpredictable keys on every benchmark iteration. +`StdRng` and `DefaultHasher` are not cross-version reproducibility contracts; +use the recorded toolchain and dependency versions to reproduce the exact +corpus and mapping. + +| Group | Timed work | +| --- | --- | +| `fresh` (primary comparison) | Key hashing, construction, streaming checksum of `k` values, destruction | +| `seeded` | Same query, reusing prehashed seeds; no free algorithm-specific cache | +| `setup` | Hash alone, or construction and destruction from a prehashed seed | +| `stream_only` | Consume `k` values from separately prepared iterators; construction and destruction excluded equally with `iter_batched_ref` | +| `collect` | Prehashed construction, allocate/fill/drop `Vec` with the same capacity `k`, destroy iterator | +| `slot` | Prehashed construction plus the result at `k-1`; existing `.nth(k-1)` replays, new `replica_at(k-1)` does not | + +The last row is a comparison of ways to answer a single-slot query, **not** a +claim that the existing API has a constant-time random-access operation. +Streaming-only is a diagnostic component benchmark, not a substitute for the +fresh-query comparison: prepared state has different cache footprints. +Inputs pass through `black_box`, and outputs are consumed. Both collection +paths use `u64` output elements, even though the old API returns `u32`. + +The full paired matrix uses +`n = 1,3,7,8,9,15,16,17,31,32,33,255,256,257,1000,1023,1024,1025,65535,65536,65537,1000000,2^30-1,2^30`. +For each, it uses valid `k` from `1,2,3,8,16`; at `n <= 1024`, it also uses +`floor(n/4)` and `n`, removing zeros and duplicates. There are 131 `(n,k)` +pairs in each of the fresh and seeded groups. Component groups use +`n = 17,257,1000,65537,2^30`. All paired timings stay within the old +implementation's supported domain. No wider-domain speedup is inferred. + +## Fresh-query results + +Representative mean **ns/query**, with bootstrap 95% confidence intervals in +brackets. Ratio is new/existing time: **above one means a regression**. + +| n | k | Existing layered | New virtual | Ratio | +| ---: | ---: | ---: | ---: | ---: | +| 1 | 1 | 58.04 [57.76, 58.27] | 7.38 [7.08, 7.82] | 0.13x | +| 8 | 1 | 51.31 [50.97, 51.63] | 114.80 [114.46, 115.16] | 2.24x | +| 8 | 8 | 227.41 [223.70, 230.99] | 1,005.85 [1,000.85, 1,014.19] | 4.42x | +| 17 | 3 | 110.06 [108.94, 111.07] | 622.33 [619.42, 625.40] | 5.65x | +| 33 | 33 | 600.19 [593.91, 604.57] | 7,988.45 [7,913.42, 8,104.45] | 13.31x | +| 255 | 3 | 38.10 [37.82, 38.39] | 142.31 [141.60, 143.18] | 3.74x | +| 256 | 3 | 39.69 [39.27, 40.12] | 143.02 [142.31, 143.75] | 3.60x | +| 257 | 3 | 63.41 [62.81, 63.97] | 305.48 [302.11, 310.19] | 4.82x | +| 1,000 | 1 | 22.39 [21.38, 24.14] | 49.53 [47.21, 52.43] | 2.21x | +| 1,000 | 3 | 35.58 [35.43, 35.72] | 104.90 [102.88, 107.36] | 2.95x | +| 1,000 | 16 | 103.18 [102.78, 103.58] | 536.52 [519.93, 561.57] | 5.20x | +| 1,000 | 250 | 2,042.64 [2,013.97, 2,070.34] | 14,996.65 [14,959.80, 15,025.20] | 7.34x | +| 1,000 | 1,000 | 9,735.29 [9,668.93, 9,808.83] | 85,990.56 [84,682.97, 87,389.54] | 8.83x | +| 1,023 | 3 | 36.29 [35.55, 37.30] | 106.26 [99.49, 118.27] | 2.93x | +| 1,024 | 3 | 35.79 [35.60, 35.99] | 97.85 [96.85, 98.99] | 2.73x | +| 1,025 | 3 | 64.33 [64.06, 64.61] | 266.90 [265.96, 268.03] | 4.15x | +| 65,535 | 3 | 35.19 [35.02, 35.39] | 88.23 [87.26, 89.16] | 2.51x | +| 65,536 | 3 | 36.40 [36.06, 36.77] | 90.23 [89.02, 91.43] | 2.48x | +| 65,537 | 3 | 66.23 [65.47, 66.87] | 268.53 [265.08, 272.36] | 4.05x | +| 1,000,000 | 3 | 36.14 [35.67, 36.65] | 96.65 [95.81, 97.70] | 2.67x | +| 2^30 - 1 | 3 | 38.49 [38.01, 38.91] | 87.74 [87.45, 88.21] | 2.28x | +| 2^30 | 3 | 38.86 [38.60, 39.12] | 89.96 [89.23, 90.68] | 2.31x | + +For example, `n=1000,k=3` is 11.86 versus 34.97 **ns/replica**; +`k=1000` is 9.74 versus 85.99 ns/replica. A full permutation has more +upper-half input slots, which need inverse walks; a short initial prefix often +avoids that work. Consequently mean cost per replica is not constant across +all `k`, even though the ideal-model expectation is bounded uniformly. + +Across the entire primary matrix the measured ratio ranges from 0.13x +(`n=1,k=1`, a degenerate constant-output case) to 13.31x +(`n=33,k=33`). The new implementation was slower in all 130 nontrivial cases. +The existing code's inexpensive four-round upper-layer +primitive and shared streaming counters generally beat repeated evaluation of +an 8/16/24-round, fully mixed forward/inverse primitive. Avoiding a small state +allocation does not make up for that extra arithmetic and traversal. +This compares the actual implementations, not algorithms normalized to an +identical primitive: both traversal strategy and mixing cost contribute. +All reported timings use the final strengthened schedule, not the rejected +eight-round version discussed below. + +## Separating setup, streaming and direct-slot costs + +Mean ns/query [95% confidence interval]. All rows below are prehashed; in the +`slot` rows, `k` means "query slot `k-1`", **not** consume `k` replicas. + +| Mode | n | k | Existing layered | New virtual | +| --- | ---: | ---: | ---: | ---: | +| Constructor + drop | 1,000 | - | 9.73 [9.62, 9.83] | 1.11 [1.10, 1.12] | +| Constructor + drop | 2^30 | - | 9.77 [9.72, 9.84] | 1.08 [1.08, 1.09] | +| Seeded stream query | 1,000 | 3 | 32.45 [32.21, 32.66] | 103.99 [102.70, 105.32] | +| Seeded stream query | 1,000 | 1,000 | 9,737.78 [9,680.29, 9,789.97] | 85,750.68 [85,646.50, 85,868.42] | +| Stream only | 1,000 | 3 | 14.62 [14.45, 14.82] | 91.78 [91.32, 92.34] | +| Stream only | 1,000 | 1,000 | 9,681.61 [9,618.72, 9,750.56] | 86,659.57 [86,278.25, 87,113.38] | +| Collect | 1,000 | 3 | 43.94 [43.02, 45.48] | 113.49 [112.83, 114.09] | +| Collect | 1,000 | 1,000 | 9,895.86 [9,786.30, 10,060.80] | 90,027.62 [88,127.94, 92,190.79] | +| Slot | 1,000 | 3 | 29.93 [29.79, 30.07] | 30.83 [30.69, 30.97] | +| Slot | 1,000 | 16 | 94.70 [93.68, 95.85] | 28.88 [28.76, 28.98] | +| Slot | 1,000 | 250 | 2,088.37 [2,042.18, 2,135.62] | 50.60 [50.48, 50.77] | +| Slot | 1,000 | 1,000 | 9,498.73 [9,419.35, 9,599.28] | 72.18 [72.02, 72.36] | +| Slot | 65,537 | 3 | 60.03 [59.29, 60.85] | 79.55 [78.88, 80.39] | + +Hashing a `u64` alone measured 7.48 [7.46, 7.50] ns. Component costs should +not be added or subtracted as exact identities: compiler optimization, +instruction overlap and cache state differ between the groups. +For `n=1000,k=3`, the seeded-query sample standard deviations were 0.54 ns +(existing) and 2.95 ns (new); collecting 1,000 replicas had standard deviations +315.22 ns and 4,646.27 ns respectively. All 721 benchmark estimates, including +their standard deviations, can be regenerated with the CSV script. + +Direct access becomes useful for later slots: querying slot 999 at `n=1000` +was about 132x faster than replaying the old iterator. Early slots need not +benefit, as the slot-2 rows show. This does not change the sequential-enumeration +regression, and the saved allocation is not being silently excluded from the +primary comparison. + +## State and allocations + +On this target, `size_of::()` is 40 bytes plus one +allocation for `4 * max(1, ceil(log2(n)/2))` bytes of counters: 20 bytes at +`n=1000`, 36 at `n=65537`, and 60 at `n=2^30`, excluding allocator metadata. +`size_of::()` is 24 bytes and it has no heap allocation. +It uses an additional constant-sized parameter block and scalars on the stack +during evaluation, not a recursion stack or a per-`n` cache. + +These allocation counts follow the source, rather than a custom allocator +installed in the timed benchmark. The diagnostic example prints the actual +struct sizes. The collection benchmark adds one capacity-`k` `Vec` +allocation to **both** algorithms, so it has two total allocations for the +existing iterator and one for the new one. Optional output storage is `8k` +bytes in both cases. + +## Distribution diagnostics and the rejected first version + +The deterministic example evaluates 200,000 hashed seeds per method for +`n = 3,4,5,7,8,9,16,17,32,64`. It reports each of the first up to eight slot +marginals, ordered pairs `(slot 0, slot 1)` and `(slot 0, last sampled slot)`, +unordered first-three subsets, and consecutive-key primary pairs. Expected +cells exclude repeated nodes for within-key pairs and include them for +cross-key pairs. All nontrivial cells are printed even when sparse (for +example, the `n=64` choose-three diagnostic has only 4.80 expected per cell). +These are exploratory diagnostics, not random p-value CI gates or a proof of +independence; the histograms overlap and are not independent tests. + +The primary corpus hashes `0x7065726d75746531 XOR i`, `i=0..199999`, with +`DefaultHasher`. A disjoint, deterministic held-out corpus uses +`0x686f6c646f757431 XOR i`. The held-out corpus was first evaluated **after** +fixing the stronger schedule; no additional tuning followed its results. +It is independent input data for diagnosis, not a claim that a deterministic +hash function supplies mathematically independent random variables. + +The initial **eight-round-at-every-width** implementation was rejected: +at `n=8`, its ordered `(0,1)` pair statistic was 1753.19 on 55 degrees of +freedom even though its primary marginal statistic was 4.24 on 7 degrees of +freedom. Clean marginals were insufficient. The final schedule uses 24 rounds +at widths 2--4, 16 at 5--7 and 8 at 8--64, fixed solely by width. This +strengthens mixing instead of reducing rounds to improve performance. + +| Metric, n=8 | Existing primary / held-out chi-square | Final virtual primary / held-out chi-square | Degrees of freedom | +| --- | ---: | ---: | ---: | +| Primary marginal | 4.79 / 4.66 | 8.63 / 1.22 | 7 | +| Slot 7 marginal | 11.50 / 5.73 | 6.44 / 1.26 | 7 | +| Ordered slots (0,1) | 65.97 / 43.73 | 54.15 / 44.41 | 55 | +| Ordered slots (0,7) | 66.92 / 44.74 | 60.47 / 45.43 | 55 | +| Unordered first-three subset | 53.47 / 59.83 | 47.66 / 40.41 | 55 | +| Consecutive-key primaries | 50.26 / 61.98 | 68.31 / 44.11 | 63 | + +At `n=32`, the final virtual first-three subset statistics are 4863.03 and +4873.25 on 4959 degrees of freedom, versus 5068.47 and 5024.53 for the existing +implementation. Its ordered `(0,1)` statistics are 1033.88 and 920.17 on 991 +degrees of freedom, versus 1032.74 and 985.50 for the existing implementation. + +No similarly large deviations appeared in the final virtual family's measured +marginals, ordered pairs, subsets or consecutive-key pairs on either corpus. +That observation does not establish exact uniformity, bound unseen-key bias, +or validate cryptographic properties. + +The **unchanged existing implementation** also has detectable deviations in +these broader diagnostics: at `n=9`, slot 4 has chi-square 253.52 on 8 degrees +of freedom in the primary corpus and 167.39 in the held-out corpus; slots 3 +and 5 also deviate. This is a limitation of the measured baseline, not a +reason to alter its behavior in this PR or to declare the new algorithm +uniform. Timing and statistical evidence answer different questions. + +## Untimed operation counts and tails + +The ignored diagnostic test wraps the **same evaluator and primitive** with +forward/inverse counters, outside any timed benchmark. For each `(n,slot)`, +it uses 10,000 deterministic SplitMix64-mixed seeds from integers `0..9999` +(the test source fixes the exact mixer and offset). Percentiles are nearest- +rank percentiles of total forward plus inverse calls; one call can contain +8, 16 or 24 Feistel rounds depending on width. + +| n | slot | Mean forward | Mean inverse | Mean visited levels | p50 calls | p99 calls | Observed max | +| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| 256 | 0 | 1.9808 | 0 | 1.9808 | 1 | 7 | 8 | +| 257 | 0 | 3.9806 | 1.0051 | 2.9715 | 4 | 16 | 28 | +| 257 | 256 | 3.9660 | 3.9270 | 2.9786 | 7 | 22 | 41 | +| 65,536 | 0 | 1.9922 | 0 | 1.9922 | 1 | 7 | 15 | +| 65,537 | 0 | 3.9822 | 0.9900 | 2.9922 | 4 | 16 | 30 | +| 65,537 | 65,536 | 4.0104 | 4.0244 | 3.0065 | 7 | 23 | 36 | +| 2^30 | 2^29 | 1.9907 | 1.4876 | 1.9907 | 1 | 16 | 49 | +| 2^63 + 1 | 2^63 | 3.9795 | 4.0169 | 2.9878 | 7 | 23 | 40 | +| u64::MAX | 0 | 2.0194 | 0 | 2.0194 | 2 | 7 | 16 | + +The last two rows are **unpaired, wider-domain diagnostics**, not performance +comparisons with the old API. Near a half-full top domain, traversal work +increases substantially; constant expected work does not mean flat latency. +The 8.0348-call sample mean at `n=65537,slot=65536` is not a contradiction of +the ideal-model bound: it is a finite sample of a finite-key approximation, +not the ideal ensemble expectation. + +Observed maxima are not guarantees. A deterministic adversarial-oracle unit +test uses ascending cycles and needs over 4,096 calls for `n=2049,slot=2048`; +the evaluator still returns the correct result without a retry cap. Long +cycles and linear-in-domain worst cases remain possible. Neither the +diagnostics nor the timing confidence intervals establish a latency bound. diff --git a/crates/consistent-choose-k/docs/virtual-permutation.md b/crates/consistent-choose-k/docs/virtual-permutation.md new file mode 100644 index 0000000..2d05a37 --- /dev/null +++ b/crates/consistent-choose-k/docs/virtual-permutation.md @@ -0,0 +1,233 @@ +# VirtualPermutation: cycle-consistent replica slots + +`VirtualPermutation` is an additional experimental algorithm, not a replacement +for `ConsistentPermutation`. Both produce deterministic, distinct candidates, +and every `k`-selection is a prefix of the same per-key list. Their membership +semantics are **different**: + +| Property | Existing `ConsistentPermutation` | New `VirtualPermutation` | +| --- | --- | --- | +| Append one node | Insert it into the ranking; later replica slots can shift | At most one old replica slot changes, to the new node | +| Remove the last node | Remove it from the ranking; later slots can shift | Only a surviving slot that named the removed node changes | +| Survivor list order | Preserved | Not promised | +| Projection between sizes | Delete entries from the output list | Delete labels from the permutation's cycles | +| Direct slot lookup | Replay iterator through that slot | `replica_at(slot)`, without replay | +| Supported `n` | `1..=2^30` (`u32`) | `1..=u64::MAX` | +| Mutable state | Per-layer `Vec` counters | Three `u64` fields; no heap allocation | + +For example, the permutation written as an output list `[2, 0, 1]` is the +cycle `0 -> 2 -> 1 -> 0`. Cycle-deleting node 2 produces `[1, 0]`, +**not** the survivor list `[0, 1]`. The changed old slot is 0, the slot that +named the removed node. This difference matters for failover and bounded-load +policies that depend on survivor priority; do not substitute the new API into +`ConsistentNodeMap` or existing consumers without considering their semantics. + +Membership is consecutive IDs `0..n`. Additions append IDs and removals remove +a suffix; arbitrary holes, weights and physical-node remapping are out of scope. +The one-slot bound is per single-node change, not for an entire suffix at once. + +## API and randomness model + +```rust +use consistent_choose_k::VirtualPermutation; +use std::hash::{DefaultHasher, Hash, Hasher}; + +let mut hasher = DefaultHasher::new(); +"object-key".hash(&mut hasher); +let seed = hasher.finish(); +let permutation = VirtualPermutation::new(1_000, seed); +let third_replica = permutation.replica_at(2); +let replicas: Vec = permutation.take(3).collect(); +assert_eq!(third_replica, replicas[2]); +``` + +Like the existing iterator, the constructor accepts an already well-mixed +64-bit seed, not an arbitrary application key. Use the same seed for all `n` +and `k`. `DefaultHasher` above is convenient for local examples and matches the +benchmark convention; Rust does not promise its mapping is stable across +versions. Distributed deployments need a specified, versioned key hash and +identical algorithm versions on every participant. + +`new(0, seed)` and `replica_at(slot >= n)` panic, following the existing +constructor's assertion convention. `replica_at` is absolute, independent of +the cursor. The iterator is cloneable and fused, reports its remaining size, +and implements `nth` without replay. `.take(0)` is empty, `.take(n)` is the full +permutation, and `.take(k)` for `k > n` stops at exhaustion, as usual for Rust +iterators. There is no separate `k` constructor argument. + +Under **independent uniform ideal permutations for each key and bit width**, +the construction below gives a uniform permutation for every fixed `n`: +each ordered `k`-tuple of distinct nodes is equally likely, and different keys' +preference permutations are independent. Within a key, replicas are sampled +without replacement, not independently. + +The actual primitive is an alternating-XOR Feistel network with SplitMix64's +avalanche finalizer as its round mixer. It uses 24 rounds for widths 2 through +4, 16 for widths 5 through 7, and 8 for widths 8 through 64. Tiny half-domains +need extra rounds: the initial eight-round version had a strong ordered-pair +bias at width three despite clean marginals. This fixed policy depends **only** +on bit width, never active `n`, requested `k`, or observed outputs. Unequal +halves support odd widths; the one-bit case is a keyed XOR. +Widths are domain-separated through a +mixed seed; round keys use distinct Weyl offsets. The inverse undoes the same +updates in reverse order. Geometry never requires a shift by 64: the largest +half-width is 32 and the evaluator's largest half boundary is `1 << 63`. +The conceptual full width-64 domain has size `2^64`, but that cardinality is +never materialized in a `u64`. + +This finite, 64-bit seeded family is **noncryptographic and only a practical +pseudorandom approximation**, not independently sampled ideal permutations, +not a proven secure PRP, and not exactly uniform over all `n!` permutations. +Domain separation and avalanche do not prove independence. Seed collisions +give identical permutations. Even conventional XOR-Feistel families have +structural restrictions (for example, even permutation parity when both +halves have at least two bits). The mixing schedule is fixed rather than +weakened to make a benchmark win. Do not use this as encryption or with +adversarial keys requiring a cryptographic guarantee. + +The existing `layer_apply` is deliberately unchanged: it only supports even +widths through 30, has no inverse, and has its own key/round schedule. Extending +or replacing it would change existing mappings. The new private primitive +therefore lives with the new evaluator. + +## Dyadic lift and cycle deletion + +The following is a derived construction and evaluator, not an implementation +of a published constant-time replica-selection algorithm. + +Let `h = 2^(b-1)`, let `A = [0,h)` be the old labels, and let `Q = P(seed,b)` +be an ordinary permutation of `[0,2h)`. Let `R` be its cycle projection onto +`A`: follow `Q` until the next label in `A`. Starting with `F_1(0) = 0`, define + +```text +F_(2h) = extend(F_h composed with inverse(R), fixing upper labels) composed with Q +``` + +Every `Q`-chain starting at old label `a` ends at old label `R(a)`. +The left composition changes only that chain's final destination, from +`R(a)` to `F_h(a)`. Upper edges and upper-only cycles are unchanged. Thus +cycle-projecting `F_(2h)` onto `A` gives `F_h`. For `h < n < 2h`, define `F_n` +by deleting all labels at least `n` from `F_(2h)`'s cycles. Cycle projections +compose, including across power-of-two boundaries. + +**Uniformity in the ideal model.** Conditional on any fixed `Q`, independent +uniform `F_h` makes `F_h composed with inverse(R)` uniform on the old-label +permutation group. This factor is consequently independent of `Q`; composing +it with uniform `Q` makes `F_(2h)` uniform. Cycle projection preserves +uniformity: each permutation of `n-1` labels has exactly `n` extensions +(insert the new label after any old label, or as a singleton cycle). +Induction establishes uniformity at every size. This proof uses ideal +uniformity, not the diagnostics of the implemented finite-key family. + +**Consistency.** Deleting label `n` from `F_(n+1)` only redirects its +predecessor to its successor, or removes its singleton cycle. For each +surviving input `r < n`, `F_(n+1)(r)` is therefore either `F_n(r)` or `n`. +At most one old input changes. A fixed prefix of input slots inherits this +property, while bijectivity supplies distinct outputs and prefix stability. +Enumerate *distinct inputs* `0,1,...,k-1`; following one output cycle instead +would not enumerate a full permutation. + +## Evaluator + +```text +replica_at(seed, n, r): + require 0 <= r < n + b = bit_length(n - 1) + x = r + while b > 0: + half = 1 << (b - 1) + y = P(seed, b, x) + if y >= n: + x = y + continue + if y >= half: + return y + while x >= half: + x = P_inverse(seed, b, x) + n = half + b -= 1 + return 0 +``` + +Walking inactive upper labels can use `Q` rather than the recursively defined +`F`, because edges whose outputs are upper labels were unchanged by the lift. +On reaching a lower output `y`, walking backward from its predecessor `x` +finds the old input `a = inverse(R)(y)`. The next level evaluates `F_h(a)`. +This proves the evaluator agrees with the explicit lift. + +Walks terminate because they follow a finite bijection's cycle. A forward walk +starts in the retained set, so it cannot remain forever in an inactive-only +cycle. A backward walk is entered only after reaching a lower label, so that +cycle necessarily meets the lower half. There are no retry caps or mapping- +changing fallbacks. + +## Expected work, not a worst-case guarantee + +Count each forward or inverse permutation evaluation as one constant-cost +operation. The following bounds are derived for **ideal independent uniform** +permutations and a fixed input, not asserted as a published theorem or a +guarantee for every seed of the implemented mixer. + +At a full dyadic level, descent probability is exactly one half. For an +upper input conditional on descent, the mean inverse-walk length is +`2h/(h+1)`: predecessors are sampled without replacement from `2h-1` labels, +`h` of them lower. The terminal lower input is uniform. If `T_s(a)` is mean +cost at full size `s`, and `U_s` its average over inputs, then + +```text +T_(2h)(a) = 1 + T_h(a)/2 if a < h +T_(2h)(a) = 1 + h/(h+1) + U_h/2 if a >= h +U_(2h) = 1 + h/(2(h+1)) + U_h/2 +T_1 = U_1 = 0 +``` + +Induction gives `U_s < 3` and `T_s(a) < 3.5`. The initial partial level is +different: its descent probability `p = h/n` can approach one. Its forward +length has mean `ell = (2h+1)/(n+1) < 2`; the retained endpoint is uniform +and independent of that length. On descent, the inverse walk retraces those +`ell-1` inactive edges. Starting from an upper input also requires finding a +lower predecessor; conditional on forward length `L`, this takes +`(2h-L+1)/(h+1)` further inverse calls on average. Thus + +```text +E C_n(r) = ell + p * (ell - 1 + T_h(r)) if r < h +E C_n(r) = ell + p * (ell - 1 + (2h+1-ell)/(h+1) + U_h) if r >= h +``` + +In particular, mean total work is **less than eight primitive calls per +slot**, uniformly in `n,r` in the ideal model. The upper-input bound approaches +eight at `n=h+1, r=h` as `h` grows. All later levels are full, so descent is +geometric after the initial level. The entering lower input depends on upper +permutations, but is independent of lower-width permutations; no independence +between walks, or between replica costs, is assumed. Linearity of expectation +gives expected `O(k)` enumeration. + +Long cycles still permit linear-in-domain walks for an unlucky permutation. +This is **not worst-case `O(k)`**, nor an adversarial-latency guarantee. State +is `O(1)` machine words, excluding optional returned output, with a +constant-sized width parameter block and no recursive stack, ring, permutation +array, or duplicate set. See [measurements and diagnostics](virtual-permutation-performance.md) +for observed forward/inverse counts and tails on the actual mixer. + +## References and verification + +The mathematical antecedent is a *virtual permutation*, using **cycle** +projection: + +- Neretin, [Virtual permutations and polymorphisms, section 1.2](https://arxiv.org/html/2202.12978v1#S1): + cycle deletion and its equivariance. +- Bourgade, Najnudel and Nikeghbali, + [A unitary extension of virtual permutations, section 1](https://arxiv.org/html/1102.2633v1#S1): + cycle projections, the Chinese restaurant construction, and the uniform + family as the Ewens parameter-one case. + +These references do **not** supply this evaluator, Feistel schedule, or its +performance bound. + +Tests exercise all word widths and extreme values, exhaustive small PRP +round-trips, full small-domain permutations, `k` prefixes, append/delete +consistency including dyadic boundaries and `u64::MAX`, and an independent +table-based lift/projection oracle. Exhaustive ideal permutations check equal +lift multiplicities at size four and all extensions of one fixed size-four +permutation through sizes five to eight. These are regression checks, not +substitutes for the ideal-model proof or evidence of cryptographic security. diff --git a/crates/consistent-choose-k/examples/permutation_diagnostics.rs b/crates/consistent-choose-k/examples/permutation_diagnostics.rs new file mode 100644 index 0000000..20ec0cf --- /dev/null +++ b/crates/consistent-choose-k/examples/permutation_diagnostics.rs @@ -0,0 +1,99 @@ +//! Deterministic distribution diagnostics, not probabilistic CI gates. +//! cargo run --release -p consistent-choose-k --example permutation_diagnostics + +use std::hash::{DefaultHasher, Hash, Hasher}; + +use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; + +const SAMPLES: u64 = 200_000; +fn key_seed(key: u64, workload_seed: u64) -> u64 { + let mut hasher = DefaultHasher::new(); + (workload_seed ^ key).hash(&mut hasher); + hasher.finish() +} + +fn report(algorithm: &str, n: usize, metric: &str, cells: &[u64]) { + let expected = cells.iter().sum::() as f64 / cells.len() as f64; + let chi2: f64 = cells + .iter() + .map(|&v| (v as f64 - expected).powi(2) / expected) + .sum(); + println!( + "{algorithm},{n},{metric},{},{expected:.3},{chi2:.3},{},{}", + cells.len() - 1, + cells.iter().min().expect("nonempty histogram"), + cells.iter().max().expect("nonempty histogram"), + ); +} + +fn main() { + let mut args = std::env::args().skip(1); + let workload_seed = match args.next().as_deref() { + None => 0x7065_726d_7574_6531, + Some("--held-out") => 0x686f_6c64_6f75_7431, + Some(_) => panic!("usage: permutation_diagnostics [--held-out]"), + }; + assert!( + args.next().is_none(), + "usage: permutation_diagnostics [--held-out]" + ); + println!("# samples={SAMPLES}, key_seed={workload_seed:#x}"); + println!( + "# iterator_bytes: layered={}, virtual={}; layered also owns heap counters", + std::mem::size_of::(), + std::mem::size_of::(), + ); + println!("algorithm,n,metric,df,expected_per_cell,chi2,min,max"); + for n in [3, 4, 5, 7, 8, 9, 16, 17, 32, 64] { + for algorithm in ["layered", "virtual"] { + let slots = n.min(8); + let mut marginal = vec![vec![0u64; n]; slots]; + let mut pairs = vec![0; n * n]; + let mut distant_pairs = vec![0; n * n]; + let mut triples = vec![0; n * (n - 1) * (n - 2) / 6]; + let mut consecutive_keys = vec![0; n * n]; + let mut previous_primary = None; + for key in 0..SAMPLES { + let seed = key_seed(key, workload_seed); + let values: Vec = if algorithm == "layered" { + ConsistentPermutation::new(n as u32, seed) + .take(slots) + .map(|v| v as usize) + .collect() + } else { + VirtualPermutation::new(n as u64, seed) + .take(slots) + .map(|v| v as usize) + .collect() + }; + for (histogram, &v) in marginal.iter_mut().zip(&values) { + histogram[v] += 1; + } + pairs[values[0] * n + values[1]] += 1; + distant_pairs[values[0] * n + values[slots - 1]] += 1; + let mut triple = [values[0], values[1], values[2]]; + triple.sort_unstable(); + let [a, b, c] = triple; + triples[c * (c - 1) * (c - 2) / 6 + b * (b - 1) / 2 + a] += 1; + if let Some(previous) = previous_primary { + consecutive_keys[previous * n + values[0]] += 1; + } + previous_primary = Some(values[0]); + } + for (slot, cells) in marginal.iter().enumerate() { + report(algorithm, n, &format!("slot_{slot}"), cells); + } + for (metric, cells) in [("ordered_0_1", pairs), ("ordered_0_last", distant_pairs)] { + assert!((0..n).all(|v| cells[v * n + v] == 0)); + let off_diagonal: Vec<_> = cells + .into_iter() + .enumerate() + .filter_map(|(i, count)| (i / n != i % n).then_some(count)) + .collect(); + report(algorithm, n, metric, &off_diagonal); + } + report(algorithm, n, "choose_3", &triples); + report(algorithm, n, "consecutive_key_primaries", &consecutive_keys); + } + } +} diff --git a/crates/consistent-choose-k/src/lib.rs b/crates/consistent-choose-k/src/lib.rs index 25b7e0f..4278f83 100644 --- a/crates/consistent-choose-k/src/lib.rs +++ b/crates/consistent-choose-k/src/lib.rs @@ -3,6 +3,7 @@ mod consistent_hash; mod consistent_permutation; mod consistent_reservoir; mod node_map; +mod virtual_permutation; pub use choose_k::ConsistentChooseKHasher; pub use consistent_hash::{ ConsistentHashIterator, ConsistentHashRevIterator, ConsistentHasher, HashSeqBuilder, @@ -11,3 +12,4 @@ pub use consistent_hash::{ pub use consistent_permutation::ConsistentPermutation; pub use consistent_reservoir::ConsistentReservoir; pub use node_map::ConsistentNodeMap; +pub use virtual_permutation::VirtualPermutation; diff --git a/crates/consistent-choose-k/src/virtual_permutation.rs b/crates/consistent-choose-k/src/virtual_permutation.rs new file mode 100644 index 0000000..76be660 --- /dev/null +++ b/crates/consistent-choose-k/src/virtual_permutation.rs @@ -0,0 +1,563 @@ +//! Cycle-consistent permutations, as opposed to the survivor-list consistency +//! of `ConsistentPermutation`. See `docs/virtual-permutation.md`. + +use std::iter::FusedIterator; + +const WEYL: u64 = 0x9E37_79B9_7F4A_7C15; + +#[inline] +fn mix(mut x: u64) -> u64 { + x = (x ^ (x >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + x = (x ^ (x >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + x ^ (x >> 31) +} + +trait Permutation { + fn forward(&self, x: u64) -> u64; + fn inverse(&self, x: u64) -> u64; +} + +/// Alternating Feistel half updates, including unequal halves for odd widths. +/// This is a noncryptographic family, not a uniform or secure PRP. +struct WordPermutation { + key: u64, + bits: u32, + right_bits: u32, + left_mask: u64, + right_mask: u64, +} + +impl WordPermutation { + #[inline] + fn new(seed: u64, bits: u32) -> Self { + debug_assert!((1..=64).contains(&bits)); + let left_bits = bits / 2; + let right_bits = bits - left_bits; + Self { + key: mix(seed ^ WEYL.wrapping_mul(u64::from(bits))), + bits, + right_bits, + left_mask: (1u64 << left_bits) - 1, + right_mask: (1u64 << right_bits) - 1, + } + } + + #[inline] + fn round(&self, x: u64, round: u64) -> u64 { + mix(x ^ self.key.wrapping_add(WEYL.wrapping_mul(round + 1))) + } + + #[inline] + fn transform_rounds(&self, x: u64) -> u64 { + let mut left = x >> self.right_bits; + let mut right = x & self.right_mask; + for i in 0..PAIRS { + let pair = if INVERSE { PAIRS - 1 - i } else { i }; + if INVERSE { + right ^= self.round(left, 2 * pair + 1) & self.right_mask; + left ^= self.round(right, 2 * pair) & self.left_mask; + } else { + left ^= self.round(right, 2 * pair) & self.left_mask; + right ^= self.round(left, 2 * pair + 1) & self.right_mask; + } + } + (left << self.right_bits) | right + } + + #[inline] + fn transform(&self, x: u64) -> u64 { + // Tiny half-domains need more mixing: eight rounds had strong + // ordered-pair bias at width three despite clean marginals. + match self.bits { + 1 => x ^ (self.key & 1), + 2..=4 => self.transform_rounds::<12, INVERSE>(x), + 5..=7 => self.transform_rounds::<8, INVERSE>(x), + _ => self.transform_rounds::<4, INVERSE>(x), + } + } +} + +impl Permutation for WordPermutation { + #[inline] + fn forward(&self, x: u64) -> u64 { + self.transform::(x) + } + + #[inline] + fn inverse(&self, x: u64) -> u64 { + self.transform::(x) + } +} + +#[inline] +fn evaluate(mut n: u64, mut x: u64, mut layer: impl FnMut(u32) -> P) -> u64 { + let mut bits = u64::BITS - (n - 1).leading_zeros(); + while bits > 0 { + let half = 1u64 << (bits - 1); + let permutation = layer(bits); + loop { + let y = permutation.forward(x); + if y >= n { + x = y; + continue; + } + if y >= half { + return y; + } + // Find the old input at the start of this Q-chain, not its old + // output y: this applies inverse(cycle_projection(Q)) before + // evaluating the smaller consistent permutation. + while x >= half { + x = permutation.inverse(x); + } + break; + } + n = half; + bits -= 1; + } + 0 +} + +/// An allocation-free, per-key permutation of `0..n` with stable replica slots. +/// +/// For fixed `(n, seed)`, taking `k` items gives `k` distinct nodes and is a +/// prefix of every longer selection. Appending node `n` changes at most one +/// existing slot, and that slot changes to `n`; removing the last node changes +/// only its slot (among slots that still exist). Membership must be a prefix +/// of consecutive IDs. +/// +/// **Not a drop-in replacement for [`crate::ConsistentPermutation`]:** this +/// preserves cycle projections, not survivor list order. Removing a node from +/// the output list does not generally produce the smaller permutation. +/// +/// Uniform, independent permutations per key and bit width give uniform +/// selections and expected `O(k)` evaluation with constant-cost forward/inverse +/// primitives. The implemented 64-bit seeded, 8/16/24-round Feistel family is +/// only a practical noncryptographic approximation to that ideal. Neither +/// exact uniformity, cryptographic security, nor worst-case `O(k)` is claimed. +/// State is constant size; no rings, permutation tables, or duplicate sets +/// are constructed. Walks have no artificial retry limit. +/// +/// ``` +/// use consistent_choose_k::VirtualPermutation; +/// +/// let seed = 0x1234_5678_9abc_def0; // normally a well-mixed hash of the key +/// let selection = VirtualPermutation::new(100, seed); +/// let third = selection.replica_at(2); +/// assert_eq!(selection.clone().nth(2), Some(third)); +/// let replicas: Vec<_> = selection.take(3).collect(); +/// assert_eq!(replicas.len(), 3); +/// ``` +#[derive(Clone, Debug)] +pub struct VirtualPermutation { + seed: u64, + n: u64, + next: u64, +} + +impl VirtualPermutation { + /// Construct a permutation for `1..=u64::MAX` nodes. + /// + /// As with [`crate::ConsistentPermutation::new`], supply a well-mixed + /// 64-bit hash of the key. Width/round domain separation is internal and + /// independent of `n` and of the requested replica count. + /// + /// # Panics + /// + /// Panics if `n == 0`. + pub fn new(n: u64, seed: u64) -> Self { + assert!(n > 0, "n must be at least 1"); + Self { seed, n, next: 0 } + } + + /// Universe size, independent of the iterator's current position. + pub fn n(&self) -> u64 { + self.n + } + + /// Evaluate an absolute zero-based replica slot without advancing the + /// iterator. Expected constant work under ideal independent permutations; + /// no worst-case constant-time guarantee. + /// + /// # Panics + /// + /// Panics if `slot >= self.n()`. + pub fn replica_at(&self, slot: u64) -> u64 { + assert!(slot < self.n, "replica slot must be less than n"); + evaluate(self.n, slot, |bits| WordPermutation::new(self.seed, bits)) + } +} + +impl Iterator for VirtualPermutation { + type Item = u64; + + fn next(&mut self) -> Option { + if self.next == self.n { + return None; + } + let value = self.replica_at(self.next); + self.next += 1; + Some(value) + } + + fn size_hint(&self) -> (usize, Option) { + match usize::try_from(self.n - self.next) { + Ok(remaining) => (remaining, Some(remaining)), + Err(_) => (usize::MAX, None), + } + } + + fn nth(&mut self, n: usize) -> Option { + self.next = self.next.saturating_add(n as u64).min(self.n); + self.next() + } +} + +impl FusedIterator for VirtualPermutation {} + +#[cfg(test)] +mod tests { + use super::*; + + fn seed(i: u64) -> u64 { + mix(i.wrapping_add(WEYL)) + } + + #[test] + fn word_round_trips_all_widths() { + for bits in 1..=64 { + let mask = u64::MAX >> (64 - bits); + for key in [0, 1, u64::MAX, seed(42)] { + let p = WordPermutation::new(key, bits); + for x in [0, 1, mask / 2, mask / 2 + 1, mask] + .into_iter() + .chain((0..128).map(|i| seed(i) & mask)) + { + let y = p.forward(x); + assert_eq!(y & !mask, 0); + assert_eq!(p.inverse(y), x, "bits={bits} key={key} x={x}"); + assert_eq!(p.forward(p.inverse(x)), x); + } + } + } + } + + #[test] + fn word_exhaustive_small_domains() { + for bits in 1..=10 { + for key in 0..16 { + let p = WordPermutation::new(seed(key), bits); + let mut outputs: Vec<_> = (0..1 << bits) + .map(|x| { + let y = p.forward(x); + assert_eq!(p.inverse(y), x); + y + }) + .collect(); + outputs.sort_unstable(); + assert_eq!(outputs, (0..1 << bits).collect::>()); + } + } + } + + #[test] + fn full_permutations_prefixes_and_membership() { + for key in 0..32 { + let key = seed(key); + let mut previous = vec![]; + for n in 1..=257 { + let permutation = VirtualPermutation::new(n, key); + let values: Vec<_> = permutation.clone().collect(); + let mut sorted = values.clone(); + sorted.sort_unstable(); + assert_eq!(sorted, (0..n).collect::>()); + for k in [0, 1, 2, 3, 8, n / 2, n] { + if k <= n { + assert_eq!( + permutation.clone().take(k as usize).collect::>(), + values[..k as usize] + ); + } + } + let mut changed = 0; + for (r, old) in previous.iter().enumerate() { + assert_eq!(permutation.replica_at(r as u64), values[r]); + if *old != values[r] { + assert_eq!(values[r], n - 1); + changed += 1; + } + // Cycle-deleting the new node recovers every old slot. + let projected = if values[r] == n - 1 { + values[n as usize - 1] + } else { + values[r] + }; + assert_eq!(*old, projected); + } + assert!(changed <= 1); + previous = values; + } + } + } + + #[test] + fn large_domains_and_power_boundaries() { + for bits in 1..64 { + let half = 1u64 << bits; + for n in [half - 1, half, half + 1, u64::MAX - 1] { + for key in [0, 1, u64::MAX, seed(123)] { + let p = VirtualPermutation::new(n, key); + let larger = VirtualPermutation::new(n + 1, key); + for r in [0, n / 2, n - 1] { + let old = p.replica_at(r); + let new = larger.replica_at(r); + assert!(old < n && new <= n); + assert!(new == old || new == n); + assert_eq!(old, if new == n { larger.replica_at(n) } else { new }); + } + } + } + } + let mut p = VirtualPermutation::new(u64::MAX, seed(9)); + p.next = u64::MAX - 1; + assert!(p.next().is_some()); + assert_eq!(p.next(), None); + assert_eq!(p.next(), None); + } + + #[test] + fn iterator_boundaries() { + let mut p = VirtualPermutation::new(1, 0); + assert_eq!(p.n(), 1); + assert_eq!(p.size_hint(), (1, Some(1))); + assert_eq!(p.clone().take(0).count(), 0); + assert_eq!(p.next(), Some(0)); + assert_eq!(p.size_hint(), (0, Some(0))); + assert_eq!(p.next(), None); + assert_eq!(p.replica_at(0), 0); + assert_eq!(p.nth(usize::MAX), None); + let mut p = VirtualPermutation::new(100, seed(4)); + assert_eq!(p.nth(12), Some(p.replica_at(12))); + assert_eq!(p.next(), Some(p.replica_at(13))); + assert_eq!(p.nth(usize::MAX), None); + assert_eq!(p.next(), None); + } + + #[test] + #[should_panic(expected = "n must be at least 1")] + fn invalid_empty_domain() { + VirtualPermutation::new(0, 0); + } + + #[test] + #[should_panic(expected = "replica slot must be less than n")] + fn invalid_slot() { + VirtualPermutation::new(10, 0).replica_at(10); + } + + #[test] + fn long_cycles_are_not_truncated() { + use std::cell::Cell; + + struct Rotation<'a> { + mask: u64, + calls: &'a Cell, + } + impl Permutation for Rotation<'_> { + fn forward(&self, x: u64) -> u64 { + self.calls.set(self.calls.get() + 1); + x.wrapping_add(1) & self.mask + } + fn inverse(&self, x: u64) -> u64 { + self.calls.set(self.calls.get() + 1); + x.wrapping_sub(1) & self.mask + } + } + // Every dyadic lift of these ascending cycles is the same ascending + // cycle. A last-slot query just above a half boundary takes long walks. + let calls = Cell::new(0); + let result = evaluate(2049, 2048, |bits| Rotation { + mask: (1 << bits) - 1, + calls: &calls, + }); + assert_eq!(result, 0); + assert!(calls.get() > 4096); + } + + // Deliberately explicit tables only in the oracle, never in the evaluator. + fn project(p: &[u64], n: usize) -> Vec { + (0..n) + .map(|x| { + let mut y = p[x]; + while y >= n as u64 { + y = p[y as usize]; + } + y + }) + .collect() + } + + fn lift(lower: &[u64], q: &[u64]) -> Vec { + let h = lower.len(); + let r = project(q, h); + let mut inverse = vec![0; h]; + for (x, &y) in r.iter().enumerate() { + inverse[y as usize] = x; + } + q.iter() + .map(|&y| { + if y < h as u64 { + lower[inverse[y as usize]] + } else { + y + } + }) + .collect() + } + + #[test] + fn matches_explicit_lift_and_cycle_projection() { + for key in 0..32 { + let key = seed(key); + let mut full = vec![0]; + for bits in 1..=8 { + let p = WordPermutation::new(key, bits); + let q: Vec<_> = (0..1 << bits).map(|x| p.forward(x)).collect(); + full = lift(&full, &q); + for n in (full.len() / 2 + 1)..=full.len() { + let actual: Vec<_> = VirtualPermutation::new(n as u64, key).collect(); + assert_eq!(actual, project(&full, n), "bits={bits} n={n}"); + } + } + } + } + + fn permutations(n: usize) -> Vec> { + fn visit(values: &mut [u64], start: usize, out: &mut Vec>) { + if start == values.len() { + out.push(values.to_vec()); + } else { + for i in start..values.len() { + values.swap(start, i); + visit(values, start + 1, out); + values.swap(start, i); + } + } + } + let mut out = vec![]; + visit(&mut (0..n as u64).collect::>(), 0, &mut out); + out + } + + #[test] + fn ideal_uniform_lift_fibers() { + use std::collections::BTreeMap; + + // Every ideal Q_1,Q_2 combination: F_4 has equal multiplicities. + let mut counts = BTreeMap::new(); + for lower in permutations(2) { + for q in permutations(4) { + *counts.entry(lift(&lower, &q)).or_insert(0) += 1; + } + } + assert_eq!(counts.len(), 24); + assert!(counts.values().all(|&count| count == 2)); + + // Fix F_4; every extension through each intermediate n occurs 4! times + // at n=8, and equally often for n=5,6,7 after cycle projection. + let lower = vec![2, 0, 3, 1]; + let mut counts: Vec, usize>> = (5..=8).map(|_| BTreeMap::new()).collect(); + for q in permutations(8) { + let full = lift(&lower, &q); + assert_eq!(project(&full, 4), lower); + for (i, n) in (5..=8).enumerate() { + *counts[i].entry(project(&full, n)).or_insert(0) += 1; + } + } + for (i, count) in counts.iter().enumerate() { + let expected_distinct: usize = (5..=i + 5).product(); + assert_eq!(count.len(), expected_distinct); + assert!(count.values().all(|&v| v == 40320 / expected_distinct)); + } + } + + #[test] + #[ignore = "deterministic operation-count report; not a timing or statistical CI gate"] + fn operation_count_diagnostics() { + use std::{cell::Cell, rc::Rc}; + + struct Counted { + permutation: WordPermutation, + calls: Rc>, + } + impl Permutation for Counted { + fn forward(&self, x: u64) -> u64 { + let mut calls = self.calls.get(); + calls[0] += 1; + self.calls.set(calls); + self.permutation.forward(x) + } + fn inverse(&self, x: u64) -> u64 { + let mut calls = self.calls.get(); + calls[1] += 1; + self.calls.set(calls); + self.permutation.inverse(x) + } + } + + println!("n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls"); + for n in [ + 1, + 7, + 8, + 9, + 255, + 256, + 257, + 1023, + 1024, + 1025, + 65535, + 65536, + 65537, + (1 << 30) - 1, + 1 << 30, + (1 << 63) + 1, + u64::MAX, + ] { + for slot in [0, n / 2, n - 1] { + let calls = Rc::new(Cell::new([0; 3])); + let mut totals = [0u64; 3]; + let mut samples = vec![]; + for key in 0..10_000 { + calls.set([0; 3]); + let result = evaluate(n, slot, |bits| { + let mut counts = calls.get(); + counts[2] += 1; + calls.set(counts); + Counted { + permutation: WordPermutation::new(seed(key), bits), + calls: Rc::clone(&calls), + } + }); + assert!(result < n); + let counts = calls.get(); + for i in 0..3 { + totals[i] += counts[i]; + } + samples.push(counts[0] + counts[1]); + } + samples.sort_unstable(); + println!( + "{n},{slot},{:.4},{:.4},{:.4},{},{},{}", + totals[0] as f64 / 10_000.0, + totals[1] as f64 / 10_000.0, + totals[2] as f64 / 10_000.0, + samples[4999], + samples[9899], + samples[9999], + ); + } + } + } +} From 316ecdadb614cc6fc1ade91702b6ee58280ee150 Mon Sep 17 00:00:00 2001 From: Alexander Neubeck Date: Thu, 24 Sep 2026 09:35:48 -0700 Subject: [PATCH 2/3] Compare cycle projection with the existing balanced Feistel Add an explicitly experimental two-bit variant using the exact existing permutation and its inverse. Preserve existing mappings and the stronger variant, and rerun three-way timing, held-out statistics, and traversal diagnostics. Document both recovered throughput and repeatable small-domain bias. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- crates/consistent-choose-k/README.md | 11 +- .../benchmarks/replica_comparison.rs | 50 ++- .../docs/virtual-permutation-performance.md | 166 ++++++++- .../docs/virtual-permutation.md | 45 +++ .../examples/permutation_diagnostics.rs | 14 +- .../src/consistent_permutation.rs | 69 +++- crates/consistent-choose-k/src/lib.rs | 2 +- .../src/virtual_permutation.rs | 328 +++++++++++++++--- 8 files changed, 629 insertions(+), 56 deletions(-) diff --git a/crates/consistent-choose-k/README.md b/crates/consistent-choose-k/README.md index 1049a38..3cef1e5 100644 --- a/crates/consistent-choose-k/README.md +++ b/crates/consistent-choose-k/README.md @@ -37,7 +37,7 @@ Why replication matters - Distributes read/write load across multiple owners, reducing hotspots. - Enables fast recovery and higher tail-latency resilience. -## Two permutation APIs, different membership semantics +## Permutation APIs and membership semantics The existing `ConsistentPermutation` preserves **survivor list order** when nodes are appended or removed from the end of `0..n`. The additional, @@ -58,6 +58,15 @@ See the [algorithm, proof assumptions and API guide](docs/virtual-permutation.md and the [reproducible comparison with the existing algorithm](docs/virtual-permutation-performance.md). The existing APIs and their mappings remain unchanged. +`BalancedVirtualPermutation` is a matched-network experiment: it gives the +same cycle/slot semantics as `VirtualPermutation`, but uses **exactly** the +existing `ConsistentPermutation` Feistel, with even widths and two bits per +lift, over `1..=2^30`. Its state is also allocation-free. The +[three-way comparison](docs/virtual-permutation-performance.md#matched-network-follow-up) +repeats performance and primary/held-out statistical diagnostics. This variant +has repeatable small-domain distribution bias and is not the default or a +statistically equivalent replacement for the stronger mixer. + ## Applications beyond replication The `ConsistentChooseK` iterator produces a per-key ranking of all `n` nodes in priority order — consistently and with zero memory overhead. This ranking is a strict superset of simple replication and enables drop-in replacements for several well-known algorithms that traditionally require maintaining expensive data structures such as hash rings. diff --git a/crates/consistent-choose-k/benchmarks/replica_comparison.rs b/crates/consistent-choose-k/benchmarks/replica_comparison.rs index ac9b5af..60ac48e 100644 --- a/crates/consistent-choose-k/benchmarks/replica_comparison.rs +++ b/crates/consistent-choose-k/benchmarks/replica_comparison.rs @@ -7,7 +7,7 @@ use std::{ time::Duration, }; -use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use consistent_choose_k::{BalancedVirtualPermutation, ConsistentPermutation, VirtualPermutation}; use criterion::{criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, Throughput}; use rand::{rngs::StdRng, RngExt, SeedableRng}; @@ -101,6 +101,17 @@ fn end_to_end(c: &mut Criterion) { } }) }); + group.bench_function(BenchmarkId::new("balanced", format!("n{n}_k{k}")), |b| { + b.iter(|| { + for &key in black_box(input) { + let seed = if mode == "fresh" { hash_key(key) } else { key }; + consume( + BalancedVirtualPermutation::new(black_box(n), seed), + black_box(k), + ); + } + }) + }); } } group.finish(); @@ -134,6 +145,13 @@ fn cost_components(c: &mut Criterion) { } }) }); + setup.bench_function(BenchmarkId::new("balanced", n), |b| { + b.iter(|| { + for &seed in black_box(&seeds) { + black_box(BalancedVirtualPermutation::new(black_box(n), seed)); + } + }) + }); } setup.finish(); @@ -204,6 +222,36 @@ fn cost_components(c: &mut Criterion) { }), } }); + group.bench_function(BenchmarkId::new("balanced", format!("n{n}_k{k}")), |b| { + match mode { + "stream_only" => b.iter_batched_ref( + || { + seeds + .iter() + .map(|&seed| BalancedVirtualPermutation::new(n, seed)) + .collect::>() + }, + |iterators| { + for iter in black_box(iterators) { + consume(iter, black_box(k)); + } + }, + BatchSize::SmallInput, + ), + _ => b.iter(|| { + for &seed in black_box(&seeds) { + let iter = BalancedVirtualPermutation::new(black_box(n), seed); + if mode == "collect" { + let mut out = Vec::with_capacity(black_box(k)); + out.extend(iter.take(k)); + black_box(out); + } else { + black_box(iter.replica_at(black_box(k as u64 - 1))); + } + } + }), + } + }); } } group.finish(); diff --git a/crates/consistent-choose-k/docs/virtual-permutation-performance.md b/crates/consistent-choose-k/docs/virtual-permutation-performance.md index 11ebc35..e16731d 100644 --- a/crates/consistent-choose-k/docs/virtual-permutation-performance.md +++ b/crates/consistent-choose-k/docs/virtual-permutation-performance.md @@ -1,9 +1,16 @@ # Replica-selection comparison -The new `VirtualPermutation` trades slower sequential enumeration for +The original `VirtualPermutation` trades slower sequential enumeration for allocation-free state and direct replica-slot evaluation. It is **not a faster drop-in replacement** for the existing `ConsistentPermutation`. +The [matched-network follow-up](#matched-network-follow-up) adds +`BalancedVirtualPermutation`, which uses exactly the existing Feistel on +two-bit layers. All three methods are remeasured together; repeatable +small-domain bias in the matched variant is reported rather than hidden. +The original two-way tables below are historical measurements at commit +`d51abfb`, not fresh measurements of the follow-up. + The existing algorithm preserves survivor **list order**; the new one preserves each old **slot** except a slot replaced by an appended node (or previously naming a removed node). Both provide distinct, `k`-prefix-stable selections, @@ -11,6 +18,10 @@ but these membership policies are not interchangeable. The [design note](virtual-permutation.md) explains the construction, ideal-model proof, noncryptographic approximation and expected-versus-worst-case distinction. +Runner names are `layered` for the existing streaming iterator, `virtual` +for the stronger one-bit cycle construction, and `balanced` for the +matched-Feistel two-bit cycle construction. + ## Reproduce From the repository root: @@ -62,7 +73,7 @@ differences should not be interpreted as portable wins. The fixture is 128 `u64` keys generated by `StdRng` with seed `0x7065726d75746531`. The fresh-key mode hashes each key with `DefaultHasher` -inside every query, identically for both algorithms. Width/round preparation +inside every query, identically for all methods. Width/round preparation and all inverse walks in the new evaluator are included, as is the existing iterator's counter allocation. These are fresh **queries** of a repeated deterministic corpus, not new unpredictable keys on every benchmark iteration. @@ -94,7 +105,7 @@ pairs in each of the fresh and seeded groups. Component groups use `n = 17,257,1000,65537,2^30`. All paired timings stay within the old implementation's supported domain. No wider-domain speedup is inferred. -## Fresh-query results +## Original fresh-query results Representative mean **ns/query**, with bootstrap 95% confidence intervals in brackets. Ratio is new/existing time: **above one means a regression**. @@ -139,7 +150,7 @@ an 8/16/24-round, fully mixed forward/inverse primitive. Avoiding a small state allocation does not make up for that extra arithmetic and traversal. This compares the actual implementations, not algorithms normalized to an identical primitive: both traversal strategy and mixing cost contribute. -All reported timings use the final strengthened schedule, not the rejected +All timings in this original comparison use the final strengthened schedule, not the rejected eight-round version discussed below. ## Separating setup, streaming and direct-slot costs @@ -168,8 +179,9 @@ not be added or subtracted as exact identities: compiler optimization, instruction overlap and cache state differ between the groups. For `n=1000,k=3`, the seeded-query sample standard deviations were 0.54 ns (existing) and 2.95 ns (new); collecting 1,000 replicas had standard deviations -315.22 ns and 4,646.27 ns respectively. All 721 benchmark estimates, including -their standard deviations, can be regenerated with the CSV script. +315.22 ns and 4,646.27 ns respectively. The original run had 721 benchmark +estimates. The current three-method runner emits 1,081 estimates, including +standard deviations, with the same CSV script. Direct access becomes useful for later slots: querying slot 999 at `n=1000` was about 132x faster than replaying the old iterator. Early slots need not @@ -278,3 +290,145 @@ test uses ascending cycles and needs over 4,096 calls for `n=2049,slot=2048`; the evaluator still returns the correct result without a retry cap. Long cycles and linear-in-domain worst cases remain possible. Neither the diagnostics nor the timing confidence intervals establish a latency bound. + +## Matched-network follow-up + +At the user's request, `BalancedVirtualPermutation` now uses exactly the +existing `layer_apply` network: same seed, mixer, balanced halves, round +counts, rotation and Weyl key schedule. The cycle construction descends by +two bits, as the existing iterator does. A new inverse shares the same round +function and reverses that exact key schedule. It does not introduce a +different mixer, extra seed hash, or independently tuned round count. + +The same machine, compiler, flags, key corpus, 131-case matrix and Criterion +settings were used again, sequentially in `layered`, `virtual`, `balanced` +order per case. There are 1,081 estimates across all groups. All three +implementations are rerun; comparisons below use this run, not subtraction of +timings from the earlier run. The shared-host and fixed-corpus limitations +still apply. State is 24 bytes with no allocation for either cycle iterator, +versus 40 bytes plus the heap counters for the existing iterator. + +### Fresh queries: matching the primitive removes much of the overhead + +Mean ns/query [bootstrap 95% interval]. The final ratio is +**matched/existing**; below one favors the matched variant. + +| n | k | Existing | Stronger one-bit | Matched two-bit | Matched/existing | +| ---: | ---: | ---: | ---: | ---: | ---: | +| 8 | 1 | 45.27 [44.67, 45.87] | 110.60 [110.18, 111.13] | 61.42 [60.35, 62.62] | 1.36x | +| 8 | 8 | 205.76 [203.68, 207.92] | 1,030.44 [1,021.06, 1,041.23] | 608.10 [592.46, 623.97] | 2.96x | +| 16 | 3 | 57.51 [57.16, 57.82] | 334.54 [333.57, 335.44] | 67.91 [66.89, 69.14] | 1.18x | +| 17 | 3 | 105.10 [104.22, 105.84] | 619.91 [618.09, 621.79] | 261.59 [258.54, 266.26] | 2.49x | +| 33 | 33 | 637.60 [604.96, 684.35] | 8,604.99 [8,085.82, 9,388.97] | 2,155.89 [2,115.56, 2,192.34] | 3.38x | +| 255 | 3 | 35.09 [34.77, 35.39] | 140.78 [140.32, 141.18] | 30.76 [30.58, 30.97] | 0.88x | +| 256 | 3 | 35.99 [35.64, 36.29] | 139.85 [139.28, 140.62] | 31.11 [30.80, 31.46] | 0.86x | +| 257 | 3 | 61.67 [61.43, 61.91] | 304.03 [302.58, 305.93] | 198.41 [195.88, 201.23] | 3.22x | +| 257 | 8 | 155.53 [151.73, 159.47] | 872.01 [846.66, 902.67] | 536.47 [521.46, 564.32] | 3.45x | +| 1,000 | 1 | 19.75 [19.70, 19.81] | 43.94 [43.47, 44.47] | 16.26 [16.03, 16.48] | 0.82x | +| 1,000 | 3 | 32.83 [32.63, 33.02] | 101.72 [101.21, 102.22] | 27.72 [27.58, 27.86] | 0.84x | +| 1,000 | 16 | 98.73 [98.11, 99.45] | 539.84 [504.74, 605.05] | 137.11 [135.35, 138.99] | 1.39x | +| 1,000 | 250 | 2,059.36 [2,047.90, 2,070.14] | 14,914.99 [14,779.03, 15,043.08] | 3,374.48 [3,336.90, 3,412.61] | 1.64x | +| 1,000 | 1,000 | 9,808.45 [9,658.26, 10,006.25] | 83,713.94 [83,369.99, 84,114.21] | 22,289.60 [22,158.85, 22,422.35] | 2.27x | +| 1,024 | 3 | 33.67 [33.45, 33.88] | 98.44 [97.52, 99.27] | 29.54 [28.99, 30.18] | 0.88x | +| 1,025 | 3 | 65.79 [65.44, 66.12] | 267.49 [266.66, 268.47] | 182.40 [180.91, 184.61] | 2.77x | +| 65,536 | 3 | 33.12 [32.94, 33.28] | 83.63 [83.24, 83.98] | 28.43 [28.16, 28.67] | 0.86x | +| 65,537 | 3 | 65.87 [65.58, 66.16] | 262.49 [262.17, 262.80] | 171.81 [169.97, 174.83] | 2.61x | +| 1,000,000 | 3 | 35.39 [34.87, 36.04] | 95.33 [94.27, 96.33] | 27.43 [27.22, 27.61] | 0.77x | +| 2^30 | 3 | 34.24 [33.94, 34.51] | 90.35 [89.09, 91.52] | 26.34 [26.18, 26.47] | 0.77x | + +At `n=1000,k=3`, matching the network and stride reduces the cycle +construction from 101.72 to 27.72 ns/query (3.67x faster), making it about +16% faster than the existing iterator on this workload. That is 9.24 +ns/replica versus 10.94 existing and 33.91 stronger-one-bit. For a full +1,000-node permutation it is still 2.27x slower than existing (22.29 versus +9.81 ns/replica), although much faster than the stronger variant's 83.71. + +The matched mean is below the existing mean in 31 of 131 cases, including the +degenerate `n=1` case; small differences are not blanket claims of significance. +Its largest observed mean ratio is 3.45x at `n=257,k=8`. Just above powers of +four, the top domain is nearly three-quarters inactive, and cycle walking plus +inverse traversal is still expensive. This experiment changes primitive +**and stride together**; it does not isolate the machine cost of odd halves +alone. It demonstrates that the earlier slowdown was not an unavoidable cost +of cycle consistency, but it does not show that the cycle method always wins. + +The component measurements help distinguish allocation from streaming work. +Below are mean ns/query from the same run, with prehashed seeds; slot queries +return **one** value, while the other `k` rows consume or collect `k` values. + +| Workload, n=1000 | Existing | Stronger one-bit | Matched two-bit | +| --- | ---: | ---: | ---: | +| Constructor + drop | 9.71 | 1.09 | 1.17 | +| Seeded query, k=3 | 31.25 | 103.66 | 27.08 | +| Stream only, k=3 | 14.45 | 89.51 | 20.95 | +| Stream only, k=1000 | 9,541.23 | 86,465.65 | 22,863.95 | +| Collect, k=3 | 42.90 | 112.43 | 36.66 | +| Collect, k=1000 | 9,727.46 | 87,758.08 | 22,936.49 | +| Query slot 2 | 30.03 | 31.54 | 7.89 | +| Query slot 999 | 9,604.66 | 77.43 | 12.69 | + +For `k=3` stream-only, the existing/matched 95% intervals are +[14.25, 14.66] / [20.80, 21.11] ns, and sample standard deviations +0.47 / 0.39 ns. For the seeded complete query, the intervals are +[30.98, 31.47] / [26.92, 27.29] ns. The fresh-query win is therefore consistent +with avoiding allocation, **not** evidence that the cycle traversal itself is +cheaper. These components still cannot be added as exact identities because +their compilation and cache conditions differ. + +The matched direct slot-999 query has interval [12.53, 12.88] ns and sample +standard deviation 0.42 ns. The existing iterator must replay to reach that +slot; this is an API/workload advantage, not a claim of comparable sequential +enumeration speed. + +### Statistical quality is not equivalent + +Both deterministic 200,000-key corpora were rerun without tuning the matched +network. The existing and stronger variants' reported diagnostic rows are +unchanged from the prior run. The matched variant has repeatable deviations: + +| Metric | Degrees of freedom | Matched primary chi-square | Matched held-out chi-square | +| --- | ---: | ---: | ---: | +| n=5, slot 4 | 4 | 82.80 | 84.94 | +| n=5, ordered slots (0,4) | 19 | 98.15 | 98.63 | +| n=8, primary marginal | 7 | 6.21 | 7.26 | +| n=8, ordered slots (0,1) | 55 | 112.66 | 131.31 | +| n=8, first-three subset | 55 | 133.99 | 146.63 | +| n=9, first-three subset | 83 | 148.11 | 141.19 | + +For the `n=5,slot=4` marginal, the most frequent node occurs 41,613 and 41,612 +times against 40,000 expected in each corpus, about a 4% excess. At `n=8`, +the stronger variant's first-three subset statistics remain 47.66 / 40.41, +and the existing iterator's are 53.47 / 59.83, versus the matched variant's +133.99 / 146.63 (all 55 degrees of freedom). + +The same primitive need not have the same statistical behavior after list +projection versus cycle projection. These results do not distinguish +within-width weaknesses from cross-width seed correlations, nor do they +prove exact uniformity for any method. They do show that replacing the +stronger network is **not statistically neutral**. The matched variant is +retained as an explicitly experimental comparison, not silently substituted +for `VirtualPermutation` or promoted as meeting the ideal randomness model. + +### Traversal counts explain remaining boundary costs + +The operation-count diagnostic now prints both cycle variants, using the same +10,000 seeds per `(n,slot)` as before. Means count forward plus inverse calls, +not Feistel rounds or nanoseconds: + +| n | slot | Stronger mean calls | Matched mean calls | Stronger p99 | Matched p99 | Matched observed max | +| ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| 256 | 0 | 1.9808 | 1.3247 | 7 | 4 | 4 | +| 257 | 0 | 4.9857 | 8.2873 | 16 | 34 | 59 | +| 257 | 256 | 7.8930 | 13.2029 | 22 | 41 | 134 | +| 65,536 | 0 | 1.9922 | 1.3330 | 7 | 4 | 8 | +| 65,537 | 0 | 4.9722 | 8.3462 | 16 | 34 | 68 | +| 65,537 | 65,536 | 8.0348 | 13.4067 | 23 | 41 | 73 | + +At full powers of four, primary lookup uses fewer levels; a regression test +also confirms that the matched and existing algorithms give the **same +primary value** there. Just above those boundaries, however, the two-bit +construction must traverse a much larger inactive upper region and then +often retrace it. Its cheaper primitive can still yield faster timings than +the stronger variant despite more primitive calls. These measured tails are +not worst-case guarantees, and the one-bit variant's ideal less-than-eight- +call bound does not apply to the two-bit variant. diff --git a/crates/consistent-choose-k/docs/virtual-permutation.md b/crates/consistent-choose-k/docs/virtual-permutation.md index 2d05a37..ae5b797 100644 --- a/crates/consistent-choose-k/docs/virtual-permutation.md +++ b/crates/consistent-choose-k/docs/virtual-permutation.md @@ -15,6 +15,14 @@ semantics are **different**: | Supported `n` | `1..=2^30` (`u32`) | `1..=u64::MAX` | | Mutable state | Per-layer `Vec` counters | Three `u64` fields; no heap allocation | +An additional **matched-network experiment**, `BalancedVirtualPermutation`, +uses the old iterator's exact Feistel network inside the new cycle-projection +construction. It supports `1..=2^30`, returns `u64` values, and also has 24-byte +allocation-free state. Its `new(n: u32, seed)`, `n`, `replica_at`, `nth` and +iterator operations have the cycle/slot semantics, not survivor-list semantics. +The [matched comparison](virtual-permutation-performance.md#matched-network-follow-up) +records both its performance and its repeatable small-domain statistical bias. + For example, the permutation written as an output list `[2, 0, 1]` is the cycle `0 -> 2 -> 1 -> 0`. Cycle-deleting node 2 produces `[1, 0]`, **not** the survivor list `[0, 1]`. The changed old slot is 0, the slot that @@ -209,6 +217,43 @@ constant-sized width parameter block and no recursive stack, ring, permutation array, or duplicate set. See [measurements and diagnostics](virtual-permutation-performance.md) for observed forward/inverse counts and tails on the actual mixer. +## Matched even-width, two-bit variant + +The cycle-lift identity is not restricted to doubling. For the matched variant +let the full domain have size `4h`, with old set `[0,h)`. Reconnect old +destinations with the same `extend(F_h composed with inverse(R))` operation. +The chain argument and ideal-model uniformity proof above work unchanged. +Start at the smallest **even** bit width covering `n`, use boundary +`1 << (bits - 2)`, and descend by two bits. Intermediate sizes still use cycle +deletion, not output-list deletion. + +`BalancedVirtualPermutation` directly calls the existing `layer_apply` for +every forward evaluation. It uses the same master seed, round function, +12/10/6/4 round schedule at widths 2/4/6/8-and-above, and rotated-key/Weyl +schedule. There is no additional per-width seed hash. The new `layer_inverse` +undoes that exact mapping and key schedule, sharing the extracted round +function. Known-answer vectors, exhaustive small domains, all supported +widths, and extreme keys/inputs check the inverse and unchanged forward +mapping. + +Full-level descent now has probability one quarter in the ideal model, +instead of one half. The initial partial level can be less than half full, +however, so its forward and inverse walks can be longer. Expected work is +still bounded per slot under independent ideal permutations; the specific +less-than-eight-call bound above is for the one-bit variant and is **not** +claimed for the two-bit variant. Inverse evaluation also has real work to +recover the final round key before reversing the rounds; the benchmarks +charge that cost and do not cache it for free. + +This changes **both** layer stride and primitive compared with +`VirtualPermutation`; timing differences are not a pure attribution to odd +versus even halves alone. It does make the primitive and stride identical to +the existing streaming iterator. Neither implementation samples independent +uniform permutations at different widths: the shared 64-bit seed and the +finite Feistel family remain approximations. The matched variant's observed +bias is a substantive limitation, not explained away by the ideal proof. +The stronger one-bit variant remains available unchanged. + ## References and verification The mathematical antecedent is a *virtual permutation*, using **cycle** diff --git a/crates/consistent-choose-k/examples/permutation_diagnostics.rs b/crates/consistent-choose-k/examples/permutation_diagnostics.rs index 20ec0cf..ad1e7d6 100644 --- a/crates/consistent-choose-k/examples/permutation_diagnostics.rs +++ b/crates/consistent-choose-k/examples/permutation_diagnostics.rs @@ -3,7 +3,7 @@ use std::hash::{DefaultHasher, Hash, Hasher}; -use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use consistent_choose_k::{BalancedVirtualPermutation, ConsistentPermutation, VirtualPermutation}; const SAMPLES: u64 = 200_000; fn key_seed(key: u64, workload_seed: u64) -> u64 { @@ -39,13 +39,14 @@ fn main() { ); println!("# samples={SAMPLES}, key_seed={workload_seed:#x}"); println!( - "# iterator_bytes: layered={}, virtual={}; layered also owns heap counters", + "# iterator_bytes: layered={}, virtual={}, balanced={}; layered also owns heap counters", std::mem::size_of::(), std::mem::size_of::(), + std::mem::size_of::(), ); println!("algorithm,n,metric,df,expected_per_cell,chi2,min,max"); for n in [3, 4, 5, 7, 8, 9, 16, 17, 32, 64] { - for algorithm in ["layered", "virtual"] { + for algorithm in ["layered", "virtual", "balanced"] { let slots = n.min(8); let mut marginal = vec![vec![0u64; n]; slots]; let mut pairs = vec![0; n * n]; @@ -60,11 +61,16 @@ fn main() { .take(slots) .map(|v| v as usize) .collect() - } else { + } else if algorithm == "virtual" { VirtualPermutation::new(n as u64, seed) .take(slots) .map(|v| v as usize) .collect() + } else { + BalancedVirtualPermutation::new(n as u32, seed) + .take(slots) + .map(|v| v as usize) + .collect() }; for (histogram, &v) in marginal.iter_mut().zip(&values) { histogram[v] += 1; diff --git a/crates/consistent-choose-k/src/consistent_permutation.rs b/crates/consistent-choose-k/src/consistent_permutation.rs index 4d715c8..fb1d3a5 100644 --- a/crates/consistent-choose-k/src/consistent_permutation.rs +++ b/crates/consistent-choose-k/src/consistent_permutation.rs @@ -75,6 +75,14 @@ fn splitmix64(seed: u64) -> u64 { z ^ (z >> 31) } +#[inline] +fn layer_round(r: u32, key: u64, half_mask: u32) -> u32 { + let k_xor = key & 0xFFFF_FFFF; + let k_mul = (key >> 32) | 1; + let mixed = ((r as u64) ^ k_xor).wrapping_mul(k_mul); + (mixed as u32).wrapping_add((mixed >> 32) as u32) & half_mask +} + /// Apply the per-layer Feistel bijection on `[0, 2^n_bits)`. /// `master_key` must be well-avalanched: two near-identical keys /// will yield two highly correlated permutations. @@ -104,10 +112,7 @@ pub(crate) fn layer_apply(n_bits: u32, master_key: u64, x: u32) -> u32 { // high half (forced odd) is the multiplier. Higher bits stay // set on purpose — they contribute via the multiplicative // mix. - let k_xor = k & 0xFFFF_FFFF; - let k_mul = (k >> 32) | 1; - let mixed = ((r as u64) ^ k_xor).wrapping_mul(k_mul); - let f = (mixed as u32).wrapping_add((mixed >> 32) as u32) & half_mask; + let f = layer_round(r, k, half_mask); let new_l = r; let new_r = l ^ f; l = new_l; @@ -122,6 +127,30 @@ pub(crate) fn layer_apply(n_bits: u32, master_key: u64, x: u32) -> u32 { ((l << half_bits) | r) & n_mask } +/// Invert exactly the existing layer permutation, including its key schedule. +#[inline] +pub(crate) fn layer_inverse(n_bits: u32, master_key: u64, x: u32) -> u32 { + debug_assert!((2..=30).contains(&n_bits) && n_bits.is_multiple_of(2)); + let rounds = rounds_for_n_bits(n_bits); + let half_bits = n_bits / 2; + let half_mask = (1u32 << half_bits) - 1; + debug_assert!(x < 1u32 << n_bits, "input out of range"); + let mut l = x >> half_bits; + let mut r = x & half_mask; + let mut key = master_key; + for _ in 1..rounds { + key = key.rotate_right(n_bits).wrapping_add(0x9E37_79B9_7F4A_7C15); + } + for _ in 0..rounds { + let old_r = l; + let old_l = r ^ layer_round(l, key, half_mask); + l = old_l; + r = old_r; + key = key.wrapping_sub(0x9E37_79B9_7F4A_7C15).rotate_left(n_bits); + } + (l << half_bits) | r +} + /// `n`-consistent permutation iterator over `0..n` driven by one /// bijection per level (see module docs). pub struct ConsistentPermutation { @@ -237,6 +266,38 @@ mod tests { use super::*; + #[test] + fn layer_mapping_known_answers_and_inverse() { + for (bits, key, x, y) in [ + (2, 0, 0, 0), + (4, 0x1234_5678_9abc_def0, 9, 2), + (6, 1, 63, 62), + (8, 0x1234_5678_9abc_def0, 200, 100), + (16, u64::MAX, 65535, 0x3a83), + (30, 0x1234_5678_9abc_def0, 0x3fff_ffff, 0x1949_4328), + ] { + assert_eq!(layer_apply(bits, key, x), y); + assert_eq!(layer_inverse(bits, key, y), x); + } + for bits in (2..=30).step_by(2) { + let mask = (1u32 << bits) - 1; + for key in [0, 1, u64::MAX, splitmix64(42)] { + let inputs: Vec<_> = if bits <= 10 { + (0..=mask).collect() + } else { + [0, 1, mask / 2, mask / 2 + 1, mask] + .into_iter() + .chain((0..128).map(|i| splitmix64(i) as u32 & mask)) + .collect() + }; + for x in inputs { + assert_eq!(layer_inverse(bits, key, layer_apply(bits, key, x)), x); + assert_eq!(layer_apply(bits, key, layer_inverse(bits, key, x)), x); + } + } + } + } + /// Across many seeds, `layer_apply(_, key, 0)` must not always /// collapse to the same value (regression test for the "F(0, k) = /// 0" symmetry bug). diff --git a/crates/consistent-choose-k/src/lib.rs b/crates/consistent-choose-k/src/lib.rs index 4278f83..1b2e45a 100644 --- a/crates/consistent-choose-k/src/lib.rs +++ b/crates/consistent-choose-k/src/lib.rs @@ -12,4 +12,4 @@ pub use consistent_hash::{ pub use consistent_permutation::ConsistentPermutation; pub use consistent_reservoir::ConsistentReservoir; pub use node_map::ConsistentNodeMap; -pub use virtual_permutation::VirtualPermutation; +pub use virtual_permutation::{BalancedVirtualPermutation, VirtualPermutation}; diff --git a/crates/consistent-choose-k/src/virtual_permutation.rs b/crates/consistent-choose-k/src/virtual_permutation.rs index 76be660..d55bd3b 100644 --- a/crates/consistent-choose-k/src/virtual_permutation.rs +++ b/crates/consistent-choose-k/src/virtual_permutation.rs @@ -3,6 +3,8 @@ use std::iter::FusedIterator; +use crate::consistent_permutation::{layer_apply, layer_inverse}; + const WEYL: u64 = 0x9E37_79B9_7F4A_7C15; #[inline] @@ -90,10 +92,21 @@ impl Permutation for WordPermutation { } #[inline] -fn evaluate(mut n: u64, mut x: u64, mut layer: impl FnMut(u32) -> P) -> u64 { +fn evaluate(n: u64, x: u64, layer: impl FnMut(u32) -> P) -> u64 { + evaluate_with_stride::<1, P>(n, x, layer) +} + +#[inline] +fn evaluate_with_stride( + mut n: u64, + mut x: u64, + mut layer: impl FnMut(u32) -> P, +) -> u64 { + debug_assert!(STEP == 1 || STEP == 2); let mut bits = u64::BITS - (n - 1).leading_zeros(); + bits = bits.div_ceil(STEP) * STEP; while bits > 0 { - let half = 1u64 << (bits - 1); + let boundary = 1u64 << (bits - STEP); let permutation = layer(bits); loop { let y = permutation.forward(x); @@ -101,19 +114,19 @@ fn evaluate(mut n: u64, mut x: u64, mut layer: impl FnMut(u32) - x = y; continue; } - if y >= half { + if y >= boundary { return y; } // Find the old input at the start of this Q-chain, not its old // output y: this applies inverse(cycle_projection(Q)) before // evaluating the smaller consistent permutation. - while x >= half { + while x >= boundary { x = permutation.inverse(x); } break; } - n = half; - bits -= 1; + n = boundary; + bits -= STEP; } 0 } @@ -215,6 +228,110 @@ impl Iterator for VirtualPermutation { impl FusedIterator for VirtualPermutation {} +struct ExistingPermutation { + bits: u32, + seed: u64, +} + +impl Permutation for ExistingPermutation { + #[inline] + fn forward(&self, x: u64) -> u64 { + u64::from(layer_apply(self.bits, self.seed, x as u32)) + } + + #[inline] + fn inverse(&self, x: u64) -> u64 { + u64::from(layer_inverse(self.bits, self.seed, x as u32)) + } +} + +/// Experimental cycle-consistent replica slots using exactly the Feistel +/// network of [`crate::ConsistentPermutation`], including its round counts and +/// key schedule, with two bits per lift. +/// +/// This has the slot-consistency semantics of [`VirtualPermutation`], **not** +/// the survivor-list semantics of `ConsistentPermutation`. The supplied +/// well-mixed seed is passed unchanged to the existing network; no additional +/// width-domain seed mixer is inserted. Consequently its statistical quality +/// must be assessed separately from the independently keyed ideal model. +/// It is noncryptographic and does not promise exact uniformity. +/// +/// **Statistical caution:** the matched-network diagnostics show repeatable +/// small-domain bias. This variant is provided for comparison, not as a +/// statistically equivalent substitute for `VirtualPermutation`. +/// +/// State is allocation-free. The supported domain matches the existing +/// network: `1..=2^30` nodes. Iteration returns `u64`, as `VirtualPermutation` +/// does, and direct slot evaluation does not replay earlier slots. +/// +/// ``` +/// use consistent_choose_k::BalancedVirtualPermutation; +/// +/// let permutation = BalancedVirtualPermutation::new(100, 0x1234_5678_9abc_def0); +/// assert_eq!(permutation.clone().nth(2), Some(permutation.replica_at(2))); +/// assert_eq!(permutation.take(3).count(), 3); +/// ``` +#[derive(Clone, Debug)] +pub struct BalancedVirtualPermutation { + inner: VirtualPermutation, +} + +impl BalancedVirtualPermutation { + /// Construct an iterator with the existing Feistel network and seed. + /// + /// # Panics + /// + /// Panics unless `1 <= n <= 2^30`. + pub fn new(n: u32, seed: u64) -> Self { + assert!(n <= 1u32 << 30, "n must be at most 2^30"); + Self { + inner: VirtualPermutation::new(u64::from(n), seed), + } + } + + /// Universe size, independent of the iterator's position. + pub fn n(&self) -> u64 { + self.inner.n() + } + + /// Evaluate an absolute slot without advancing the iterator. + /// + /// # Panics + /// + /// Panics if `slot >= self.n()`. + pub fn replica_at(&self, slot: u64) -> u64 { + assert!(slot < self.inner.n, "replica slot must be less than n"); + evaluate_with_stride::<2, _>(self.inner.n, slot, |bits| ExistingPermutation { + bits, + seed: self.inner.seed, + }) + } +} + +impl Iterator for BalancedVirtualPermutation { + type Item = u64; + + fn next(&mut self) -> Option { + if self.inner.next == self.inner.n { + return None; + } + let value = self.replica_at(self.inner.next); + self.inner.next += 1; + Some(value) + } + + fn size_hint(&self) -> (usize, Option) { + self.inner.size_hint() + } + + fn nth(&mut self, n: usize) -> Option { + self.inner.next = self.inner.next.saturating_add(n as u64).min(self.inner.n); + self.next() + } +} + +impl FusedIterator for BalancedVirtualPermutation {} + #[cfg(test)] mod tests { use super::*; @@ -432,6 +549,115 @@ mod tests { } } + #[test] + fn balanced_permutations_match_explicit_quarter_lifts() { + for key in 0..32 { + let key = seed(key); + let mut full = vec![0]; + for bits in (2..=8).step_by(2) { + let q: Vec<_> = (0..1 << bits) + .map(|x| u64::from(layer_apply(bits, key, x))) + .collect(); + full = lift(&full, &q); + for n in (full.len() / 4 + 1)..=full.len() { + let iter = BalancedVirtualPermutation::new(n as u32, key); + let actual: Vec<_> = iter.clone().collect(); + assert_eq!(actual, project(&full, n)); + let mut sorted = actual.clone(); + sorted.sort_unstable(); + assert_eq!(sorted, (0..n as u64).collect::>()); + for k in [0, 1, n / 2, n] { + assert_eq!(iter.clone().take(k).collect::>(), actual[..k]); + } + for (slot, &value) in actual.iter().enumerate() { + assert_eq!(iter.replica_at(slot as u64), value); + } + let smaller: Vec<_> = + BalancedVirtualPermutation::new(n as u32 - 1, key).collect(); + assert_eq!(project(&actual, n - 1), smaller); + let changed: Vec<_> = smaller + .iter() + .zip(&actual) + .filter(|(old, new)| old != new) + .collect(); + assert!(changed.len() <= 1); + assert!(changed.iter().all(|(_, new)| **new == n as u64 - 1)); + } + } + } + } + + #[test] + fn balanced_large_domains_and_iterator_boundaries() { + for bits in 1..30 { + for n in [(1u32 << bits) - 1, 1 << bits, (1 << bits) + 1] { + for key in [0, 1, u64::MAX, seed(42)] { + let p = BalancedVirtualPermutation::new(n, key); + let larger = BalancedVirtualPermutation::new(n + 1, key); + for slot in [0, u64::from(n / 2), u64::from(n - 1)] { + let old = p.replica_at(slot); + let new = larger.replica_at(slot); + assert!(old < u64::from(n)); + assert!(new == old || new == u64::from(n)); + assert_eq!( + old, + if new == u64::from(n) { + larger.replica_at(new) + } else { + new + } + ); + } + } + } + } + let mut p = BalancedVirtualPermutation::new(1, 0); + assert_eq!(p.n(), 1); + assert_eq!(p.size_hint(), (1, Some(1))); + assert_eq!(p.next(), Some(0)); + assert_eq!(p.size_hint(), (0, Some(0))); + assert_eq!(p.next(), None); + assert_eq!(p.nth(usize::MAX), None); + let mut p = BalancedVirtualPermutation::new(1 << 30, seed(9)); + assert_eq!(p.nth((1 << 30) - 1), Some(p.replica_at((1 << 30) - 1))); + assert_eq!(p.next(), None); + } + + #[test] + fn balanced_primary_matches_existing_at_full_powers_of_four() { + for bits in (0..=30).step_by(2) { + for key in 0..128 { + let key = seed(key); + let n = 1u32 << bits; + let expected = crate::ConsistentPermutation::new(n, key) + .next() + .expect("nonempty domain"); + assert_eq!( + BalancedVirtualPermutation::new(n, key).replica_at(0), + u64::from(expected) + ); + } + } + } + + #[test] + #[should_panic(expected = "n must be at least 1")] + fn balanced_invalid_empty_domain() { + BalancedVirtualPermutation::new(0, 0); + } + + #[test] + #[should_panic(expected = "n must be at most 2^30")] + fn balanced_invalid_large_domain() { + BalancedVirtualPermutation::new((1 << 30) + 1, 0); + } + + #[test] + #[should_panic(expected = "replica slot must be less than n")] + fn balanced_invalid_slot() { + BalancedVirtualPermutation::new(1, 0).replica_at(1); + } + fn permutations(n: usize) -> Vec> { fn visit(values: &mut [u64], start: usize, out: &mut Vec>) { if start == values.len() { @@ -486,11 +712,22 @@ mod tests { fn operation_count_diagnostics() { use std::{cell::Cell, rc::Rc}; - struct Counted { - permutation: WordPermutation, + struct Counted

{ + permutation: P, calls: Rc>, } - impl Permutation for Counted { + impl

Counted

{ + fn new(permutation: P, calls: &Rc>) -> Self { + let mut counts = calls.get(); + counts[2] += 1; + calls.set(counts); + Self { + permutation, + calls: Rc::clone(calls), + } + } + } + impl Permutation for Counted

{ fn forward(&self, x: u64) -> u64 { let mut calls = self.calls.get(); calls[0] += 1; @@ -505,7 +742,9 @@ mod tests { } } - println!("n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls"); + println!( + "algorithm,n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls" + ); for n in [ 1, 7, @@ -525,38 +764,49 @@ mod tests { (1 << 63) + 1, u64::MAX, ] { - for slot in [0, n / 2, n - 1] { - let calls = Rc::new(Cell::new([0; 3])); - let mut totals = [0u64; 3]; - let mut samples = vec![]; - for key in 0..10_000 { - calls.set([0; 3]); - let result = evaluate(n, slot, |bits| { - let mut counts = calls.get(); - counts[2] += 1; - calls.set(counts); - Counted { - permutation: WordPermutation::new(seed(key), bits), - calls: Rc::clone(&calls), + for algorithm in ["virtual", "balanced"] { + if algorithm == "balanced" && n > 1 << 30 { + continue; + } + for slot in [0, n / 2, n - 1] { + let calls = Rc::new(Cell::new([0; 3])); + let mut totals = [0u64; 3]; + let mut samples = vec![]; + for key in 0..10_000 { + calls.set([0; 3]); + let result = if algorithm == "virtual" { + evaluate(n, slot, |bits| { + Counted::new(WordPermutation::new(seed(key), bits), &calls) + }) + } else { + evaluate_with_stride::<2, _>(n, slot, |bits| { + Counted::new( + ExistingPermutation { + bits, + seed: seed(key), + }, + &calls, + ) + }) + }; + assert!(result < n); + let counts = calls.get(); + for i in 0..3 { + totals[i] += counts[i]; } - }); - assert!(result < n); - let counts = calls.get(); - for i in 0..3 { - totals[i] += counts[i]; + samples.push(counts[0] + counts[1]); } - samples.push(counts[0] + counts[1]); + samples.sort_unstable(); + println!( + "{algorithm},{n},{slot},{:.4},{:.4},{:.4},{},{},{}", + totals[0] as f64 / 10_000.0, + totals[1] as f64 / 10_000.0, + totals[2] as f64 / 10_000.0, + samples[4999], + samples[9899], + samples[9999], + ); } - samples.sort_unstable(); - println!( - "{n},{slot},{:.4},{:.4},{:.4},{},{},{}", - totals[0] as f64 / 10_000.0, - totals[1] as f64 / 10_000.0, - totals[2] as f64 / 10_000.0, - samples[4999], - samples[9899], - samples[9999], - ); } } } From 44bd94d0d3136caa4775521c20a28ba70d399fb4 Mon Sep 17 00:00:00 2001 From: Alexander Neubeck Date: Wed, 30 Sep 2026 11:32:04 +0200 Subject: [PATCH 3/3] Replace slot alternative with sentinel-rooted survivor ordering Traverse one consistent cycle from a permanent sentinel, restore the existing implementation unchanged, and replace direct-rank APIs with sequential replay. Add full survivor-order and amortization invariants, final-construction randomness diagnostics, and a fresh paired performance report. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- crates/consistent-choose-k/README.md | 43 +- .../benchmarks/replica_comparison.rs | 87 +- .../summarize_replica_comparison.py | 6 +- .../docs/permutation-design.md | 9 +- .../docs/virtual-permutation-performance.md | 708 ++++++------- .../docs/virtual-permutation.md | 447 ++++----- .../examples/permutation_diagnostics.rs | 273 +++-- .../src/consistent_permutation.rs | 69 +- crates/consistent-choose-k/src/lib.rs | 2 +- .../src/virtual_permutation.rs | 929 ++++++++---------- 10 files changed, 1172 insertions(+), 1401 deletions(-) diff --git a/crates/consistent-choose-k/README.md b/crates/consistent-choose-k/README.md index 3cef1e5..fd9c105 100644 --- a/crates/consistent-choose-k/README.md +++ b/crates/consistent-choose-k/README.md @@ -40,32 +40,29 @@ Why replication matters ## Permutation APIs and membership semantics The existing `ConsistentPermutation` preserves **survivor list order** when -nodes are appended or removed from the end of `0..n`. The additional, -experimental `VirtualPermutation` instead preserves **replica slots**: on a -single-node append, at most one old slot changes, to the new node; on removal, -only a surviving slot that named that node changes. It does not preserve list -restriction and is not a drop-in replacement for the existing iterator or -its failover policies. Both yield distinct nodes and stable `k` prefixes. - -`VirtualPermutation::new(n, seed)` supports `1..=u64::MAX`, allocates no state, -and offers both an iterator and absolute `replica_at(slot)` lookup. Expected -`O(k)` enumeration follows under ideal independent uniform permutations with -constant-cost forward/inverse evaluation, **not** as a worst-case guarantee. -The implemented noncryptographic, 64-bit seeded Feistel family approximates -that randomness model; exact uniformity and independence are not claimed. +nodes are appended or removed from the end of `0..n`. The experimental +`VirtualPermutation` now supplies the same order-restriction contract using a +different construction: it traverses a single consistent cycle from a permanent +sentinel. Removing the appended node from the larger complete order recovers +the smaller complete order. Both yield distinct nodes and stable `k` prefixes; +replica ranks may shift when membership changes. + +`VirtualPermutation::new(n, seed)` supports `1..=u64::MAX - 1` real nodes, +uses four `u64` fields and no heap state, and offers sequential iteration. +The extra internal label is reserved for the sentinel. `nth(r)` replays +`r + 1` successors; there is no direct rank lookup. Expected `O(k)` prefix +enumeration follows under ideal independent uniform permutations with +constant-cost forward/inverse primitives, **not** as a worst-case guarantee. +The noncryptographic, 64-bit seeded Feistel family approximates that model; +exact uniformity and independence are not claimed. This implementation +replaces the earlier experimental slot-based variants and changes their +output mappings; the existing `ConsistentPermutation` mapping is unchanged. See the [algorithm, proof assumptions and API guide](docs/virtual-permutation.md) and the [reproducible comparison with the existing algorithm](docs/virtual-permutation-performance.md). -The existing APIs and their mappings remain unchanged. - -`BalancedVirtualPermutation` is a matched-network experiment: it gives the -same cycle/slot semantics as `VirtualPermutation`, but uses **exactly** the -existing `ConsistentPermutation` Feistel, with even widths and two bits per -lift, over `1..=2^30`. Its state is also allocation-free. The -[three-way comparison](docs/virtual-permutation-performance.md#matched-network-follow-up) -repeats performance and primary/held-out statistical diagnostics. This variant -has repeatable small-domain distribution bias and is not the default or a -statistically equivalent replacement for the stronger mixer. +The comparison reports fresh-key performance regressions as well as fresh +primary and held-out randomness diagnostics; neither algorithm is replaced +in existing consumers. ## Applications beyond replication diff --git a/crates/consistent-choose-k/benchmarks/replica_comparison.rs b/crates/consistent-choose-k/benchmarks/replica_comparison.rs index 60ac48e..c39d9df 100644 --- a/crates/consistent-choose-k/benchmarks/replica_comparison.rs +++ b/crates/consistent-choose-k/benchmarks/replica_comparison.rs @@ -7,35 +7,46 @@ use std::{ time::Duration, }; -use consistent_choose_k::{BalancedVirtualPermutation, ConsistentPermutation, VirtualPermutation}; -use criterion::{criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, Throughput}; +use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; +use criterion::{ + criterion_group, criterion_main, BatchSize, BenchmarkId, Criterion, SamplingMode, Throughput, +}; use rand::{rngs::StdRng, RngExt, SeedableRng}; const WORKLOAD_SEED: u64 = 0x7065_726d_7574_6531; const KEY_COUNT: usize = 128; const NODES: &[u32] = &[ 1, + 2, 3, + 4, + 6, 7, 8, 9, + 14, 15, 16, 17, + 30, 31, 32, 33, + 254, 255, 256, 257, 1000, + 1022, 1023, 1024, 1025, + 65534, 65535, 65536, 65537, 1_000_000, + (1 << 30) - 2, (1 << 30) - 1, 1 << 30, ]; @@ -75,10 +86,11 @@ fn end_to_end(c: &mut Criterion) { let keys = keys(); let seeds: Vec<_> = keys.iter().copied().map(hash_key).collect(); for mode in ["fresh", "seeded"] { - let mut group = c.benchmark_group(format!("replicas/{mode}")); + let mut group = c.benchmark_group(format!("sentinel_replicas/{mode}")); // Both execute exactly one complete query per key, including // construction, streaming consumption and state destruction. group.throughput(Throughput::Elements(KEY_COUNT as u64)); + group.sampling_mode(SamplingMode::Flat); for &n in NODES { for k in counts(n) { let input = if mode == "fresh" { &keys } else { &seeds }; @@ -90,7 +102,7 @@ fn end_to_end(c: &mut Criterion) { } }) }); - group.bench_function(BenchmarkId::new("virtual", format!("n{n}_k{k}")), |b| { + group.bench_function(BenchmarkId::new("sentinel", format!("n{n}_k{k}")), |b| { b.iter(|| { for &key in black_box(input) { let seed = if mode == "fresh" { hash_key(key) } else { key }; @@ -101,17 +113,6 @@ fn end_to_end(c: &mut Criterion) { } }) }); - group.bench_function(BenchmarkId::new("balanced", format!("n{n}_k{k}")), |b| { - b.iter(|| { - for &key in black_box(input) { - let seed = if mode == "fresh" { hash_key(key) } else { key }; - consume( - BalancedVirtualPermutation::new(black_box(n), seed), - black_box(k), - ); - } - }) - }); } } group.finish(); @@ -121,8 +122,9 @@ fn end_to_end(c: &mut Criterion) { fn cost_components(c: &mut Criterion) { let keys = keys(); let seeds: Vec<_> = keys.iter().copied().map(hash_key).collect(); - let mut setup = c.benchmark_group("replicas/setup"); + let mut setup = c.benchmark_group("sentinel_replicas/setup"); setup.throughput(Throughput::Elements(KEY_COUNT as u64)); + setup.sampling_mode(SamplingMode::Flat); setup.bench_function("hash_u64", |b| { b.iter(|| { for &key in black_box(&keys) { @@ -138,26 +140,20 @@ fn cost_components(c: &mut Criterion) { } }) }); - setup.bench_function(BenchmarkId::new("virtual", n), |b| { + setup.bench_function(BenchmarkId::new("sentinel", n), |b| { b.iter(|| { for &seed in black_box(&seeds) { black_box(VirtualPermutation::new(u64::from(black_box(n)), seed)); } }) }); - setup.bench_function(BenchmarkId::new("balanced", n), |b| { - b.iter(|| { - for &seed in black_box(&seeds) { - black_box(BalancedVirtualPermutation::new(black_box(n), seed)); - } - }) - }); } setup.finish(); - for mode in ["stream_only", "collect", "slot"] { - let mut group = c.benchmark_group(format!("replicas/{mode}")); + for mode in ["stream_only", "collect", "rank_replay"] { + let mut group = c.benchmark_group(format!("sentinel_replicas/{mode}")); group.throughput(Throughput::Elements(KEY_COUNT as u64)); + group.sampling_mode(SamplingMode::Flat); for &n in &[17, 257, 1000, 65537, 1 << 30] { for k in counts(n) { group.bench_function(BenchmarkId::new("layered", format!("n{n}_k{k}")), |b| { @@ -185,14 +181,14 @@ fn cost_components(c: &mut Criterion) { out.extend(iter.take(k).map(u64::from)); black_box(out); } else { - // No random-access API in the baseline: nth must replay. + // Both methods replay the prefix to answer a rank query. black_box(iter.nth(black_box(k - 1))); } } }), } }); - group.bench_function(BenchmarkId::new("virtual", format!("n{n}_k{k}")), |b| { + group.bench_function(BenchmarkId::new("sentinel", format!("n{n}_k{k}")), |b| { match mode { "stream_only" => b.iter_batched_ref( || { @@ -210,43 +206,14 @@ fn cost_components(c: &mut Criterion) { ), _ => b.iter(|| { for &seed in black_box(&seeds) { - let iter = VirtualPermutation::new(u64::from(black_box(n)), seed); - if mode == "collect" { - let mut out = Vec::with_capacity(black_box(k)); - out.extend(iter.take(k)); - black_box(out); - } else { - black_box(iter.replica_at(black_box(k as u64 - 1))); - } - } - }), - } - }); - group.bench_function(BenchmarkId::new("balanced", format!("n{n}_k{k}")), |b| { - match mode { - "stream_only" => b.iter_batched_ref( - || { - seeds - .iter() - .map(|&seed| BalancedVirtualPermutation::new(n, seed)) - .collect::>() - }, - |iterators| { - for iter in black_box(iterators) { - consume(iter, black_box(k)); - } - }, - BatchSize::SmallInput, - ), - _ => b.iter(|| { - for &seed in black_box(&seeds) { - let iter = BalancedVirtualPermutation::new(black_box(n), seed); + let mut iter = + VirtualPermutation::new(u64::from(black_box(n)), seed); if mode == "collect" { let mut out = Vec::with_capacity(black_box(k)); out.extend(iter.take(k)); black_box(out); } else { - black_box(iter.replica_at(black_box(k as u64 - 1))); + black_box(iter.nth(black_box(k - 1))); } } }), diff --git a/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py b/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py index 0fa9f65..e43de38 100644 --- a/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py +++ b/crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py @@ -17,9 +17,9 @@ def summarize(root): for metadata in root.glob("**/new/benchmark.json"): benchmark = json.loads(metadata.read_text()) group = benchmark["group_id"] - if not group.startswith("replicas/"): + if not group.startswith("sentinel_replicas/"): continue - mode = group.removeprefix("replicas/") + mode = group.removeprefix("sentinel_replicas/") algorithm = benchmark["function_id"] value = benchmark.get("value_str") or "" match = re.fullmatch(r"n(\d+)_k(\d+)", value) @@ -38,7 +38,7 @@ def summarize(root): interval["lower_bound"] / divisor, interval["upper_bound"] / divisor, estimates["std_dev"]["point_estimate"] / divisor, - mean["point_estimate"] / divisor / (k if k and mode != "slot" else 1), + mean["point_estimate"] / divisor / (k if k and mode != "rank_replay" else 1), ) ) if not rows: diff --git a/crates/consistent-choose-k/docs/permutation-design.md b/crates/consistent-choose-k/docs/permutation-design.md index 9fd144c..79a9360 100644 --- a/crates/consistent-choose-k/docs/permutation-design.md +++ b/crates/consistent-choose-k/docs/permutation-design.md @@ -4,10 +4,11 @@ This document explains the design of [`ConsistentPermutation`], the per-layer Feistel permutation iterator that this crate uses to drive its `n`-consistent ranking. -This is **survivor-list consistency**, not the cycle-projection/replica-slot -consistency of the additional [`VirtualPermutation`](virtual-permutation.md). -The algorithms are not interchangeable membership policies. See their -[paired performance comparison](virtual-permutation-performance.md). +This is **survivor-list consistency**. The experimental +[`VirtualPermutation`](virtual-permutation.md) now supplies the same membership +contract through a sentinel-rooted single cycle, with different mappings, +primitive costs and state requirements. See their +[paired performance and randomness comparison](virtual-permutation-performance.md). Uniformity arguments below model the per-layer bijections as independent uniform permutations; the actual finite-key, noncryptographic Feistel family is a practical approximation, not an exact uniform sample from all permutations. diff --git a/crates/consistent-choose-k/docs/virtual-permutation-performance.md b/crates/consistent-choose-k/docs/virtual-permutation-performance.md index e16731d..acb7072 100644 --- a/crates/consistent-choose-k/docs/virtual-permutation-performance.md +++ b/crates/consistent-choose-k/docs/virtual-permutation-performance.md @@ -1,434 +1,314 @@ -# Replica-selection comparison - -The original `VirtualPermutation` trades slower sequential enumeration for -allocation-free state and direct replica-slot evaluation. It is **not a faster -drop-in replacement** for the existing `ConsistentPermutation`. - -The [matched-network follow-up](#matched-network-follow-up) adds -`BalancedVirtualPermutation`, which uses exactly the existing Feistel on -two-bit layers. All three methods are remeasured together; repeatable -small-domain bias in the matched variant is reported rather than hidden. -The original two-way tables below are historical measurements at commit -`d51abfb`, not fresh measurements of the follow-up. - -The existing algorithm preserves survivor **list order**; the new one preserves -each old **slot** except a slot replaced by an appended node (or previously -naming a removed node). Both provide distinct, `k`-prefix-stable selections, -but these membership policies are not interchangeable. The -[design note](virtual-permutation.md) explains the construction, ideal-model -proof, noncryptographic approximation and expected-versus-worst-case distinction. - -Runner names are `layered` for the existing streaming iterator, `virtual` -for the stronger one-bit cycle construction, and `balanced` for the -matched-Feistel two-bit cycle construction. +# Sentinel-rooted permutation comparison + +The revised `VirtualPermutation` preserves complete survivor-list order, +like the unchanged `ConsistentPermutation`, but is **substantially slower** +in this implementation. It removes heap state, not computational work. +Fresh-key `n=1000,k=3` costs 351.25 versus 36.48 ns/query (9.63x slower); +full enumeration costs 195,818.68 versus 10,761.56 ns (18.20x slower). +No comparably large, repeatable distribution deviations appeared for the +new construction in the primary/held-out diagnostic matrix; that is not +a proof of uniformity or independence. + +All results below were measured anew on the **single-cycle plus sentinel** +construction. They replace the earlier slot-based and matched-network reports. +There is no constant-time output-rank lookup or direct-slot speedup: +both iterators replay a prefix for `nth`. See the +[design, bounds and ideal-model derivation](virtual-permutation.md). +Runner names are `layered` (existing) and `sentinel` (new). ## Reproduce -From the repository root: +From the repository root, run sequentially, without other CPU-heavy work: ```sh cargo test -p consistent-choose-k +cargo test --release -p consistent-choose-k cargo bench -p consistent-choose-k-benchmarks --bench replica_comparison -- --noplot python3 crates/consistent-choose-k/benchmarks/summarize_replica_comparison.py \ target/criterion > comparison.csv -cargo run --release -p consistent-choose-k --example permutation_diagnostics -cargo run --release -p consistent-choose-k --example permutation_diagnostics -- --held-out +cargo run --release -p consistent-choose-k --example permutation_diagnostics \ + > diagnostics.csv +cargo run --release -p consistent-choose-k --example permutation_diagnostics \ + -- --held-out > held-out.csv cargo test --release -p consistent-choose-k operation_count_diagnostics \ -- --ignored --nocapture ``` -Run these **sequentially**, not alongside other CPU-intensive tasks. Criterion -filters can restrict a rerun, for example -`-- 'replicas/fresh/.*/n1000_k3$' --noplot`. -Raw sample timings and estimates remain under `target/criterion`; the script -emits mean ns/query, bootstrap 95% confidence interval, sample standard deviation -in ns/query, and mean ns/replica. Criterion's console times are for a **batch of -128 queries**, so divide by 128, not by `k`, to obtain ns/query. Divide again by -`k` for ns/replica (a slot query returns just one result). - -### Recorded environment - -Measured on September 24, 2026, on an Apple M4 Max, native -`aarch64-apple-darwin`, macOS 27.0 build 26A428: - -- `rustc 1.92.0 (ded5c06cf 2025-12-08)`, LLVM 21.1.3. -- Apple clang 21.0.0 (`clang-2100.1.1.101`). -- Repository bench profile: optimized with debug info; configured - `-C target-feature=+neon`; no extra LTO, native-CPU flags or PGO. -- Criterion 0.8.2, rand 0.10.3; 20 samples, 100 ms warmup, 300 ms target - measurement time per case, 1,000 bootstrap resamples, no plots. -- The default selected Xcode could not link because its license was - unaccepted. All successful commands used the independently installed - Command Line Tools via - `DEVELOPER_DIR=/Library/Developer/CommandLineTools`. - No license was accepted and no system setting was changed. - -These are bounded microbenchmarks on a shared host, without CPU pinning, -frequency control or isolation. Intervals describe repeated timing samples of -one fixed key corpus, **not** uncertainty across all keys, machines or compiler -versions. Algorithms ran sequentially, existing first in each pair. Small -differences should not be interpreted as portable wins. +A Criterion filter can select a bounded rerun, for example +`-- 'sentinel_replicas/fresh/.*/n1000_k3$' --noplot`. +The new `sentinel_replicas/` result namespace excludes stale measurements of +the replaced algorithms. Raw samples/estimates remain under `target/criterion`. +The CSV script reports mean ns/query, bootstrap 95% confidence limits, sample +standard deviation and ns/output. Criterion's console times are **batches of +128 queries**: divide by 128 for ns/query, then by `k` for ns/output. +`rank_replay` returns only one output, so its ns/output equals ns/query. + +### Environment and limitations + +Measured September 30, 2026, on Apple M4 Max, native `aarch64-apple-darwin`, +macOS 27.0 build 26A428; `rustc 1.92.0 (ded5c06cf 2025-12-08)`, +LLVM 21.1.3; Apple clang 21.0.0 (`clang-2100.1.1.101`). +The repository bench profile is optimized with debug information and its +configured `-C target-feature=+neon`; no additional LTO, PGO or native-CPU +flags. Criterion 0.8.2 and rand 0.10.3 were resolved locally. + +Each case uses 20 flat samples, 100 ms warmup, 300 ms target measurement, +1,000 bootstrap resamples and no plots. Flat sampling bounds expensive +full-prefix cases; actual durations can exceed the target. All successful +local Cargo commands used +`DEVELOPER_DIR=/Library/Developer/CommandLineTools` to select the independently +installed CLT instead of the default Xcode whose license was unaccepted. +No license was accepted or system setting changed. + +This is a shared host without CPU pinning, frequency control or isolation. +Algorithms run sequentially, existing first in each pair. Intervals concern +repeated timing samples of **one fixed key corpus**, not uncertainty across +all keys, machines or compilers. Some cases are noisy; the wide interval at +`n=257,k=3` is retained rather than discarded. Small differences are not +portable wins. ### Workload and accounting -The fixture is 128 `u64` keys generated by `StdRng` with seed -`0x7065726d75746531`. The fresh-key mode hashes each key with `DefaultHasher` -inside every query, identically for all methods. Width/round preparation -and all inverse walks in the new evaluator are included, as is the existing -iterator's counter allocation. These are fresh **queries** of a repeated -deterministic corpus, not new unpredictable keys on every benchmark iteration. -`StdRng` and `DefaultHasher` are not cross-version reproducibility contracts; -use the recorded toolchain and dependency versions to reproduce the exact -corpus and mapping. - -| Group | Timed work | +The fixture contains 128 `u64` keys from `StdRng`, seed +`0x7065726d75746531`. Fresh queries hash with `DefaultHasher` inside the timed +region, equally for both algorithms. This is repeated fresh **setup** on a +fixed corpus, not unpredictable new keys every iteration. `StdRng` and +`DefaultHasher` are not cross-version mapping contracts: use the recorded +compiler/dependency versions for the exact corpus. + +| Mode | Timed work | | --- | --- | -| `fresh` (primary comparison) | Key hashing, construction, streaming checksum of `k` values, destruction | -| `seeded` | Same query, reusing prehashed seeds; no free algorithm-specific cache | -| `setup` | Hash alone, or construction and destruction from a prehashed seed | -| `stream_only` | Consume `k` values from separately prepared iterators; construction and destruction excluded equally with `iter_batched_ref` | -| `collect` | Prehashed construction, allocate/fill/drop `Vec` with the same capacity `k`, destroy iterator | -| `slot` | Prehashed construction plus the result at `k-1`; existing `.nth(k-1)` replays, new `replica_at(k-1)` does not | - -The last row is a comparison of ways to answer a single-slot query, **not** a -claim that the existing API has a constant-time random-access operation. -Streaming-only is a diagnostic component benchmark, not a substitute for the -fresh-query comparison: prepared state has different cache footprints. -Inputs pass through `black_box`, and outputs are consumed. Both collection -paths use `u64` output elements, even though the old API returns `u32`. - -The full paired matrix uses -`n = 1,3,7,8,9,15,16,17,31,32,33,255,256,257,1000,1023,1024,1025,65535,65536,65537,1000000,2^30-1,2^30`. -For each, it uses valid `k` from `1,2,3,8,16`; at `n <= 1024`, it also uses -`floor(n/4)` and `n`, removing zeros and duplicates. There are 131 `(n,k)` -pairs in each of the fresh and seeded groups. Component groups use -`n = 17,257,1000,65537,2^30`. All paired timings stay within the old -implementation's supported domain. No wider-domain speedup is inferred. - -## Original fresh-query results - -Representative mean **ns/query**, with bootstrap 95% confidence intervals in -brackets. Ratio is new/existing time: **above one means a regression**. - -| n | k | Existing layered | New virtual | Ratio | +| `fresh` (primary) | Hash key, construct, stream/checksum `k` nodes, destroy | +| `seeded` | Same, but with prehashed seeds; no algorithm-specific cache | +| `setup` | Hash alone, or construct/drop an iterator from a seed | +| `stream_only` | Consume separately prepared iterators with `iter_batched_ref`; construction/destruction excluded for both | +| `collect` | Prehashed construction plus allocate/fill/drop the same-capacity `Vec` | +| `rank_replay` | Prehashed construction and `.nth(k-1)`; both replay `k` successors to return one result | + +All width/key preparation, forward/inverse work, and baseline counter +allocation are charged in the primary comparison. Inputs use `black_box` and +outputs are consumed. Collection uses `u64` elements for both, despite the +existing API's `u32` output. Streaming-only is a component experiment, not +the primary comparison; its prepared-state cache footprints differ. + +The full paired matrix is: + +```text +n = 1,2,3,4,6,7,8,9,14,15,16,17,30,31,32,33, + 254,255,256,257,1000,1022,1023,1024,1025, + 65534,65535,65536,65537,1000000,2^30-2,2^30-1,2^30 +k = valid values from 1,2,3,8,16; also floor(n/4) and n when n<=1024 +``` + +Zeros and duplicates are removed. This covers powers of two and sentinel +boundaries `n+1=2^b`. There are **177 `(n,k)` pairs** in each fresh/seeded +group. Component groups use `n=17,257,1000,65537,2^30`. The completed run +contains **905 estimates**. All paired timings are in the existing iterator's +supported range; larger new domains are not claimed as performance wins. + +## Fresh-query results + +Mean **ns/query [bootstrap 95% confidence interval]**. Ratio is new/existing: +above one is a regression. + +| n | k | Existing layered | New sentinel | Ratio | | ---: | ---: | ---: | ---: | ---: | -| 1 | 1 | 58.04 [57.76, 58.27] | 7.38 [7.08, 7.82] | 0.13x | -| 8 | 1 | 51.31 [50.97, 51.63] | 114.80 [114.46, 115.16] | 2.24x | -| 8 | 8 | 227.41 [223.70, 230.99] | 1,005.85 [1,000.85, 1,014.19] | 4.42x | -| 17 | 3 | 110.06 [108.94, 111.07] | 622.33 [619.42, 625.40] | 5.65x | -| 33 | 33 | 600.19 [593.91, 604.57] | 7,988.45 [7,913.42, 8,104.45] | 13.31x | -| 255 | 3 | 38.10 [37.82, 38.39] | 142.31 [141.60, 143.18] | 3.74x | -| 256 | 3 | 39.69 [39.27, 40.12] | 143.02 [142.31, 143.75] | 3.60x | -| 257 | 3 | 63.41 [62.81, 63.97] | 305.48 [302.11, 310.19] | 4.82x | -| 1,000 | 1 | 22.39 [21.38, 24.14] | 49.53 [47.21, 52.43] | 2.21x | -| 1,000 | 3 | 35.58 [35.43, 35.72] | 104.90 [102.88, 107.36] | 2.95x | -| 1,000 | 16 | 103.18 [102.78, 103.58] | 536.52 [519.93, 561.57] | 5.20x | -| 1,000 | 250 | 2,042.64 [2,013.97, 2,070.34] | 14,996.65 [14,959.80, 15,025.20] | 7.34x | -| 1,000 | 1,000 | 9,735.29 [9,668.93, 9,808.83] | 85,990.56 [84,682.97, 87,389.54] | 8.83x | -| 1,023 | 3 | 36.29 [35.55, 37.30] | 106.26 [99.49, 118.27] | 2.93x | -| 1,024 | 3 | 35.79 [35.60, 35.99] | 97.85 [96.85, 98.99] | 2.73x | -| 1,025 | 3 | 64.33 [64.06, 64.61] | 266.90 [265.96, 268.03] | 4.15x | -| 65,535 | 3 | 35.19 [35.02, 35.39] | 88.23 [87.26, 89.16] | 2.51x | -| 65,536 | 3 | 36.40 [36.06, 36.77] | 90.23 [89.02, 91.43] | 2.48x | -| 65,537 | 3 | 66.23 [65.47, 66.87] | 268.53 [265.08, 272.36] | 4.05x | -| 1,000,000 | 3 | 36.14 [35.67, 36.65] | 96.65 [95.81, 97.70] | 2.67x | -| 2^30 - 1 | 3 | 38.49 [38.01, 38.91] | 87.74 [87.45, 88.21] | 2.28x | -| 2^30 | 3 | 38.86 [38.60, 39.12] | 89.96 [89.23, 90.68] | 2.31x | - -For example, `n=1000,k=3` is 11.86 versus 34.97 **ns/replica**; -`k=1000` is 9.74 versus 85.99 ns/replica. A full permutation has more -upper-half input slots, which need inverse walks; a short initial prefix often -avoids that work. Consequently mean cost per replica is not constant across -all `k`, even though the ideal-model expectation is bounded uniformly. - -Across the entire primary matrix the measured ratio ranges from 0.13x -(`n=1,k=1`, a degenerate constant-output case) to 13.31x -(`n=33,k=33`). The new implementation was slower in all 130 nontrivial cases. -The existing code's inexpensive four-round upper-layer -primitive and shared streaming counters generally beat repeated evaluation of -an 8/16/24-round, fully mixed forward/inverse primitive. Avoiding a small state -allocation does not make up for that extra arithmetic and traversal. -This compares the actual implementations, not algorithms normalized to an -identical primitive: both traversal strategy and mixing cost contribute. -All timings in this original comparison use the final strengthened schedule, not the rejected -eight-round version discussed below. - -## Separating setup, streaming and direct-slot costs - -Mean ns/query [95% confidence interval]. All rows below are prehashed; in the -`slot` rows, `k` means "query slot `k-1`", **not** consume `k` replicas. - -| Mode | n | k | Existing layered | New virtual | -| --- | ---: | ---: | ---: | ---: | -| Constructor + drop | 1,000 | - | 9.73 [9.62, 9.83] | 1.11 [1.10, 1.12] | -| Constructor + drop | 2^30 | - | 9.77 [9.72, 9.84] | 1.08 [1.08, 1.09] | -| Seeded stream query | 1,000 | 3 | 32.45 [32.21, 32.66] | 103.99 [102.70, 105.32] | -| Seeded stream query | 1,000 | 1,000 | 9,737.78 [9,680.29, 9,789.97] | 85,750.68 [85,646.50, 85,868.42] | -| Stream only | 1,000 | 3 | 14.62 [14.45, 14.82] | 91.78 [91.32, 92.34] | -| Stream only | 1,000 | 1,000 | 9,681.61 [9,618.72, 9,750.56] | 86,659.57 [86,278.25, 87,113.38] | -| Collect | 1,000 | 3 | 43.94 [43.02, 45.48] | 113.49 [112.83, 114.09] | -| Collect | 1,000 | 1,000 | 9,895.86 [9,786.30, 10,060.80] | 90,027.62 [88,127.94, 92,190.79] | -| Slot | 1,000 | 3 | 29.93 [29.79, 30.07] | 30.83 [30.69, 30.97] | -| Slot | 1,000 | 16 | 94.70 [93.68, 95.85] | 28.88 [28.76, 28.98] | -| Slot | 1,000 | 250 | 2,088.37 [2,042.18, 2,135.62] | 50.60 [50.48, 50.77] | -| Slot | 1,000 | 1,000 | 9,498.73 [9,419.35, 9,599.28] | 72.18 [72.02, 72.36] | -| Slot | 65,537 | 3 | 60.03 [59.29, 60.85] | 79.55 [78.88, 80.39] | - -Hashing a `u64` alone measured 7.48 [7.46, 7.50] ns. Component costs should -not be added or subtracted as exact identities: compiler optimization, -instruction overlap and cache state differ between the groups. -For `n=1000,k=3`, the seeded-query sample standard deviations were 0.54 ns -(existing) and 2.95 ns (new); collecting 1,000 replicas had standard deviations -315.22 ns and 4,646.27 ns respectively. The original run had 721 benchmark -estimates. The current three-method runner emits 1,081 estimates, including -standard deviations, with the same CSV script. - -Direct access becomes useful for later slots: querying slot 999 at `n=1000` -was about 132x faster than replaying the old iterator. Early slots need not -benefit, as the slot-2 rows show. This does not change the sequential-enumeration -regression, and the saved allocation is not being silently excluded from the -primary comparison. - -## State and allocations - -On this target, `size_of::()` is 40 bytes plus one -allocation for `4 * max(1, ceil(log2(n)/2))` bytes of counters: 20 bytes at -`n=1000`, 36 at `n=65537`, and 60 at `n=2^30`, excluding allocator metadata. -`size_of::()` is 24 bytes and it has no heap allocation. -It uses an additional constant-sized parameter block and scalars on the stack -during evaluation, not a recursion stack or a per-`n` cache. - -These allocation counts follow the source, rather than a custom allocator -installed in the timed benchmark. The diagnostic example prints the actual -struct sizes. The collection benchmark adds one capacity-`k` `Vec` -allocation to **both** algorithms, so it has two total allocations for the -existing iterator and one for the new one. Optional output storage is `8k` -bytes in both cases. - -## Distribution diagnostics and the rejected first version - -The deterministic example evaluates 200,000 hashed seeds per method for -`n = 3,4,5,7,8,9,16,17,32,64`. It reports each of the first up to eight slot -marginals, ordered pairs `(slot 0, slot 1)` and `(slot 0, last sampled slot)`, -unordered first-three subsets, and consecutive-key primary pairs. Expected -cells exclude repeated nodes for within-key pairs and include them for -cross-key pairs. All nontrivial cells are printed even when sparse (for -example, the `n=64` choose-three diagnostic has only 4.80 expected per cell). -These are exploratory diagnostics, not random p-value CI gates or a proof of -independence; the histograms overlap and are not independent tests. - -The primary corpus hashes `0x7065726d75746531 XOR i`, `i=0..199999`, with -`DefaultHasher`. A disjoint, deterministic held-out corpus uses -`0x686f6c646f757431 XOR i`. The held-out corpus was first evaluated **after** -fixing the stronger schedule; no additional tuning followed its results. -It is independent input data for diagnosis, not a claim that a deterministic -hash function supplies mathematically independent random variables. - -The initial **eight-round-at-every-width** implementation was rejected: -at `n=8`, its ordered `(0,1)` pair statistic was 1753.19 on 55 degrees of -freedom even though its primary marginal statistic was 4.24 on 7 degrees of -freedom. Clean marginals were insufficient. The final schedule uses 24 rounds -at widths 2--4, 16 at 5--7 and 8 at 8--64, fixed solely by width. This -strengthens mixing instead of reducing rounds to improve performance. - -| Metric, n=8 | Existing primary / held-out chi-square | Final virtual primary / held-out chi-square | Degrees of freedom | -| --- | ---: | ---: | ---: | -| Primary marginal | 4.79 / 4.66 | 8.63 / 1.22 | 7 | -| Slot 7 marginal | 11.50 / 5.73 | 6.44 / 1.26 | 7 | -| Ordered slots (0,1) | 65.97 / 43.73 | 54.15 / 44.41 | 55 | -| Ordered slots (0,7) | 66.92 / 44.74 | 60.47 / 45.43 | 55 | -| Unordered first-three subset | 53.47 / 59.83 | 47.66 / 40.41 | 55 | -| Consecutive-key primaries | 50.26 / 61.98 | 68.31 / 44.11 | 63 | - -At `n=32`, the final virtual first-three subset statistics are 4863.03 and -4873.25 on 4959 degrees of freedom, versus 5068.47 and 5024.53 for the existing -implementation. Its ordered `(0,1)` statistics are 1033.88 and 920.17 on 991 -degrees of freedom, versus 1032.74 and 985.50 for the existing implementation. - -No similarly large deviations appeared in the final virtual family's measured -marginals, ordered pairs, subsets or consecutive-key pairs on either corpus. -That observation does not establish exact uniformity, bound unseen-key bias, -or validate cryptographic properties. - -The **unchanged existing implementation** also has detectable deviations in -these broader diagnostics: at `n=9`, slot 4 has chi-square 253.52 on 8 degrees -of freedom in the primary corpus and 167.39 in the held-out corpus; slots 3 -and 5 also deviate. This is a limitation of the measured baseline, not a -reason to alter its behavior in this PR or to declare the new algorithm -uniform. Timing and statistical evidence answer different questions. - -## Untimed operation counts and tails - -The ignored diagnostic test wraps the **same evaluator and primitive** with -forward/inverse counters, outside any timed benchmark. For each `(n,slot)`, -it uses 10,000 deterministic SplitMix64-mixed seeds from integers `0..9999` -(the test source fixes the exact mixer and offset). Percentiles are nearest- -rank percentiles of total forward plus inverse calls; one call can contain -8, 16 or 24 Feistel rounds depending on width. - -| n | slot | Mean forward | Mean inverse | Mean visited levels | p50 calls | p99 calls | Observed max | -| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | -| 256 | 0 | 1.9808 | 0 | 1.9808 | 1 | 7 | 8 | -| 257 | 0 | 3.9806 | 1.0051 | 2.9715 | 4 | 16 | 28 | -| 257 | 256 | 3.9660 | 3.9270 | 2.9786 | 7 | 22 | 41 | -| 65,536 | 0 | 1.9922 | 0 | 1.9922 | 1 | 7 | 15 | -| 65,537 | 0 | 3.9822 | 0.9900 | 2.9922 | 4 | 16 | 30 | -| 65,537 | 65,536 | 4.0104 | 4.0244 | 3.0065 | 7 | 23 | 36 | -| 2^30 | 2^29 | 1.9907 | 1.4876 | 1.9907 | 1 | 16 | 49 | -| 2^63 + 1 | 2^63 | 3.9795 | 4.0169 | 2.9878 | 7 | 23 | 40 | -| u64::MAX | 0 | 2.0194 | 0 | 2.0194 | 2 | 7 | 16 | - -The last two rows are **unpaired, wider-domain diagnostics**, not performance -comparisons with the old API. Near a half-full top domain, traversal work -increases substantially; constant expected work does not mean flat latency. -The 8.0348-call sample mean at `n=65537,slot=65536` is not a contradiction of -the ideal-model bound: it is a finite sample of a finite-key approximation, -not the ideal ensemble expectation. - -Observed maxima are not guarantees. A deterministic adversarial-oracle unit -test uses ascending cycles and needs over 4,096 calls for `n=2049,slot=2048`; -the evaluator still returns the correct result without a retry cap. Long -cycles and linear-in-domain worst cases remain possible. Neither the -diagnostics nor the timing confidence intervals establish a latency bound. - -## Matched-network follow-up - -At the user's request, `BalancedVirtualPermutation` now uses exactly the -existing `layer_apply` network: same seed, mixer, balanced halves, round -counts, rotation and Weyl key schedule. The cycle construction descends by -two bits, as the existing iterator does. A new inverse shares the same round -function and reverses that exact key schedule. It does not introduce a -different mixer, extra seed hash, or independently tuned round count. - -The same machine, compiler, flags, key corpus, 131-case matrix and Criterion -settings were used again, sequentially in `layered`, `virtual`, `balanced` -order per case. There are 1,081 estimates across all groups. All three -implementations are rerun; comparisons below use this run, not subtraction of -timings from the earlier run. The shared-host and fixed-corpus limitations -still apply. State is 24 bytes with no allocation for either cycle iterator, -versus 40 bytes plus the heap counters for the existing iterator. - -### Fresh queries: matching the primitive removes much of the overhead - -Mean ns/query [bootstrap 95% interval]. The final ratio is -**matched/existing**; below one favors the matched variant. - -| n | k | Existing | Stronger one-bit | Matched two-bit | Matched/existing | -| ---: | ---: | ---: | ---: | ---: | ---: | -| 8 | 1 | 45.27 [44.67, 45.87] | 110.60 [110.18, 111.13] | 61.42 [60.35, 62.62] | 1.36x | -| 8 | 8 | 205.76 [203.68, 207.92] | 1,030.44 [1,021.06, 1,041.23] | 608.10 [592.46, 623.97] | 2.96x | -| 16 | 3 | 57.51 [57.16, 57.82] | 334.54 [333.57, 335.44] | 67.91 [66.89, 69.14] | 1.18x | -| 17 | 3 | 105.10 [104.22, 105.84] | 619.91 [618.09, 621.79] | 261.59 [258.54, 266.26] | 2.49x | -| 33 | 33 | 637.60 [604.96, 684.35] | 8,604.99 [8,085.82, 9,388.97] | 2,155.89 [2,115.56, 2,192.34] | 3.38x | -| 255 | 3 | 35.09 [34.77, 35.39] | 140.78 [140.32, 141.18] | 30.76 [30.58, 30.97] | 0.88x | -| 256 | 3 | 35.99 [35.64, 36.29] | 139.85 [139.28, 140.62] | 31.11 [30.80, 31.46] | 0.86x | -| 257 | 3 | 61.67 [61.43, 61.91] | 304.03 [302.58, 305.93] | 198.41 [195.88, 201.23] | 3.22x | -| 257 | 8 | 155.53 [151.73, 159.47] | 872.01 [846.66, 902.67] | 536.47 [521.46, 564.32] | 3.45x | -| 1,000 | 1 | 19.75 [19.70, 19.81] | 43.94 [43.47, 44.47] | 16.26 [16.03, 16.48] | 0.82x | -| 1,000 | 3 | 32.83 [32.63, 33.02] | 101.72 [101.21, 102.22] | 27.72 [27.58, 27.86] | 0.84x | -| 1,000 | 16 | 98.73 [98.11, 99.45] | 539.84 [504.74, 605.05] | 137.11 [135.35, 138.99] | 1.39x | -| 1,000 | 250 | 2,059.36 [2,047.90, 2,070.14] | 14,914.99 [14,779.03, 15,043.08] | 3,374.48 [3,336.90, 3,412.61] | 1.64x | -| 1,000 | 1,000 | 9,808.45 [9,658.26, 10,006.25] | 83,713.94 [83,369.99, 84,114.21] | 22,289.60 [22,158.85, 22,422.35] | 2.27x | -| 1,024 | 3 | 33.67 [33.45, 33.88] | 98.44 [97.52, 99.27] | 29.54 [28.99, 30.18] | 0.88x | -| 1,025 | 3 | 65.79 [65.44, 66.12] | 267.49 [266.66, 268.47] | 182.40 [180.91, 184.61] | 2.77x | -| 65,536 | 3 | 33.12 [32.94, 33.28] | 83.63 [83.24, 83.98] | 28.43 [28.16, 28.67] | 0.86x | -| 65,537 | 3 | 65.87 [65.58, 66.16] | 262.49 [262.17, 262.80] | 171.81 [169.97, 174.83] | 2.61x | -| 1,000,000 | 3 | 35.39 [34.87, 36.04] | 95.33 [94.27, 96.33] | 27.43 [27.22, 27.61] | 0.77x | -| 2^30 | 3 | 34.24 [33.94, 34.51] | 90.35 [89.09, 91.52] | 26.34 [26.18, 26.47] | 0.77x | - -At `n=1000,k=3`, matching the network and stride reduces the cycle -construction from 101.72 to 27.72 ns/query (3.67x faster), making it about -16% faster than the existing iterator on this workload. That is 9.24 -ns/replica versus 10.94 existing and 33.91 stronger-one-bit. For a full -1,000-node permutation it is still 2.27x slower than existing (22.29 versus -9.81 ns/replica), although much faster than the stronger variant's 83.71. - -The matched mean is below the existing mean in 31 of 131 cases, including the -degenerate `n=1` case; small differences are not blanket claims of significance. -Its largest observed mean ratio is 3.45x at `n=257,k=8`. Just above powers of -four, the top domain is nearly three-quarters inactive, and cycle walking plus -inverse traversal is still expensive. This experiment changes primitive -**and stride together**; it does not isolate the machine cost of odd halves -alone. It demonstrates that the earlier slowdown was not an unavoidable cost -of cycle consistency, but it does not show that the cycle method always wins. - -The component measurements help distinguish allocation from streaming work. -Below are mean ns/query from the same run, with prehashed seeds; slot queries -return **one** value, while the other `k` rows consume or collect `k` values. - -| Workload, n=1000 | Existing | Stronger one-bit | Matched two-bit | +| 1 | 1 | 58.79 [58.58, 59.01] | 14.80 [14.60, 14.95] | 0.25x | +| 7 | 3 | 103.16 [102.76, 103.49] | 864.91 [845.97, 886.02] | 8.38x | +| 8 | 3 | 93.23 [92.57, 94.10] | 2,046.05 [2,036.71, 2,057.79] | 21.95x | +| 8 | 8 | 219.11 [216.88, 221.17] | 6,167.64 [6,071.71, 6,264.64] | 28.15x | +| 17 | 3 | 112.01 [111.87, 112.13] | 1,536.05 [1,533.94, 1,538.30] | 13.71x | +| 33 | 33 | 636.62 [632.07, 641.34] | 23,294.31 [22,161.71, 24,440.97] | 36.59x | +| 254 | 3 | 37.48 [37.27, 37.65] | 468.70 [456.92, 483.16] | 12.51x | +| 255 | 3 | 38.07 [37.72, 38.40] | 494.86 [483.43, 505.74] | 13.00x | +| 256 | 3 | 38.74 [38.62, 38.86] | 956.42 [929.65, 988.09] | 24.69x | +| 257 | 3 | 73.57 [65.12, 88.32] | 903.16 [895.14, 912.31] | 12.28x | +| 1,000 | 1 | 21.30 [21.25, 21.34] | 103.96 [102.76, 105.34] | 4.88x | +| 1,000 | 2 | 28.00 [27.96, 28.04] | 218.57 [215.73, 221.40] | 7.80x | +| 1,000 | 3 | 36.48 [36.11, 36.84] | 351.25 [349.34, 353.43] | 9.63x | +| 1,000 | 8 | 59.65 [59.41, 59.91] | 1,079.05 [1,040.73, 1,151.32] | 18.09x | +| 1,000 | 16 | 101.34 [101.04, 101.63] | 2,393.84 [2,338.73, 2,453.82] | 23.62x | +| 1,000 | 250 | 2,182.76 [2,164.34, 2,198.22] | 44,957.17 [44,920.88, 44,995.31] | 20.60x | +| 1,000 | 1,000 | 10,761.56 [10,554.57, 10,972.06] | 195,818.68 [191,456.60, 200,639.88] | 18.20x | +| 1,022 | 3 | 35.23 [34.97, 35.48] | 343.55 [341.82, 345.57] | 9.75x | +| 1,023 | 3 | 35.90 [35.64, 36.16] | 347.39 [345.30, 349.88] | 9.68x | +| 1,024 | 3 | 36.13 [35.93, 36.30] | 759.81 [758.32, 761.42] | 21.03x | +| 1,024 | 16 | 106.36 [105.76, 107.03] | 5,685.25 [5,593.80, 5,766.20] | 53.45x | +| 1,025 | 3 | 69.08 [68.58, 69.59] | 759.49 [756.87, 762.46] | 10.99x | +| 65,535 | 3 | 38.91 [35.65, 44.28] | 328.64 [318.92, 338.79] | 8.44x | +| 65,536 | 3 | 39.12 [38.49, 39.68] | 747.60 [742.93, 751.59] | 19.11x | +| 65,537 | 3 | 67.87 [66.62, 69.11] | 749.19 [739.39, 760.31] | 11.04x | +| 1,000,000 | 3 | 35.77 [35.24, 36.34] | 328.07 [320.83, 336.15] | 9.17x | +| 2^30 - 2 | 3 | 38.40 [38.06, 38.75] | 305.69 [302.15, 309.49] | 7.96x | +| 2^30 - 1 | 3 | 40.79 [40.24, 41.33] | 300.83 [299.63, 302.02] | 7.37x | +| 2^30 | 3 | 39.51 [39.32, 39.71] | 727.97 [726.74, 729.28] | 18.42x | + +The new implementation is slower in all **176 nontrivial fresh cases**. +The only win is the degenerate `n=1` constant order. Nontrivial ratios range +from 3.51x to 53.45x. At `n=1000,k=3`, the figures are 12.16 versus +117.09 **ns/output**; at `k=1000`, 10.76 versus 195.82 ns/output. +Sample standard deviations are 0.88/4.88 ns for the three-output query and +495.29/10,391.66 ns for full enumeration. + +Every conjugated-cycle step evaluates both Q and its inverse; normalized +lifts also retrace chains. The new Q uses more expensive mixing and more +rounds than the existing iterator's upper layers. This is not an attribution +solely to odd versus even Feistel widths. The sentinel shifts padding +boundaries: `n=1023` has a full internal 1024-label domain, whereas `n=1024` +requires a nearly half-empty 2048-label domain. Operation counts below expose +that cost. Avoiding counter allocation does not offset these extra operations. + +## Setup, streaming, collection and rank replay + +Mean ns/query [95% interval], all prehashed; `n=1000`. +For rank replay, `k` denotes fetching rank `k-1`, not returning `k` outputs. + +| Mode | k | Existing layered | New sentinel | | --- | ---: | ---: | ---: | -| Constructor + drop | 9.71 | 1.09 | 1.17 | -| Seeded query, k=3 | 31.25 | 103.66 | 27.08 | -| Stream only, k=3 | 14.45 | 89.51 | 20.95 | -| Stream only, k=1000 | 9,541.23 | 86,465.65 | 22,863.95 | -| Collect, k=3 | 42.90 | 112.43 | 36.66 | -| Collect, k=1000 | 9,727.46 | 87,758.08 | 22,936.49 | -| Query slot 2 | 30.03 | 31.54 | 7.89 | -| Query slot 999 | 9,604.66 | 77.43 | 12.69 | - -For `k=3` stream-only, the existing/matched 95% intervals are -[14.25, 14.66] / [20.80, 21.11] ns, and sample standard deviations -0.47 / 0.39 ns. For the seeded complete query, the intervals are -[30.98, 31.47] / [26.92, 27.29] ns. The fresh-query win is therefore consistent -with avoiding allocation, **not** evidence that the cycle traversal itself is -cheaper. These components still cannot be added as exact identities because -their compilation and cache conditions differ. - -The matched direct slot-999 query has interval [12.53, 12.88] ns and sample -standard deviation 0.42 ns. The existing iterator must replay to reach that -slot; this is an API/workload advantage, not a claim of comparable sequential -enumeration speed. - -### Statistical quality is not equivalent - -Both deterministic 200,000-key corpora were rerun without tuning the matched -network. The existing and stronger variants' reported diagnostic rows are -unchanged from the prior run. The matched variant has repeatable deviations: - -| Metric | Degrees of freedom | Matched primary chi-square | Matched held-out chi-square | +| Constructor + drop | - | 9.59 [9.46, 9.73] | 1.58 [1.57, 1.59] | +| Seeded stream query | 3 | 31.60 [31.27, 31.96] | 336.88 [335.32, 339.24] | +| Seeded stream query | 1,000 | 9,765.81 [9,702.08, 9,834.64] | 186,574.59 [185,570.04, 187,827.35] | +| Stream only | 3 | 14.67 [14.47, 14.87] | 347.11 [338.09, 357.54] | +| Stream only | 1,000 | 9,595.61 [9,574.26, 9,622.33] | 215,975.98 [208,624.79, 224,187.29] | +| Collect | 3 | 42.41 [42.21, 42.60] | 372.48 [371.57, 373.44] | +| Collect | 1,000 | 9,965.63 [9,906.14, 10,036.13] | 206,992.98 [198,395.13, 216,335.19] | +| Rank replay | 3 | 33.34 [32.45, 34.14] | 363.04 [355.80, 371.01] | +| Rank replay | 1,000 | 9,779.07 [9,746.28, 9,820.64] | 187,699.87 [186,407.45, 189,040.15] | + +Hashing a `u64` alone measured 5.41 [4.84, 5.98] ns. Component means are +not additive identities: optimizer behavior, instruction overlap, prepared +state and host noise differ. In particular, component results do not recover +a hidden direct-rank advantage. + +### State and allocation + +On this target the baseline struct is 40 bytes plus one allocation for +`4 * max(1, ceil(log2(n)/2))` bytes of counters: 20 bytes at `n=1000`, +36 at `n=65537`, 60 at `n=2^30`, excluding allocator metadata. +The sentinel iterator is **32 bytes with no heap allocation**. Its temporary +width parameters and scalars are constant-sized, without recursive stack, +prebuilt ring, permutation table or duplicate set. + +Allocation counts follow source inspection, not a custom allocator in the +timed region; the diagnostic prints actual struct sizes. Collection adds one +capacity-`k` `Vec` allocation (8k bytes) to both methods: two total +allocations for the existing iterator and one for the new iterator. + +## Fresh randomness diagnostics + +The example evaluates **3,760,000 complete orders per method per corpus**: +200,000 keys at each `n<=33`, 50,000 at `62,63,64,65`, and 20,000 at +`126,127,128,129,254,255,256,257`. The smaller sizes are +`1,2,3,4,5,6,7,8,9,14,15,16,17,30,31,32,33`. +They include small/odd-width domains and both real-node and sentinel boundaries. + +Primary keys hash `0x7065726d75746531 XOR i`; a disjoint held-out key corpus +hashes `0x686f6c646f757431 XOR i`, using `DefaultHasher`. +The fixed 24/16/8-round Q schedule was not changed or tuned to either corpus +for this construction. Held-out means separate input data, not mathematical +independence supplied by a deterministic hash. + +Metrics cover ranks `0,1,2,n/2,n-2,n-1` where valid, first/middle adjacent +pairs, first/middle and first/last distant pairs, first-three unordered subsets, +all full orders through `n=7`, consecutive application-key primary pairs, +and primaries for `seed` versus `seed XOR (1<<63)`. Duplicate rank choices +are removed. + +Pairs are exact through `n=65`; larger domains use eight contiguous buckets. +For within-key bucket pairs the expected weight is +`size[a]*(size[b] - (a==b))/(n*(n-1))`, not a uniform 64-cell assumption. +Cross-key pairs use `size[a]*size[b]/n^2`. Repeated exact nodes are impossible +within-key and allowed cross-key. Triple histograms are skipped when expected +cell counts would be below ten (all tested sizes above 33); full-order +histograms stop at seven. The actual minimum expected cell count is 11.834. + +The output contains 694 rows across both algorithms per corpus, including +observations, degrees of freedom, minimum expected count, chi-square, maximum +relative cell deviation and empirical total variation (TV). These are +overlapping exploratory diagnostics, **not p-value CI gates**. Consecutive-key +pairs overlap in their keys; rows and nearby sizes are not independent tests. +Bucketed tests can miss correlations inside a bucket. + +### Representative chi-square results + +Values are primary / held-out. All rows use 200,000 observations except +`n=64,65` (50,000) and `n=256` (20,000). + +| n, metric (zero-based ranks) | df | Existing layered | New sentinel | | --- | ---: | ---: | ---: | -| n=5, slot 4 | 4 | 82.80 | 84.94 | -| n=5, ordered slots (0,4) | 19 | 98.15 | 98.63 | -| n=8, primary marginal | 7 | 6.21 | 7.26 | -| n=8, ordered slots (0,1) | 55 | 112.66 | 131.31 | -| n=8, first-three subset | 55 | 133.99 | 146.63 | -| n=9, first-three subset | 83 | 148.11 | 141.19 | - -For the `n=5,slot=4` marginal, the most frequent node occurs 41,613 and 41,612 -times against 40,000 expected in each corpus, about a 4% excess. At `n=8`, -the stronger variant's first-three subset statistics remain 47.66 / 40.41, -and the existing iterator's are 53.47 / 59.83, versus the matched variant's -133.99 / 146.63 (all 55 degrees of freedom). - -The same primitive need not have the same statistical behavior after list -projection versus cycle projection. These results do not distinguish -within-width weaknesses from cross-width seed correlations, nor do they -prove exact uniformity for any method. They do show that replacing the -stronger network is **not statistically neutral**. The matched variant is -retained as an explicitly experimental comparison, not silently substituted -for `VirtualPermutation` or promoted as meeting the ideal randomness model. - -### Traversal counts explain remaining boundary costs - -The operation-count diagnostic now prints both cycle variants, using the same -10,000 seeds per `(n,slot)` as before. Means count forward plus inverse calls, -not Feistel rounds or nanoseconds: - -| n | slot | Stronger mean calls | Matched mean calls | Stronger p99 | Matched p99 | Matched observed max | -| ---: | ---: | ---: | ---: | ---: | ---: | ---: | -| 256 | 0 | 1.9808 | 1.3247 | 7 | 4 | 4 | -| 257 | 0 | 4.9857 | 8.2873 | 16 | 34 | 59 | -| 257 | 256 | 7.8930 | 13.2029 | 22 | 41 | 134 | -| 65,536 | 0 | 1.9922 | 1.3330 | 7 | 4 | 8 | -| 65,537 | 0 | 4.9722 | 8.3462 | 16 | 34 | 68 | -| 65,537 | 65,536 | 8.0348 | 13.4067 | 23 | 41 | 73 | - -At full powers of four, primary lookup uses fewer levels; a regression test -also confirms that the matched and existing algorithms give the **same -primary value** there. Just above those boundaries, however, the two-bit -construction must traverse a much larger inactive upper region and then -often retrace it. Its cheaper primitive can still yield faster timings than -the stronger variant despite more primitive calls. These measured tails are -not worst-case guarantees, and the one-bit variant's ideal less-than-eight- -call bound does not apply to the two-bit variant. +| 7, full order | 5,039 | 5,191.353 / 5,431.710 | 5,188.278 / 5,135.056 | +| 7, first-three subset | 34 | 77.207 / 81.965 | 37.004 / 45.729 | +| 8, first rank | 7 | 4.790 / 4.656 | 4.560 / 9.774 | +| 8, last rank | 7 | 11.501 / 5.734 | 6.189 / 13.977 | +| 8, ordered ranks (0,1) | 55 | 65.968 / 43.725 | 55.089 / 51.515 | +| 8, ordered ranks (4,5) | 55 | 57.991 / 53.807 | 42.932 / 48.067 | +| 8, ordered ranks (0,7) | 55 | 66.922 / 44.737 | 58.735 / 60.360 | +| 8, first-three subset | 55 | 53.474 / 59.828 | 69.037 / 50.819 | +| 8, related-seed primaries | 63 | 86.673 / 51.715 | 83.164 / 76.575 | +| 9, middle rank 4 | 8 | 253.520 / 167.394 | 4.610 / 0.708 | +| 9, ordered ranks (4,5) | 71 | 387.530 / 305.222 | 52.348 / 76.608 | +| 33, middle rank 16 | 32 | 127.629 / 100.534 | 27.809 / 29.831 | +| 33, first-three subset | 5,455 | 5,553.436 / 5,455.392 | 5,429.912 / 5,576.133 | +| 64, ordered ranks (0,1) | 4,031 | 3,967.352 / 3,973.320 | 3,828.813 / 4,140.406 | +| 65, ordered ranks (0,64) | 4,159 | 4,096.640 / 4,155.046 | 4,042.726 / 4,180.339 | +| 256, ordered ranks (0,128), eight buckets | 63 | 500.392 / 488.941 | 74.926 / 77.442 | + +The unchanged baseline has repeatable deviations, especially middle ranks +and distant pairs. At `n=9,rank=4`, its maximum relative cell deviations are +9.91%/7.94%, versus sentinel's 0.90%/0.25%; empirical TV is +1.10%/0.92% versus 0.20%/0.08%. At `n=256`, the distant bucketed pair has +maximum relative deviations 53.96%/53.64% versus 17.76%/14.75%, and TV +5.71%/5.59% versus 2.36%/2.54%. These observations do not justify changing +the user's mapping in this PR. + +Not every new statistic is small. Sentinel's first rank at `n=4` has +chi-square 2.852/13.863 on df=3, maximum relative deviation 0.56%/1.22%. +Its last rank at `n=30` has 55.466/36.187 on df=29, and consecutive-key +primaries at `n=15` have 297.327/203.974 on df=224. Isolated fluctuations +must be interpreted alongside hundreds of correlated checks, not optimized +away by changing the mixer after viewing results. + +TV and maximum cell error include sampling noise and are not corrected +estimates of true family bias: even the new full-order `n=7` histograms have +TV 6.46%/6.39% with only 39.68 expected observations per bin. Diagnostics +cannot prove independence, exact uniformity, unseen-key bounds or +cryptographic security. The new family remains experimental. + +## Sequential operation counts and tails + +The ignored test instruments the actual evaluator **outside timed code**. +It uses 10,000 deterministic mixed seeds per `(n,k)`, 56 cases. Counts are +for whole sequential prefixes, not independent input slots. +One P or P_inverse operation costs two ordinary Q/Q_inverse evaluations; +each Q uses 8, 16 or 24 rounds according to width (width one is XOR). +Percentiles below are nearest-rank percentiles of total **P + P_inverse** +calls per query. `max step` is the largest individual successor evaluation. + +| n | k | Mean P forward | Mean P inverse | Mean Q calls/output | p99 query calls | Max query calls | Max step | +| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| 1 | 1 | 1.0000 | 0.0000 | 2.0000 | 1 | 1 | 1 | +| 8 | 1 | 3.1700 | 0.6973 | 7.7346 | 12 | 18 | 18 | +| 8 | 8 | 25.2413 | 11.0429 | 9.0710 | 40 | 40 | 20 | +| 255 | 3 | 5.9065 | 0.8688 | 4.5169 | 14 | 20 | 10 | +| 256 | 3 | 11.8445 | 3.8270 | 10.4477 | 32 | 49 | 35 | +| 256 | 256 | 1,012.0098 | 502.0404 | 11.8285 | 1,522 | 1,523 | 49 | +| 1,023 | 3 | 5.9834 | 0.8494 | 4.5552 | 14 | 20 | 12 | +| 1,024 | 3 | 11.9965 | 3.8663 | 10.5752 | 32 | 47 | 31 | +| 65,535 | 3 | 5.9705 | 0.8476 | 4.5454 | 15 | 23 | 15 | +| 65,536 | 3 | 11.9667 | 3.8438 | 10.5403 | 32 | 48 | 36 | +| 2^30 | 3 | 12.0301 | 3.9108 | 10.6273 | 33 | 50 | 36 | +| 2^63 - 1 | 16 | 32.0460 | 11.6256 | 5.4589 | 60 | 75 | 23 | +| u64::MAX - 1 | 16 | 32.1024 | 11.6484 | 5.4688 | 60 | 72 | 25 | + +Observed mean ordinary-PRP calls/output range from 2 to 11.8285. This is +compatible with, but does not prove, the conservative ideal-model expected +bound below 16 described in the design note. The mean visited levels for +three outputs are 5.9834 at `n=1023` versus 8.9764 at `n=1024`; top padding +also adds forward walks and retracing. + +The deterministic invariant test checks that lower-level invocations equal +the selected lower-label subsequence and inverse calls never exceed forward +calls per level along every tested prefix. A constructed full-cycle oracle +still requires more than 4,096 primitive calls for just two outputs at +internal count 2,049. Thus observed tails, and the expected-prefix bound, +are **not a worst-case or adversarial-latency guarantee**. diff --git a/crates/consistent-choose-k/docs/virtual-permutation.md b/crates/consistent-choose-k/docs/virtual-permutation.md index ae5b797..b700dc8 100644 --- a/crates/consistent-choose-k/docs/virtual-permutation.md +++ b/crates/consistent-choose-k/docs/virtual-permutation.md @@ -1,40 +1,28 @@ -# VirtualPermutation: cycle-consistent replica slots +# VirtualPermutation: sentinel-rooted consistent order -`VirtualPermutation` is an additional experimental algorithm, not a replacement -for `ConsistentPermutation`. Both produce deterministic, distinct candidates, -and every `k`-selection is a prefix of the same per-key list. Their membership -semantics are **different**: +`VirtualPermutation` is an experimental alternative to the unchanged +`ConsistentPermutation`. Both produce distinct nodes, stable `k` prefixes, +and **complete survivor-list restriction**: deleting real node `n` from the +complete order for `n + 1` real nodes recovers the order for `n` real nodes. +Membership is consecutive IDs `0..n`; additions append IDs and removals remove +a suffix. Arbitrary holes, weights and physical-node remapping are out of scope. | Property | Existing `ConsistentPermutation` | New `VirtualPermutation` | | --- | --- | --- | -| Append one node | Insert it into the ranking; later replica slots can shift | At most one old replica slot changes, to the new node | -| Remove the last node | Remove it from the ranking; later slots can shift | Only a surviving slot that named the removed node changes | -| Survivor list order | Preserved | Not promised | -| Projection between sizes | Delete entries from the output list | Delete labels from the permutation's cycles | -| Direct slot lookup | Replay iterator through that slot | `replica_at(slot)`, without replay | -| Supported `n` | `1..=2^30` (`u32`) | `1..=u64::MAX` | -| Mutable state | Per-layer `Vec` counters | Three `u64` fields; no heap allocation | - -An additional **matched-network experiment**, `BalancedVirtualPermutation`, -uses the old iterator's exact Feistel network inside the new cycle-projection -construction. It supports `1..=2^30`, returns `u64` values, and also has 24-byte -allocation-free state. Its `new(n: u32, seed)`, `n`, `replica_at`, `nth` and -iterator operations have the cycle/slot semantics, not survivor-list semantics. -The [matched comparison](virtual-permutation-performance.md#matched-network-follow-up) -records both its performance and its repeatable small-domain statistical bias. - -For example, the permutation written as an output list `[2, 0, 1]` is the -cycle `0 -> 2 -> 1 -> 0`. Cycle-deleting node 2 produces `[1, 0]`, -**not** the survivor list `[0, 1]`. The changed old slot is 0, the slot that -named the removed node. This difference matters for failover and bounded-load -policies that depend on survivor priority; do not substitute the new API into -`ConsistentNodeMap` or existing consumers without considering their semantics. - -Membership is consecutive IDs `0..n`. Additions append IDs and removals remove -a suffix; arbitrary holes, weights and physical-node remapping are out of scope. -The one-slot bound is per single-node change, not for an entire suffix at once. - -## API and randomness model +| Membership changes | Insert/delete entries without reordering survivors | Same order-restriction contract | +| Replica ranks | May shift after insert/delete | May shift after insert/delete | +| Rank query | Replay the iterator | Replay the iterator | +| Supported real-node count | `1..=2^30`, `u32` | `1..=u64::MAX - 1`, `u64` | +| Iterator state | Per-layer heap-allocated counters | Four `u64` fields; no allocation | +| Construction | Interleaved per-layer Feistel streams | Consistent single-cycle successor traversal from a sentinel | + +This revision **replaces the earlier experimental slot-based variants**. +Those variants did not preserve survivor-list order. The experimental output +mapping has changed, the matched-network variant has been removed, and there +is no longer an absolute constant-time rank API. Existing consumers and the +user's `ConsistentPermutation` implementation/mapping are unchanged. + +## API, bounds and key hashing ```rust use consistent_choose_k::VirtualPermutation; @@ -43,236 +31,215 @@ use std::hash::{DefaultHasher, Hash, Hasher}; let mut hasher = DefaultHasher::new(); "object-key".hash(&mut hasher); let seed = hasher.finish(); -let permutation = VirtualPermutation::new(1_000, seed); -let third_replica = permutation.replica_at(2); -let replicas: Vec = permutation.take(3).collect(); -assert_eq!(third_replica, replicas[2]); +let replicas: Vec = VirtualPermutation::new(1000, seed).take(3).collect(); +let third = VirtualPermutation::new(1000, seed).nth(2); // replays three successors +assert_eq!(third, Some(replicas[2])); +let old: Vec<_> = VirtualPermutation::new(1000, seed).collect(); +let restricted: Vec<_> = VirtualPermutation::new(1001, seed) + .filter(|&node| node != 1000).collect(); +assert_eq!(old, restricted); ``` -Like the existing iterator, the constructor accepts an already well-mixed -64-bit seed, not an arbitrary application key. Use the same seed for all `n` -and `k`. `DefaultHasher` above is convenient for local examples and matches the -benchmark convention; Rust does not promise its mapping is stable across +The constructor accepts a well-mixed 64-bit seed. Keep it fixed across all +membership and replica counts. `DefaultHasher` matches the examples and +benchmark convention, but Rust does not promise a stable mapping across versions. Distributed deployments need a specified, versioned key hash and -identical algorithm versions on every participant. - -`new(0, seed)` and `replica_at(slot >= n)` panic, following the existing -constructor's assertion convention. `replica_at` is absolute, independent of -the cursor. The iterator is cloneable and fused, reports its remaining size, -and implements `nth` without replay. `.take(0)` is empty, `.take(n)` is the full -permutation, and `.take(k)` for `k > n` stops at exhaustion, as usual for Rust -iterators. There is no separate `k` constructor argument. - -Under **independent uniform ideal permutations for each key and bit width**, -the construction below gives a uniform permutation for every fixed `n`: -each ordered `k`-tuple of distinct nodes is equally likely, and different keys' -preference permutations are independent. Within a key, replicas are sampled -without replacement, not independently. - -The actual primitive is an alternating-XOR Feistel network with SplitMix64's -avalanche finalizer as its round mixer. It uses 24 rounds for widths 2 through -4, 16 for widths 5 through 7, and 8 for widths 8 through 64. Tiny half-domains -need extra rounds: the initial eight-round version had a strong ordered-pair -bias at width three despite clean marginals. This fixed policy depends **only** -on bit width, never active `n`, requested `k`, or observed outputs. Unequal -halves support odd widths; the one-bit case is a keyed XOR. -Widths are domain-separated through a -mixed seed; round keys use distinct Weyl offsets. The inverse undoes the same -updates in reverse order. Geometry never requires a shift by 64: the largest -half-width is 32 and the evaluator's largest half boundary is `1 << 63`. -The conceptual full width-64 domain has size `2^64`, but that cardinality is -never materialized in a `u64`. - -This finite, 64-bit seeded family is **noncryptographic and only a practical -pseudorandom approximation**, not independently sampled ideal permutations, -not a proven secure PRP, and not exactly uniform over all `n!` permutations. -Domain separation and avalanche do not prove independence. Seed collisions -give identical permutations. Even conventional XOR-Feistel families have -structural restrictions (for example, even permutation parity when both -halves have at least two bits). The mixing schedule is fixed rather than -weakened to make a benchmark win. Do not use this as encryption or with -adversarial keys requiring a cryptographic guarantee. - -The existing `layer_apply` is deliberately unchanged: it only supports even -widths through 30, has no inverse, and has its own key/round schedule. Extending -or replacing it would change existing mappings. The new private primitive -therefore lives with the new evaluator. - -## Dyadic lift and cycle deletion - -The following is a derived construction and evaluator, not an implementation -of a published constant-time replica-selection algorithm. - -Let `h = 2^(b-1)`, let `A = [0,h)` be the old labels, and let `Q = P(seed,b)` -be an ordinary permutation of `[0,2h)`. Let `R` be its cycle projection onto -`A`: follow `Q` until the next label in `A`. Starting with `F_1(0) = 0`, define +identical algorithm versions on all participants. Seed collisions give +identical orders. + +Internal label zero is a permanent sentinel; real node `i` has internal label +`i + 1`. Internal count is `n + 1`, so `new(0, seed)` and +`new(u64::MAX, seed)` panic rather than wrap, following the existing +constructor's assertion convention. `n()` returns the original real-node +count. The cloneable, fused iterator reports its remaining size, safely even +when that count exceeds `usize`. `.take(0)` is empty; `.take(n)` is the complete +order; larger requests stop at exhaustion. There is no separate `k` constructor +argument. `nth(r)` uses ordinary iterator replay from the current position; +answering an uncached absolute rank requires replay from a new iterator. + +## Ordinary permutation and guaranteed single cycle + +For each bit width `b`, let `Q(seed,b)` be an ordinary reversible permutation +of `[0,2^b)`, with domain separation depending only on seed and width. Define +the single-cycle primitive by conjugating modular increment: + +```text +P(x) = Q_inverse((Q(x) + 1) mod 2^b) +P_inverse(x) = Q_inverse((Q(x) - 1) mod 2^b) +``` + +Each is two ordinary permutation evaluations: first `Q`, then `Q_inverse`. +Increment is a full cycle, so its conjugate is a full cycle for **every** Q, +not merely with high probability. If Q is uniform on all permutations of a +domain of size `m`, each full cycle has exactly `m` conjugators, so P is uniform +on the `(m-1)!` full cycles. + +The implemented Q is a noncryptographic alternating-XOR Feistel with the +SplitMix64 finalizer as round mixer. It uses 24 rounds at widths 2--4, +16 at widths 5--7, and 8 at widths 8--64; width one is keyed XOR. This fixed +schedule, inherited from the stronger experimental primitive, has been +rediagnosed in the **new construction**, not assumed adequate from old results. +Unequal halves support odd widths. Width seeds are mixed and round keys use +Weyl offsets. The inverse undoes the same updates in reverse order. + +This finite 64-bit family is only a practical pseudorandom approximation, +not exact independent ideal randomness or a proven secure PRP. It cannot +uniformly represent all orders once there are more orders than seeds. +Feistel families also have structural restrictions, such as permutation +parity restrictions on sufficiently large balanced halves. Neither domain +separation, a large round count nor statistical diagnostics proves independence. +Do not use this as encryption or where adversarial keys require cryptographic +security. The existing iterator's different Feistel schedule is not modified. + +All word operations are safe through width 64: modular steps use wrapping +arithmetic followed by masking, half widths never exceed 32, and the largest +dyadic half boundary is `1 << 63`. A conceptual domain cardinality `2^64` +is never stored in a `u64`. + +## Normalized cycle lift + +This is a derived construction/evaluator, not an established production +implementation or an implementation of a published constant-time algorithm. + +Let `A=[0,h)` and let P be a full cycle on `[0,2h)`. Let R be P's cycle +projection onto A: follow P until reaching the next old label. Starting with +`F_1(0)=0`, define: ```text -F_(2h) = extend(F_h composed with inverse(R), fixing upper labels) composed with Q +F_(2h) = extend(F_h composed with inverse(R), fixing upper labels) composed with P ``` -Every `Q`-chain starting at old label `a` ends at old label `R(a)`. -The left composition changes only that chain's final destination, from -`R(a)` to `F_h(a)`. Upper edges and upper-only cycles are unchanged. Thus -cycle-projecting `F_(2h)` onto `A` gives `F_h`. For `h < n < 2h`, define `F_n` -by deleting all labels at least `n` from `F_(2h)`'s cycles. Cycle projections -compose, including across power-of-two boundaries. - -**Uniformity in the ideal model.** Conditional on any fixed `Q`, independent -uniform `F_h` makes `F_h composed with inverse(R)` uniform on the old-label -permutation group. This factor is consequently independent of `Q`; composing -it with uniform `Q` makes `F_(2h)` uniform. Cycle projection preserves -uniformity: each permutation of `n-1` labels has exactly `n` extensions -(insert the new label after any old label, or as a singleton cycle). -Induction establishes uniformity at every size. This proof uses ideal -uniformity, not the diagnostics of the implemented finite-key family. - -**Consistency.** Deleting label `n` from `F_(n+1)` only redirects its -predecessor to its successor, or removes its singleton cycle. For each -surviving input `r < n`, `F_(n+1)(r)` is therefore either `F_n(r)` or `n`. -At most one old input changes. A fixed prefix of input slots inherits this -property, while bijectivity supplies distinct outputs and prefix stability. -Enumerate *distinct inputs* `0,1,...,k-1`; following one output cycle instead -would not enumerate a full permutation. - -## Evaluator +For every old label `a`, its P-chain goes through zero or more upper labels +and ends at `R(a)`. The lift changes that final destination to `F_h(a)`. +There are no upper-only cycles in P. Reconnecting all old-node chains using +the lower full cycle therefore produces exactly one full cycle. Its cycle +projection onto A is F_h. For intermediate counts, delete all inactive labels +from the larger cycle. Projection composes, including across dyadic boundaries. + +**Uniform full cycles under the ideal model.** A full cycle P decomposes into +its projected lower cycle R and one ordered upper-node chain attached to each +old label. Every combination of a lower full cycle and such a chain arrangement +corresponds to exactly one full P. Thus uniform P makes R uniform independently +of the arrangement. The lift replaces R with the independent uniform F_h +without changing the arrangement, giving a uniform full F_(2h). Projection of +a uniform full cycle is uniform: each cycle on `m-1` labels has exactly `m-1` +extensions, inserting the new label after any old label. Induction gives +uniform full cycles at every internal count. + +**Rooted list order.** Begin at sentinel zero and repeatedly follow F: + +```text +cursor = 0 +repeat k times, where 0 <= k <= n: + cursor = next_consistent(seed, n + 1, cursor) + emit cursor - 1 +``` + +The sentinel cannot reappear before all `n` real nodes. A full cycle with a +fixed sentinel corresponds bijectively to a real-node order. Cycle deletion +therefore becomes ordinary list deletion: for example `S->A->C->B->S` can +grow into `S->A->D->C->B->S`, preserving the old order. Each ideal real-node +order, ordered prefix, or subset has the appropriate uniform distribution. +Different keys have independent orders **only under independent ideal +primitives across keys**; nodes within a key are sampled without replacement. + +Evaluating successor inputs `0,1,...,k-1` instead of following the cursor would +not implement this contract. Internal input-label successor consistency is +not the public output-rank API. + +## Constant-space successor evaluator ```text -replica_at(seed, n, r): - require 0 <= r < n - b = bit_length(n - 1) - x = r +next_consistent(seed, count, x): + require 0 <= x < count + b = bit_length(count - 1) while b > 0: half = 1 << (b - 1) y = P(seed, b, x) - if y >= n: + if y >= count: x = y continue if y >= half: return y while x >= half: x = P_inverse(seed, b, x) - n = half + count = half b -= 1 return 0 ``` -Walking inactive upper labels can use `Q` rather than the recursively defined -`F`, because edges whose outputs are upper labels were unchanged by the lift. -On reaching a lower output `y`, walking backward from its predecessor `x` -finds the old input `a = inverse(R)(y)`. The next level evaluates `F_h(a)`. -This proves the evaluator agrees with the explicit lift. - -Walks terminate because they follow a finite bijection's cycle. A forward walk -starts in the retained set, so it cannot remain forever in an inactive-only -cycle. A backward walk is entered only after reaching a lower label, so that -cycle necessarily meets the lower half. There are no retry caps or mapping- -changing fallbacks. - -## Expected work, not a worst-case guarantee - -Count each forward or inverse permutation evaluation as one constant-cost -operation. The following bounds are derived for **ideal independent uniform** -permutations and a fixed input, not asserted as a published theorem or a -guarantee for every seed of the implemented mixer. - -At a full dyadic level, descent probability is exactly one half. For an -upper input conditional on descent, the mean inverse-walk length is -`2h/(h+1)`: predecessors are sampled without replacement from `2h-1` labels, -`h` of them lower. The terminal lower input is uniform. If `T_s(a)` is mean -cost at full size `s`, and `U_s` its average over inputs, then - -```text -T_(2h)(a) = 1 + T_h(a)/2 if a < h -T_(2h)(a) = 1 + h/(h+1) + U_h/2 if a >= h -U_(2h) = 1 + h/(2(h+1)) + U_h/2 -T_1 = U_1 = 0 -``` - -Induction gives `U_s < 3` and `T_s(a) < 3.5`. The initial partial level is -different: its descent probability `p = h/n` can approach one. Its forward -length has mean `ell = (2h+1)/(n+1) < 2`; the retained endpoint is uniform -and independent of that length. On descent, the inverse walk retraces those -`ell-1` inactive edges. Starting from an upper input also requires finding a -lower predecessor; conditional on forward length `L`, this takes -`(2h-L+1)/(h+1)` further inverse calls on average. Thus - -```text -E C_n(r) = ell + p * (ell - 1 + T_h(r)) if r < h -E C_n(r) = ell + p * (ell - 1 + (2h+1-ell)/(h+1) + U_h) if r >= h -``` - -In particular, mean total work is **less than eight primitive calls per -slot**, uniformly in `n,r` in the ideal model. The upper-input bound approaches -eight at `n=h+1, r=h` as `h` grows. All later levels are full, so descent is -geometric after the initial level. The entering lower input depends on upper -permutations, but is independent of lower-width permutations; no independence -between walks, or between replica costs, is assumed. Linearity of expectation -gives expected `O(k)` enumeration. - -Long cycles still permit linear-in-domain walks for an unlucky permutation. -This is **not worst-case `O(k)`**, nor an adversarial-latency guarantee. State -is `O(1)` machine words, excluding optional returned output, with a -constant-sized width parameter block and no recursive stack, ring, permutation -array, or duplicate set. See [measurements and diagnostics](virtual-permutation-performance.md) -for observed forward/inverse counts and tails on the actual mixer. - -## Matched even-width, two-bit variant - -The cycle-lift identity is not restricted to doubling. For the matched variant -let the full domain have size `4h`, with old set `[0,h)`. Reconnect old -destinations with the same `extend(F_h composed with inverse(R))` operation. -The chain argument and ideal-model uniformity proof above work unchanged. -Start at the smallest **even** bit width covering `n`, use boundary -`1 << (bits - 2)`, and descend by two bits. Intermediate sizes still use cycle -deletion, not output-list deletion. - -`BalancedVirtualPermutation` directly calls the existing `layer_apply` for -every forward evaluation. It uses the same master seed, round function, -12/10/6/4 round schedule at widths 2/4/6/8-and-above, and rotated-key/Weyl -schedule. There is no additional per-width seed hash. The new `layer_inverse` -undoes that exact mapping and key schedule, sharing the extracted round -function. Known-answer vectors, exhaustive small domains, all supported -widths, and extreme keys/inputs check the inverse and unchanged forward -mapping. - -Full-level descent now has probability one quarter in the ideal model, -instead of one half. The initial partial level can be less than half full, -however, so its forward and inverse walks can be longer. Expected work is -still bounded per slot under independent ideal permutations; the specific -less-than-eight-call bound above is for the one-bit variant and is **not** -claimed for the two-bit variant. Inverse evaluation also has real work to -recover the final round key before reversing the rounds; the benchmarks -charge that cost and do not cache it for free. - -This changes **both** layer stride and primitive compared with -`VirtualPermutation`; timing differences are not a pure attribution to odd -versus even halves alone. It does make the primitive and stride identical to -the existing streaming iterator. Neither implementation samples independent -uniform permutations at different widths: the shared 64-bit seed and the -finite Feistel family remain approximations. The matched variant's observed -bias is a substantive limitation, not explained away by the ideal proof. -The stronger one-bit variant remains available unchanged. - -## References and verification - -The mathematical antecedent is a *virtual permutation*, using **cycle** -projection: +Edges whose outputs are upper labels are unchanged by the lift, so walking +inactive upper labels can use P directly. Upon reaching a lower output, +the backward walk recovers the old input at the start of that chain; recursion +then supplies its correct lower-cycle destination. This is the explicit lift +without constructing any tables. The actual implementation uses loops, not a +recursive stack. Walks terminate because P is a full finite cycle intersecting +the retained/lower set. No retry cap or mapping-changing fallback is used. + +## Expected adaptive-prefix work + +The cursor depends on previous outputs. A fixed-input expected-cost bound +would therefore be insufficient. Instead count work over the entire rooted +prefix, under independent uniform ideal Q at each width and constant-cost +forward/inverse ordinary permutation calls. Fix `n,k` before sampling those +primitives. A zero-length prefix does no traversal; the strict bounds below +are for `1<=k<=n`. + +Let `M` be the smallest power of two at least `n+1`. The lifted full cycle +F_M (not the raw primitive P) is uniform. Reaching the first `k` active real outputs visits, in expectation, +`k*M/(n+1)` top-level successors: sample without replacement from the +`M-1` non-sentinel labels until the `k`th of the `n` active labels. This is +less than `2k`. + +At each lower full dyadic domain of size `L=M/2,M/4,...,2`, the successor +invocations form the rooted lower-label subsequence. Their count equals the +number of selected final internal labels in `1..L-1`, so its expectation is +`k*(L-1)/n`. This uses marginal uniformity of the rooted real-node order, not +independence of adaptive calls. Summing these lower-level expectations gives +less than `k*M/n`, which is at most `2k` for `n>=1`. + +At any level, a backward walk retraces an upper-node chain already traversed +by that level's forward prefix. This includes inactive labels just visited +in the top-level walk. Each upper label is retraced at most once; the prefix +does not wrap back through the sentinel. Consequently inverse calls are +bounded pathwise by forward calls across the prefix. Combining the counts +gives a conservative **expected bound below `8k` P/P_inverse calls, or `16k` +ordinary Q/Q_inverse calls**. Width setup is bounded per level invocation and +does not change expected `O(k)` work. This bounds cumulative work from the +sentinel, not work conditioned on an arbitrary already-observed prefix. + +This is an ideal-model derivation, not a published complexity theorem or a +guarantee for every finite-key seed. An unlucky full cycle can force +linear-in-domain work even for a short prefix; there is **no worst-case +`O(k)` or adversarial-latency guarantee**. The iterator uses `O(1)` auxiliary +state excluding returned outputs. There is no prebuilt ring, per-key table, +permutation array, cached prefix or duplicate set. + +## Verification and references + +Tests cover forward/inverse round trips for both Q and P through all 64 +widths, guaranteed single-cycle coverage on small domains, explicit +table-based lift/projection equivalence, rooted traversal and sentinel return, +complete survivor-list restriction, `k` prefixes, default `nth` replay, fused +exhaustion, and small/large dyadic and sentinel boundaries. They explicitly +reject overflowing sentinel counts. A constructed long-walk case checks that +evaluation is not truncated. + +Exhaustive ideal checks cover all 24 ordinary size-four conjugators and all +30,240 independent full-cycle families at sizes 2,4,8. Every real-node order +at sizes 1 through 7 appears equally often and agrees with the explicit +reference and survivor deletion. These are regression checks, not substitutes +for the ideal-model argument or proofs of security. + +The mathematical antecedents concern virtual permutations and cycle projection: - Neretin, [Virtual permutations and polymorphisms, section 1.2](https://arxiv.org/html/2202.12978v1#S1): - cycle deletion and its equivariance. + cycle deletion and equivariance. - Bourgade, Najnudel and Nikeghbali, [A unitary extension of virtual permutations, section 1](https://arxiv.org/html/1102.2633v1#S1): - cycle projections, the Chinese restaurant construction, and the uniform - family as the Ewens parameter-one case. - -These references do **not** supply this evaluator, Feistel schedule, or its -performance bound. - -Tests exercise all word widths and extreme values, exhaustive small PRP -round-trips, full small-domain permutations, `k` prefixes, append/delete -consistency including dyadic boundaries and `u64::MAX`, and an independent -table-based lift/projection oracle. Exhaustive ideal permutations check equal -lift multiplicities at size four and all extensions of one fixed size-four -permutation through sizes five to eight. These are regression checks, not -substitutes for the ideal-model proof or evidence of cryptographic security. + cycle projections, Chinese restaurant construction and uniform coherent families. + +These sources do **not** supply this evaluator, Feistel schedule, single-cycle +sentinel specialization or performance bound. The +[performance and randomness report](virtual-permutation-performance.md) +records actual final-construction measurements and their limitations. diff --git a/crates/consistent-choose-k/examples/permutation_diagnostics.rs b/crates/consistent-choose-k/examples/permutation_diagnostics.rs index ad1e7d6..647f430 100644 --- a/crates/consistent-choose-k/examples/permutation_diagnostics.rs +++ b/crates/consistent-choose-k/examples/permutation_diagnostics.rs @@ -1,34 +1,116 @@ -//! Deterministic distribution diagnostics, not probabilistic CI gates. -//! cargo run --release -p consistent-choose-k --example permutation_diagnostics +//! Deterministic distribution diagnostics, not statistical CI gates. +//! Run with no arguments for the primary corpus, or --held-out. use std::hash::{DefaultHasher, Hash, Hasher}; -use consistent_choose_k::{BalancedVirtualPermutation, ConsistentPermutation, VirtualPermutation}; +use consistent_choose_k::{ConsistentPermutation, VirtualPermutation}; -const SAMPLES: u64 = 200_000; -fn key_seed(key: u64, workload_seed: u64) -> u64 { +fn key_seed(key: u64, corpus: u64) -> u64 { let mut hasher = DefaultHasher::new(); - (workload_seed ^ key).hash(&mut hasher); + (corpus ^ key).hash(&mut hasher); hasher.finish() } -fn report(algorithm: &str, n: usize, metric: &str, cells: &[u64]) { - let expected = cells.iter().sum::() as f64 / cells.len() as f64; - let chi2: f64 = cells - .iter() - .map(|&v| (v as f64 - expected).powi(2) / expected) - .sum(); - println!( - "{algorithm},{n},{metric},{},{expected:.3},{chi2:.3},{},{}", - cells.len() - 1, - cells.iter().min().expect("nonempty histogram"), - cells.iter().max().expect("nonempty histogram"), - ); +struct Histogram { + counts: Vec, + probabilities: Vec, +} + +impl Histogram { + fn uniform(cells: usize) -> Self { + Self { + counts: vec![0; cells], + probabilities: vec![1.0 / cells as f64; cells], + } + } + + fn pairs(n: usize, buckets: usize, distinct: bool) -> Self { + let mut sizes = vec![0; buckets]; + for node in 0..n { + sizes[node * buckets / n] += 1; + } + let mut probabilities = vec![]; + for a in 0..buckets { + for b in 0..buckets { + let available = sizes[b] - usize::from(distinct && a == b); + probabilities + .push((sizes[a] * available) as f64 / (n * (n - usize::from(distinct))) as f64); + } + } + Self { + counts: vec![0; buckets * buckets], + probabilities, + } + } + + fn record(&mut self, cell: usize) { + self.counts[cell] += 1; + } + + fn report(&self, algorithm: &str, n: usize, metric: &str) { + let total: u64 = self.counts.iter().sum(); + let mut chi2 = 0.0; + let mut tv = 0.0; + let mut max_relative: f64 = 0.0; + let mut min_expected = f64::INFINITY; + let mut cells = 0; + for (&count, &probability) in self.counts.iter().zip(&self.probabilities) { + if probability == 0.0 { + assert_eq!(count, 0, "impossible pair observed"); + continue; + } + let expected = total as f64 * probability; + assert!(expected >= 9.99, "inadequate expected cell count"); + let delta = (count as f64 - expected).abs(); + chi2 += delta * delta / expected; + tv += delta / total as f64 / 2.0; + max_relative = max_relative.max(delta / expected); + min_expected = min_expected.min(expected); + cells += 1; + } + println!( + "{algorithm},{n},{metric},{total},{},{min_expected:.3},{chi2:.3},{:.3},{tv:.6}", + cells - 1, + max_relative * 100.0 + ); + } +} + +fn order(algorithm: &str, n: usize, seed: u64) -> Vec { + match algorithm { + "layered" => ConsistentPermutation::new(n as u32, seed) + .map(|v| v as usize) + .collect(), + "sentinel" => VirtualPermutation::new(n as u64, seed) + .map(|v| v as usize) + .collect(), + _ => unreachable!(), + } +} + +fn first(algorithm: &str, n: usize, seed: u64) -> usize { + match algorithm { + "layered" => ConsistentPermutation::new(n as u32, seed) + .next() + .expect("nonempty") as usize, + "sentinel" => VirtualPermutation::new(n as u64, seed) + .next() + .expect("nonempty") as usize, + _ => unreachable!(), + } +} + +fn permutation_index(values: &[usize]) -> usize { + let mut index = 0; + for (i, &value) in values.iter().enumerate() { + index = index * (values.len() - i) + values[i + 1..].iter().filter(|&&v| v < value).count(); + } + index } fn main() { let mut args = std::env::args().skip(1); - let workload_seed = match args.next().as_deref() { + let corpus = match args.next().as_deref() { None => 0x7065_726d_7574_6531, Some("--held-out") => 0x686f_6c64_6f75_7431, Some(_) => panic!("usage: permutation_diagnostics [--held-out]"), @@ -37,69 +119,112 @@ fn main() { args.next().is_none(), "usage: permutation_diagnostics [--held-out]" ); - println!("# samples={SAMPLES}, key_seed={workload_seed:#x}"); + println!("# corpus={corpus:#x}; exact bins require >=10 expected observations"); println!( - "# iterator_bytes: layered={}, virtual={}, balanced={}; layered also owns heap counters", + "# iterator_bytes: layered={}, sentinel={}; layered also owns heap counters", std::mem::size_of::(), - std::mem::size_of::(), - std::mem::size_of::(), + std::mem::size_of::() ); - println!("algorithm,n,metric,df,expected_per_cell,chi2,min,max"); - for n in [3, 4, 5, 7, 8, 9, 16, 17, 32, 64] { - for algorithm in ["layered", "virtual", "balanced"] { - let slots = n.min(8); - let mut marginal = vec![vec![0u64; n]; slots]; - let mut pairs = vec![0; n * n]; - let mut distant_pairs = vec![0; n * n]; - let mut triples = vec![0; n * (n - 1) * (n - 2) / 6]; - let mut consecutive_keys = vec![0; n * n]; - let mut previous_primary = None; - for key in 0..SAMPLES { - let seed = key_seed(key, workload_seed); - let values: Vec = if algorithm == "layered" { - ConsistentPermutation::new(n as u32, seed) - .take(slots) - .map(|v| v as usize) - .collect() - } else if algorithm == "virtual" { - VirtualPermutation::new(n as u64, seed) - .take(slots) - .map(|v| v as usize) - .collect() - } else { - BalancedVirtualPermutation::new(n as u32, seed) - .take(slots) - .map(|v| v as usize) - .collect() - }; - for (histogram, &v) in marginal.iter_mut().zip(&values) { - histogram[v] += 1; + println!( + "algorithm,n,metric,observations,df,min_expected,chi2,max_relative_percent,total_variation" + ); + for n in [ + 1usize, 2, 3, 4, 5, 6, 7, 8, 9, 14, 15, 16, 17, 30, 31, 32, 33, 62, 63, 64, 65, 126, 127, + 128, 129, 254, 255, 256, 257, + ] { + let samples = if n <= 33 { + 200_000 + } else if n <= 65 { + 50_000 + } else { + 20_000 + }; + let buckets = if samples / (n * n) >= 10 { n } else { 8 }; + let pair_kind = if buckets == n { "exact" } else { "bucket8" }; + let mut ranks = vec![ + 0, + 1.min(n - 1), + 2.min(n - 1), + n / 2, + n.saturating_sub(2), + n - 1, + ]; + ranks.sort_unstable(); + ranks.dedup(); + let mut rank_pairs = vec![ + (0, 1.min(n - 1)), + (n / 2, (n / 2 + 1).min(n - 1)), + (0, n / 2), + (0, n - 1), + ]; + rank_pairs.retain(|(a, b)| a != b); + rank_pairs.sort_unstable(); + rank_pairs.dedup(); + let triple_cells = if n >= 3 { n * (n - 1) * (n - 2) / 6 } else { 0 }; + let triple_ok = triple_cells > 0 && samples / triple_cells >= 10; + if n >= 3 && !triple_ok { + println!( + "# n={n}: skipping exact choose_3; expected={:.3}", + samples as f64 / triple_cells as f64 + ); + } + for algorithm in ["layered", "sentinel"] { + let mut marginals: Vec<_> = ranks.iter().map(|_| Histogram::uniform(n)).collect(); + let mut pairs: Vec<_> = rank_pairs + .iter() + .map(|_| Histogram::pairs(n, buckets, true)) + .collect(); + let mut triples = triple_ok.then(|| Histogram::uniform(triple_cells)); + let mut full = (n <= 7).then(|| Histogram::uniform((1..=n).product())); + let mut adjacent_keys = Histogram::pairs(n, buckets, false); + let mut related_seeds = Histogram::pairs(n, buckets, false); + let mut previous = None; + for key in 0..samples as u64 { + let seed = key_seed(key, corpus); + let values = order(algorithm, n, seed); + assert_eq!(values.len(), n); + for (hist, &rank) in marginals.iter_mut().zip(&ranks) { + hist.record(values[rank]); + } + for (hist, &(a, b)) in pairs.iter_mut().zip(&rank_pairs) { + assert_ne!(values[a], values[b]); + hist.record((values[a] * buckets / n) * buckets + values[b] * buckets / n); } - pairs[values[0] * n + values[1]] += 1; - distant_pairs[values[0] * n + values[slots - 1]] += 1; - let mut triple = [values[0], values[1], values[2]]; - triple.sort_unstable(); - let [a, b, c] = triple; - triples[c * (c - 1) * (c - 2) / 6 + b * (b - 1) / 2 + a] += 1; - if let Some(previous) = previous_primary { - consecutive_keys[previous * n + values[0]] += 1; + if let Some(hist) = &mut triples { + let mut triple = [values[0], values[1], values[2]]; + triple.sort_unstable(); + let [a, b, c] = triple; + hist.record(c * (c - 1) * (c - 2) / 6 + b * (b - 1) / 2 + a); } - previous_primary = Some(values[0]); + if let Some(hist) = &mut full { + hist.record(permutation_index(&values)); + } + let primary = values[0] * buckets / n; + if let Some(old) = previous { + adjacent_keys.record(old * buckets + primary); + } + previous = Some(primary); + let related = first(algorithm, n, seed ^ (1u64 << 63)) * buckets / n; + related_seeds.record(primary * buckets + related); + } + for (hist, rank) in marginals.iter().zip(&ranks) { + hist.report(algorithm, n, &format!("rank_{rank}")); + } + for (hist, (a, b)) in pairs.iter().zip(&rank_pairs) { + hist.report(algorithm, n, &format!("ordered_{a}_{b}_{pair_kind}")); } - for (slot, cells) in marginal.iter().enumerate() { - report(algorithm, n, &format!("slot_{slot}"), cells); + if let Some(hist) = triples { + hist.report(algorithm, n, "choose_3"); } - for (metric, cells) in [("ordered_0_1", pairs), ("ordered_0_last", distant_pairs)] { - assert!((0..n).all(|v| cells[v * n + v] == 0)); - let off_diagonal: Vec<_> = cells - .into_iter() - .enumerate() - .filter_map(|(i, count)| (i / n != i % n).then_some(count)) - .collect(); - report(algorithm, n, metric, &off_diagonal); + if let Some(hist) = full { + hist.report(algorithm, n, "full_order"); } - report(algorithm, n, "choose_3", &triples); - report(algorithm, n, "consecutive_key_primaries", &consecutive_keys); + adjacent_keys.report( + algorithm, + n, + &format!("consecutive_key_primaries_{pair_kind}"), + ); + related_seeds.report(algorithm, n, &format!("seed_bitflip_primaries_{pair_kind}")); } } } diff --git a/crates/consistent-choose-k/src/consistent_permutation.rs b/crates/consistent-choose-k/src/consistent_permutation.rs index fb1d3a5..4d715c8 100644 --- a/crates/consistent-choose-k/src/consistent_permutation.rs +++ b/crates/consistent-choose-k/src/consistent_permutation.rs @@ -75,14 +75,6 @@ fn splitmix64(seed: u64) -> u64 { z ^ (z >> 31) } -#[inline] -fn layer_round(r: u32, key: u64, half_mask: u32) -> u32 { - let k_xor = key & 0xFFFF_FFFF; - let k_mul = (key >> 32) | 1; - let mixed = ((r as u64) ^ k_xor).wrapping_mul(k_mul); - (mixed as u32).wrapping_add((mixed >> 32) as u32) & half_mask -} - /// Apply the per-layer Feistel bijection on `[0, 2^n_bits)`. /// `master_key` must be well-avalanched: two near-identical keys /// will yield two highly correlated permutations. @@ -112,7 +104,10 @@ pub(crate) fn layer_apply(n_bits: u32, master_key: u64, x: u32) -> u32 { // high half (forced odd) is the multiplier. Higher bits stay // set on purpose — they contribute via the multiplicative // mix. - let f = layer_round(r, k, half_mask); + let k_xor = k & 0xFFFF_FFFF; + let k_mul = (k >> 32) | 1; + let mixed = ((r as u64) ^ k_xor).wrapping_mul(k_mul); + let f = (mixed as u32).wrapping_add((mixed >> 32) as u32) & half_mask; let new_l = r; let new_r = l ^ f; l = new_l; @@ -127,30 +122,6 @@ pub(crate) fn layer_apply(n_bits: u32, master_key: u64, x: u32) -> u32 { ((l << half_bits) | r) & n_mask } -/// Invert exactly the existing layer permutation, including its key schedule. -#[inline] -pub(crate) fn layer_inverse(n_bits: u32, master_key: u64, x: u32) -> u32 { - debug_assert!((2..=30).contains(&n_bits) && n_bits.is_multiple_of(2)); - let rounds = rounds_for_n_bits(n_bits); - let half_bits = n_bits / 2; - let half_mask = (1u32 << half_bits) - 1; - debug_assert!(x < 1u32 << n_bits, "input out of range"); - let mut l = x >> half_bits; - let mut r = x & half_mask; - let mut key = master_key; - for _ in 1..rounds { - key = key.rotate_right(n_bits).wrapping_add(0x9E37_79B9_7F4A_7C15); - } - for _ in 0..rounds { - let old_r = l; - let old_l = r ^ layer_round(l, key, half_mask); - l = old_l; - r = old_r; - key = key.wrapping_sub(0x9E37_79B9_7F4A_7C15).rotate_left(n_bits); - } - (l << half_bits) | r -} - /// `n`-consistent permutation iterator over `0..n` driven by one /// bijection per level (see module docs). pub struct ConsistentPermutation { @@ -266,38 +237,6 @@ mod tests { use super::*; - #[test] - fn layer_mapping_known_answers_and_inverse() { - for (bits, key, x, y) in [ - (2, 0, 0, 0), - (4, 0x1234_5678_9abc_def0, 9, 2), - (6, 1, 63, 62), - (8, 0x1234_5678_9abc_def0, 200, 100), - (16, u64::MAX, 65535, 0x3a83), - (30, 0x1234_5678_9abc_def0, 0x3fff_ffff, 0x1949_4328), - ] { - assert_eq!(layer_apply(bits, key, x), y); - assert_eq!(layer_inverse(bits, key, y), x); - } - for bits in (2..=30).step_by(2) { - let mask = (1u32 << bits) - 1; - for key in [0, 1, u64::MAX, splitmix64(42)] { - let inputs: Vec<_> = if bits <= 10 { - (0..=mask).collect() - } else { - [0, 1, mask / 2, mask / 2 + 1, mask] - .into_iter() - .chain((0..128).map(|i| splitmix64(i) as u32 & mask)) - .collect() - }; - for x in inputs { - assert_eq!(layer_inverse(bits, key, layer_apply(bits, key, x)), x); - assert_eq!(layer_apply(bits, key, layer_inverse(bits, key, x)), x); - } - } - } - } - /// Across many seeds, `layer_apply(_, key, 0)` must not always /// collapse to the same value (regression test for the "F(0, k) = /// 0" symmetry bug). diff --git a/crates/consistent-choose-k/src/lib.rs b/crates/consistent-choose-k/src/lib.rs index 1b2e45a..4278f83 100644 --- a/crates/consistent-choose-k/src/lib.rs +++ b/crates/consistent-choose-k/src/lib.rs @@ -12,4 +12,4 @@ pub use consistent_hash::{ pub use consistent_permutation::ConsistentPermutation; pub use consistent_reservoir::ConsistentReservoir; pub use node_map::ConsistentNodeMap; -pub use virtual_permutation::{BalancedVirtualPermutation, VirtualPermutation}; +pub use virtual_permutation::VirtualPermutation; diff --git a/crates/consistent-choose-k/src/virtual_permutation.rs b/crates/consistent-choose-k/src/virtual_permutation.rs index d55bd3b..e12ef26 100644 --- a/crates/consistent-choose-k/src/virtual_permutation.rs +++ b/crates/consistent-choose-k/src/virtual_permutation.rs @@ -1,10 +1,8 @@ -//! Cycle-consistent permutations, as opposed to the survivor-list consistency -//! of `ConsistentPermutation`. See `docs/virtual-permutation.md`. +//! Sentinel-rooted, single-cycle consistent permutations. +//! See `docs/virtual-permutation.md` for the construction and assumptions. use std::iter::FusedIterator; -use crate::consistent_permutation::{layer_apply, layer_inverse}; - const WEYL: u64 = 0x9E37_79B9_7F4A_7C15; #[inline] @@ -19,8 +17,7 @@ trait Permutation { fn inverse(&self, x: u64) -> u64; } -/// Alternating Feistel half updates, including unequal halves for odd widths. -/// This is a noncryptographic family, not a uniform or secure PRP. +/// Ordinary noncryptographic Feistel Q, not itself the consistent map or cycle. struct WordPermutation { key: u64, bits: u32, @@ -68,8 +65,6 @@ impl WordPermutation { #[inline] fn transform(&self, x: u64) -> u64 { - // Tiny half-domains need more mixing: eight rounds had strong - // ordered-pair bias at width three despite clean marginals. match self.bits { 1 => x ^ (self.key & 1), 2..=4 => self.transform_rounds::<12, INVERSE>(x), @@ -91,113 +86,123 @@ impl Permutation for WordPermutation { } } +struct SingleCycle { + ordinary: Q, + mask: u64, +} + +impl Permutation for SingleCycle { + #[inline] + fn forward(&self, x: u64) -> u64 { + self.ordinary + .inverse(self.ordinary.forward(x).wrapping_add(1) & self.mask) + } + + #[inline] + fn inverse(&self, x: u64) -> u64 { + self.ordinary + .inverse(self.ordinary.forward(x).wrapping_sub(1) & self.mask) + } +} + #[inline] -fn evaluate(n: u64, x: u64, layer: impl FnMut(u32) -> P) -> u64 { - evaluate_with_stride::<1, P>(n, x, layer) +fn cycle(seed: u64, bits: u32) -> SingleCycle { + SingleCycle { + ordinary: WordPermutation::new(seed, bits), + mask: u64::MAX >> (64 - bits), + } } #[inline] -fn evaluate_with_stride( - mut n: u64, - mut x: u64, - mut layer: impl FnMut(u32) -> P, -) -> u64 { - debug_assert!(STEP == 1 || STEP == 2); - let mut bits = u64::BITS - (n - 1).leading_zeros(); - bits = bits.div_ceil(STEP) * STEP; +fn evaluate(mut count: u64, mut x: u64, mut layer: impl FnMut(u32) -> P) -> u64 { + debug_assert!(count > 0 && x < count); + let mut bits = u64::BITS - (count - 1).leading_zeros(); while bits > 0 { - let boundary = 1u64 << (bits - STEP); + let half = 1u64 << (bits - 1); let permutation = layer(bits); loop { let y = permutation.forward(x); - if y >= n { + if y >= count { x = y; continue; } - if y >= boundary { + if y >= half { return y; } - // Find the old input at the start of this Q-chain, not its old - // output y: this applies inverse(cycle_projection(Q)) before - // evaluating the smaller consistent permutation. - while x >= boundary { + // Recover the old input at the start of this chain, rather than + // its old destination y, before evaluating the lower-level cycle. + while x >= half { x = permutation.inverse(x); } break; } - n = boundary; - bits -= STEP; + count = half; + bits -= 1; } 0 } -/// An allocation-free, per-key permutation of `0..n` with stable replica slots. +/// An allocation-free, per-key ordering of `0..n` preserving survivor order. /// -/// For fixed `(n, seed)`, taking `k` items gives `k` distinct nodes and is a -/// prefix of every longer selection. Appending node `n` changes at most one -/// existing slot, and that slot changes to `n`; removing the last node changes -/// only its slot (among slots that still exist). Membership must be a prefix -/// of consecutive IDs. +/// Every prefix contains distinct nodes. Appending a node inserts it somewhere +/// in the complete order; removing the last node deletes it without reordering +/// survivors. Membership must be consecutive IDs `0..n`. Replica ranks may +/// shift when membership changes. /// -/// **Not a drop-in replacement for [`crate::ConsistentPermutation`]:** this -/// preserves cycle projections, not survivor list order. Removing a node from -/// the output list does not generally produce the smaller permutation. +/// The iterator traverses one consistent cycle from a permanent internal +/// sentinel. It does not evaluate independent rank inputs or cache a prefix. +/// `nth(r)` replays `r + 1` successors from the current position; there is no +/// constant-time absolute-rank API. /// -/// Uniform, independent permutations per key and bit width give uniform -/// selections and expected `O(k)` evaluation with constant-cost forward/inverse -/// primitives. The implemented 64-bit seeded, 8/16/24-round Feistel family is -/// only a practical noncryptographic approximation to that ideal. Neither -/// exact uniformity, cryptographic security, nor worst-case `O(k)` is claimed. -/// State is constant size; no rings, permutation tables, or duplicate sets -/// are constructed. Walks have no artificial retry limit. +/// Under independent uniform ideal permutations per key and bit width, the +/// rooted order is uniform and sequential work is expected `O(k)` for `k` +/// outputs with constant-cost primitives. This is not a worst-case bound. +/// The finite 64-bit seeded, 8/16/24-round Feistel family is noncryptographic; +/// exact uniformity, independence and cryptographic security are not claimed. +/// State is four `u64` fields, with no ring, table, duplicate set or allocation. /// /// ``` /// use consistent_choose_k::VirtualPermutation; /// /// let seed = 0x1234_5678_9abc_def0; // normally a well-mixed hash of the key -/// let selection = VirtualPermutation::new(100, seed); -/// let third = selection.replica_at(2); -/// assert_eq!(selection.clone().nth(2), Some(third)); -/// let replicas: Vec<_> = selection.take(3).collect(); -/// assert_eq!(replicas.len(), 3); +/// let old: Vec<_> = VirtualPermutation::new(100, seed).collect(); +/// let new: Vec<_> = VirtualPermutation::new(101, seed) +/// .filter(|&node| node != 100).collect(); +/// assert_eq!(old, new); +/// assert_eq!(VirtualPermutation::new(100, seed).take(3).collect::>(), old[..3]); /// ``` #[derive(Clone, Debug)] pub struct VirtualPermutation { seed: u64, - n: u64, - next: u64, + count: u64, + cursor: u64, + remaining: u64, } impl VirtualPermutation { - /// Construct a permutation for `1..=u64::MAX` nodes. + /// Construct an ordering of `1..=u64::MAX - 1` real nodes. /// - /// As with [`crate::ConsistentPermutation::new`], supply a well-mixed - /// 64-bit hash of the key. Width/round domain separation is internal and - /// independent of `n` and of the requested replica count. + /// Supply a well-mixed key hash, held fixed across membership and replica + /// counts. Internal label zero is the sentinel; real node `i` has label + /// `i + 1`. The extra sentinel requires representable internal count `n + 1`. /// /// # Panics /// - /// Panics if `n == 0`. + /// Panics if `n == 0` or `n == u64::MAX`. pub fn new(n: u64, seed: u64) -> Self { assert!(n > 0, "n must be at least 1"); - Self { seed, n, next: 0 } + assert!(n < u64::MAX, "n must be at most u64::MAX - 1"); + Self { + seed, + count: n + 1, + cursor: 0, + remaining: n, + } } - /// Universe size, independent of the iterator's current position. + /// Number of real nodes, independent of iterator position. pub fn n(&self) -> u64 { - self.n - } - - /// Evaluate an absolute zero-based replica slot without advancing the - /// iterator. Expected constant work under ideal independent permutations; - /// no worst-case constant-time guarantee. - /// - /// # Panics - /// - /// Panics if `slot >= self.n()`. - pub fn replica_at(&self, slot: u64) -> u64 { - assert!(slot < self.n, "replica slot must be less than n"); - evaluate(self.n, slot, |bits| WordPermutation::new(self.seed, bits)) + self.count - 1 } } @@ -205,154 +210,73 @@ impl Iterator for VirtualPermutation { type Item = u64; fn next(&mut self) -> Option { - if self.next == self.n { + if self.remaining == 0 { return None; } - let value = self.replica_at(self.next); - self.next += 1; - Some(value) + self.cursor = evaluate(self.count, self.cursor, |bits| cycle(self.seed, bits)); + debug_assert!(self.cursor > 0 && self.cursor < self.count); + self.remaining -= 1; + Some(self.cursor - 1) } fn size_hint(&self) -> (usize, Option) { - match usize::try_from(self.n - self.next) { + match usize::try_from(self.remaining) { Ok(remaining) => (remaining, Some(remaining)), Err(_) => (usize::MAX, None), } } - - fn nth(&mut self, n: usize) -> Option { - self.next = self.next.saturating_add(n as u64).min(self.n); - self.next() - } } impl FusedIterator for VirtualPermutation {} -struct ExistingPermutation { - bits: u32, - seed: u64, -} - -impl Permutation for ExistingPermutation { - #[inline] - fn forward(&self, x: u64) -> u64 { - u64::from(layer_apply(self.bits, self.seed, x as u32)) - } +#[cfg(test)] +mod tests { + use super::*; + use std::cell::Cell; - #[inline] - fn inverse(&self, x: u64) -> u64 { - u64::from(layer_inverse(self.bits, self.seed, x as u32)) + struct CountedCycle<'a> { + cycle: SingleCycle, + calls: &'a Cell<[u64; 3]>, } -} -/// Experimental cycle-consistent replica slots using exactly the Feistel -/// network of [`crate::ConsistentPermutation`], including its round counts and -/// key schedule, with two bits per lift. -/// -/// This has the slot-consistency semantics of [`VirtualPermutation`], **not** -/// the survivor-list semantics of `ConsistentPermutation`. The supplied -/// well-mixed seed is passed unchanged to the existing network; no additional -/// width-domain seed mixer is inserted. Consequently its statistical quality -/// must be assessed separately from the independently keyed ideal model. -/// It is noncryptographic and does not promise exact uniformity. -/// -/// **Statistical caution:** the matched-network diagnostics show repeatable -/// small-domain bias. This variant is provided for comparison, not as a -/// statistically equivalent substitute for `VirtualPermutation`. -/// -/// State is allocation-free. The supported domain matches the existing -/// network: `1..=2^30` nodes. Iteration returns `u64`, as `VirtualPermutation` -/// does, and direct slot evaluation does not replay earlier slots. -/// -/// ``` -/// use consistent_choose_k::BalancedVirtualPermutation; -/// -/// let permutation = BalancedVirtualPermutation::new(100, 0x1234_5678_9abc_def0); -/// assert_eq!(permutation.clone().nth(2), Some(permutation.replica_at(2))); -/// assert_eq!(permutation.take(3).count(), 3); -/// ``` -#[derive(Clone, Debug)] -pub struct BalancedVirtualPermutation { - inner: VirtualPermutation, -} - -impl BalancedVirtualPermutation { - /// Construct an iterator with the existing Feistel network and seed. - /// - /// # Panics - /// - /// Panics unless `1 <= n <= 2^30`. - pub fn new(n: u32, seed: u64) -> Self { - assert!(n <= 1u32 << 30, "n must be at most 2^30"); - Self { - inner: VirtualPermutation::new(u64::from(n), seed), + impl Permutation for CountedCycle<'_> { + fn forward(&self, x: u64) -> u64 { + let mut v = self.calls.get(); + v[0] += 1; + self.calls.set(v); + self.cycle.forward(x) } - } - - /// Universe size, independent of the iterator's position. - pub fn n(&self) -> u64 { - self.inner.n() - } - - /// Evaluate an absolute slot without advancing the iterator. - /// - /// # Panics - /// - /// Panics if `slot >= self.n()`. - pub fn replica_at(&self, slot: u64) -> u64 { - assert!(slot < self.inner.n, "replica slot must be less than n"); - evaluate_with_stride::<2, _>(self.inner.n, slot, |bits| ExistingPermutation { - bits, - seed: self.inner.seed, - }) - } -} - -impl Iterator for BalancedVirtualPermutation { - type Item = u64; - - fn next(&mut self) -> Option { - if self.inner.next == self.inner.n { - return None; + fn inverse(&self, x: u64) -> u64 { + let mut v = self.calls.get(); + v[1] += 1; + self.calls.set(v); + self.cycle.inverse(x) } - let value = self.replica_at(self.inner.next); - self.inner.next += 1; - Some(value) - } - - fn size_hint(&self) -> (usize, Option) { - self.inner.size_hint() - } - - fn nth(&mut self, n: usize) -> Option { - self.inner.next = self.inner.next.saturating_add(n as u64).min(self.inner.n); - self.next() } -} - -impl FusedIterator for BalancedVirtualPermutation {} - -#[cfg(test)] -mod tests { - use super::*; fn seed(i: u64) -> u64 { mix(i.wrapping_add(WEYL)) } + fn successor(count: u64, key: u64, x: u64) -> u64 { + evaluate(count, x, |bits| cycle(key, bits)) + } + #[test] - fn word_round_trips_all_widths() { + fn ordinary_and_cycle_round_trips_all_widths() { for bits in 1..=64 { let mask = u64::MAX >> (64 - bits); for key in [0, 1, u64::MAX, seed(42)] { - let p = WordPermutation::new(key, bits); + let q = WordPermutation::new(key, bits); + let p = cycle(key, bits); for x in [0, 1, mask / 2, mask / 2 + 1, mask] .into_iter() - .chain((0..128).map(|i| seed(i) & mask)) + .chain((0..64).map(|i| seed(i) & mask)) { - let y = p.forward(x); - assert_eq!(y & !mask, 0); - assert_eq!(p.inverse(y), x, "bits={bits} key={key} x={x}"); + assert!(q.forward(x) <= mask && p.forward(x) <= mask); + assert_eq!(q.inverse(q.forward(x)), x); + assert_eq!(q.forward(q.inverse(x)), x); + assert_eq!(p.inverse(p.forward(x)), x); assert_eq!(p.forward(p.inverse(x)), x); } } @@ -360,153 +284,29 @@ mod tests { } #[test] - fn word_exhaustive_small_domains() { - for bits in 1..=10 { + fn conjugates_are_single_cycles() { + for bits in 1..=9 { for key in 0..16 { - let p = WordPermutation::new(seed(key), bits); - let mut outputs: Vec<_> = (0..1 << bits) - .map(|x| { - let y = p.forward(x); - assert_eq!(p.inverse(y), x); - y - }) - .collect(); - outputs.sort_unstable(); - assert_eq!(outputs, (0..1 << bits).collect::>()); - } - } - } - - #[test] - fn full_permutations_prefixes_and_membership() { - for key in 0..32 { - let key = seed(key); - let mut previous = vec![]; - for n in 1..=257 { - let permutation = VirtualPermutation::new(n, key); - let values: Vec<_> = permutation.clone().collect(); - let mut sorted = values.clone(); - sorted.sort_unstable(); - assert_eq!(sorted, (0..n).collect::>()); - for k in [0, 1, 2, 3, 8, n / 2, n] { - if k <= n { - assert_eq!( - permutation.clone().take(k as usize).collect::>(), - values[..k as usize] - ); - } + let p = cycle(seed(key), bits); + let mut seen = vec![false; 1 << bits]; + let mut x = 0; + for _ in 0..1 << bits { + assert!(!seen[x as usize]); + seen[x as usize] = true; + assert_eq!(p.inverse(p.forward(x)), x); + x = p.forward(x); } - let mut changed = 0; - for (r, old) in previous.iter().enumerate() { - assert_eq!(permutation.replica_at(r as u64), values[r]); - if *old != values[r] { - assert_eq!(values[r], n - 1); - changed += 1; - } - // Cycle-deleting the new node recovers every old slot. - let projected = if values[r] == n - 1 { - values[n as usize - 1] - } else { - values[r] - }; - assert_eq!(*old, projected); - } - assert!(changed <= 1); - previous = values; - } - } - } - - #[test] - fn large_domains_and_power_boundaries() { - for bits in 1..64 { - let half = 1u64 << bits; - for n in [half - 1, half, half + 1, u64::MAX - 1] { - for key in [0, 1, u64::MAX, seed(123)] { - let p = VirtualPermutation::new(n, key); - let larger = VirtualPermutation::new(n + 1, key); - for r in [0, n / 2, n - 1] { - let old = p.replica_at(r); - let new = larger.replica_at(r); - assert!(old < n && new <= n); - assert!(new == old || new == n); - assert_eq!(old, if new == n { larger.replica_at(n) } else { new }); - } - } - } - } - let mut p = VirtualPermutation::new(u64::MAX, seed(9)); - p.next = u64::MAX - 1; - assert!(p.next().is_some()); - assert_eq!(p.next(), None); - assert_eq!(p.next(), None); - } - - #[test] - fn iterator_boundaries() { - let mut p = VirtualPermutation::new(1, 0); - assert_eq!(p.n(), 1); - assert_eq!(p.size_hint(), (1, Some(1))); - assert_eq!(p.clone().take(0).count(), 0); - assert_eq!(p.next(), Some(0)); - assert_eq!(p.size_hint(), (0, Some(0))); - assert_eq!(p.next(), None); - assert_eq!(p.replica_at(0), 0); - assert_eq!(p.nth(usize::MAX), None); - let mut p = VirtualPermutation::new(100, seed(4)); - assert_eq!(p.nth(12), Some(p.replica_at(12))); - assert_eq!(p.next(), Some(p.replica_at(13))); - assert_eq!(p.nth(usize::MAX), None); - assert_eq!(p.next(), None); - } - - #[test] - #[should_panic(expected = "n must be at least 1")] - fn invalid_empty_domain() { - VirtualPermutation::new(0, 0); - } - - #[test] - #[should_panic(expected = "replica slot must be less than n")] - fn invalid_slot() { - VirtualPermutation::new(10, 0).replica_at(10); - } - - #[test] - fn long_cycles_are_not_truncated() { - use std::cell::Cell; - - struct Rotation<'a> { - mask: u64, - calls: &'a Cell, - } - impl Permutation for Rotation<'_> { - fn forward(&self, x: u64) -> u64 { - self.calls.set(self.calls.get() + 1); - x.wrapping_add(1) & self.mask - } - fn inverse(&self, x: u64) -> u64 { - self.calls.set(self.calls.get() + 1); - x.wrapping_sub(1) & self.mask + assert_eq!(x, 0); + assert!(seen.into_iter().all(|v| v)); } } - // Every dyadic lift of these ascending cycles is the same ascending - // cycle. A last-slot query just above a half boundary takes long walks. - let calls = Cell::new(0); - let result = evaluate(2049, 2048, |bits| Rotation { - mask: (1 << bits) - 1, - calls: &calls, - }); - assert_eq!(result, 0); - assert!(calls.get() > 4096); } - // Deliberately explicit tables only in the oracle, never in the evaluator. - fn project(p: &[u64], n: usize) -> Vec { - (0..n) + fn project(p: &[u64], count: usize) -> Vec { + (0..count) .map(|x| { let mut y = p[x]; - while y >= n as u64 { + while y >= count as u64 { y = p[y as usize]; } y @@ -514,14 +314,14 @@ mod tests { .collect() } - fn lift(lower: &[u64], q: &[u64]) -> Vec { + fn lift(lower: &[u64], p: &[u64]) -> Vec { let h = lower.len(); - let r = project(q, h); + let r = project(p, h); let mut inverse = vec![0; h]; for (x, &y) in r.iter().enumerate() { inverse[y as usize] = x; } - q.iter() + p.iter() .map(|&y| { if y < h as u64 { lower[inverse[y as usize]] @@ -532,133 +332,175 @@ mod tests { .collect() } + fn rooted_order(p: &[u64]) -> Vec { + let mut x = 0; + let mut order = vec![]; + for _ in 1..p.len() { + x = p[x as usize]; + assert_ne!(x, 0, "sentinel reached early"); + order.push(x - 1); + } + assert_eq!(p[x as usize], 0, "cycle does not close at sentinel"); + order + } + #[test] - fn matches_explicit_lift_and_cycle_projection() { - for key in 0..32 { + fn complete_orders_prefixes_and_survivor_restriction() { + for key in 0..16 { let key = seed(key); - let mut full = vec![0]; - for bits in 1..=8 { - let p = WordPermutation::new(key, bits); - let q: Vec<_> = (0..1 << bits).map(|x| p.forward(x)).collect(); - full = lift(&full, &q); - for n in (full.len() / 2 + 1)..=full.len() { - let actual: Vec<_> = VirtualPermutation::new(n as u64, key).collect(); - assert_eq!(actual, project(&full, n), "bits={bits} n={n}"); + let mut previous_order = vec![]; + let mut previous_map = vec![0]; + for n in 1..=257 { + let iter = VirtualPermutation::new(n, key); + let order: Vec<_> = iter.clone().collect(); + let mut sorted = order.clone(); + sorted.sort_unstable(); + assert_eq!(sorted, (0..n).collect::>()); + assert_eq!( + order + .iter() + .copied() + .filter(|&node| node < n - 1) + .collect::>(), + previous_order + ); + for k in [0, 1, 2, 3, 8, n / 2, n] { + if k <= n { + assert_eq!( + iter.clone().take(k as usize).collect::>(), + order[..k as usize] + ); + } + } + for rank in [0, n / 2, n - 1] { + let mut replay = iter.clone(); + assert_eq!(replay.nth(rank as usize), Some(order[rank as usize])); + assert_eq!(replay.next(), order.get(rank as usize + 1).copied()); } + let map: Vec<_> = (0..=n).map(|x| successor(n + 1, key, x)).collect(); + assert_eq!(rooted_order(&map), order); + assert_eq!(project(&map, n as usize), previous_map); + previous_map = map; + previous_order = order; } } } #[test] - fn balanced_permutations_match_explicit_quarter_lifts() { - for key in 0..32 { + fn matches_explicit_single_cycle_lifts() { + for key in 0..8 { let key = seed(key); let mut full = vec![0]; - for bits in (2..=8).step_by(2) { - let q: Vec<_> = (0..1 << bits) - .map(|x| u64::from(layer_apply(bits, key, x))) - .collect(); - full = lift(&full, &q); - for n in (full.len() / 4 + 1)..=full.len() { - let iter = BalancedVirtualPermutation::new(n as u32, key); - let actual: Vec<_> = iter.clone().collect(); - assert_eq!(actual, project(&full, n)); - let mut sorted = actual.clone(); - sorted.sort_unstable(); - assert_eq!(sorted, (0..n as u64).collect::>()); - for k in [0, 1, n / 2, n] { - assert_eq!(iter.clone().take(k).collect::>(), actual[..k]); - } - for (slot, &value) in actual.iter().enumerate() { - assert_eq!(iter.replica_at(slot as u64), value); - } - let smaller: Vec<_> = - BalancedVirtualPermutation::new(n as u32 - 1, key).collect(); - assert_eq!(project(&actual, n - 1), smaller); - let changed: Vec<_> = smaller - .iter() - .zip(&actual) - .filter(|(old, new)| old != new) + for bits in 1..=8 { + let p = cycle(key, bits); + let table: Vec<_> = (0..1 << bits).map(|x| p.forward(x)).collect(); + full = lift(&full, &table); + for count in full.len() / 2 + 1..=full.len() { + let projected = project(&full, count); + assert_eq!( + (0..count as u64) + .map(|x| successor(count as u64, key, x)) + .collect::>(), + projected + ); + assert_eq!( + VirtualPermutation::new(count as u64 - 1, key).collect::>(), + rooted_order(&projected) + ); + } + } + } + } + + #[test] + fn large_boundaries_and_sentinel_overflow() { + let mut sizes = vec![u64::MAX - 2, u64::MAX - 1]; + for bits in 1..64 { + let power = 1u64 << bits; + sizes.extend([power.saturating_sub(2).max(1), power - 1, power, power + 1]); + } + for n in sizes { + for key in [0, 1, u64::MAX, seed(42)] { + let k = n.min(16) as usize; + let values: Vec<_> = VirtualPermutation::new(n, key).take(k).collect(); + assert!(values.iter().all(|&v| v < n)); + let mut distinct = values.clone(); + distinct.sort_unstable(); + distinct.dedup(); + assert_eq!(distinct.len(), k); + if n < u64::MAX - 1 { + let restricted: Vec<_> = VirtualPermutation::new(n + 1, key) + .take(k + 1) + .filter(|&node| node < n) + .take(k) .collect(); - assert!(changed.len() <= 1); - assert!(changed.iter().all(|(_, new)| **new == n as u64 - 1)); + assert_eq!(values, restricted); } } } } #[test] - fn balanced_large_domains_and_iterator_boundaries() { - for bits in 1..30 { - for n in [(1u32 << bits) - 1, 1 << bits, (1 << bits) + 1] { - for key in [0, 1, u64::MAX, seed(42)] { - let p = BalancedVirtualPermutation::new(n, key); - let larger = BalancedVirtualPermutation::new(n + 1, key); - for slot in [0, u64::from(n / 2), u64::from(n - 1)] { - let old = p.replica_at(slot); - let new = larger.replica_at(slot); - assert!(old < u64::from(n)); - assert!(new == old || new == u64::from(n)); - assert_eq!( - old, - if new == u64::from(n) { - larger.replica_at(new) - } else { - new - } + fn sequential_level_subsequences_and_inverse_charging() { + for key in 0..8 { + for n in 1u64..=129 { + let bits = u64::BITS - n.leading_zeros(); + let calls: Vec<_> = (0..=bits).map(|_| Cell::new([0; 3])).collect(); + let mut selected = vec![0; bits as usize + 1]; + let mut cursor = 0; + for k in 1..=n { + cursor = evaluate(n + 1, cursor, |width| CountedCycle { + cycle: cycle(seed(key), width), + calls: &calls[width as usize], + }); + for width in 1..=bits { + let [forward, inverse, _] = calls[width as usize].get(); + assert!( + inverse <= forward, + "each upper chain is retraced at most once" ); + if width < bits { + selected[width as usize] += u64::from(cursor < 1 << width); + assert_eq!(forward, selected[width as usize]); + } else { + assert!(forward >= k && forward < 1 << bits); + } } } } } - let mut p = BalancedVirtualPermutation::new(1, 0); + } + + #[test] + fn iterator_boundaries() { + let mut p = VirtualPermutation::new(1, 0); assert_eq!(p.n(), 1); assert_eq!(p.size_hint(), (1, Some(1))); + assert_eq!(p.clone().take(0).count(), 0); assert_eq!(p.next(), Some(0)); + assert_eq!(successor(p.count, p.seed, p.cursor), 0); assert_eq!(p.size_hint(), (0, Some(0))); assert_eq!(p.next(), None); - assert_eq!(p.nth(usize::MAX), None); - let mut p = BalancedVirtualPermutation::new(1 << 30, seed(9)); - assert_eq!(p.nth((1 << 30) - 1), Some(p.replica_at((1 << 30) - 1))); assert_eq!(p.next(), None); - } - - #[test] - fn balanced_primary_matches_existing_at_full_powers_of_four() { - for bits in (0..=30).step_by(2) { - for key in 0..128 { - let key = seed(key); - let n = 1u32 << bits; - let expected = crate::ConsistentPermutation::new(n, key) - .next() - .expect("nonempty domain"); - assert_eq!( - BalancedVirtualPermutation::new(n, key).replica_at(0), - u64::from(expected) - ); - } - } + assert_eq!(p.nth(usize::MAX), None); + let p = VirtualPermutation::new(u64::MAX - 1, 0); + let expected = usize::try_from(u64::MAX - 1).ok(); + assert_eq!(p.size_hint(), (expected.unwrap_or(usize::MAX), expected)); } #[test] #[should_panic(expected = "n must be at least 1")] - fn balanced_invalid_empty_domain() { - BalancedVirtualPermutation::new(0, 0); - } - - #[test] - #[should_panic(expected = "n must be at most 2^30")] - fn balanced_invalid_large_domain() { - BalancedVirtualPermutation::new((1 << 30) + 1, 0); + fn invalid_empty_domain() { + VirtualPermutation::new(0, 0); } #[test] - #[should_panic(expected = "replica slot must be less than n")] - fn balanced_invalid_slot() { - BalancedVirtualPermutation::new(1, 0).replica_at(1); + #[should_panic(expected = "n must be at most u64::MAX - 1")] + fn invalid_sentinel_overflow() { + VirtualPermutation::new(u64::MAX, 0); } - fn permutations(n: usize) -> Vec> { + fn permutations(mut values: Vec) -> Vec> { fn visit(values: &mut [u64], start: usize, out: &mut Vec>) { if start == values.len() { out.push(values.to_vec()); @@ -671,82 +513,132 @@ mod tests { } } let mut out = vec![]; - visit(&mut (0..n as u64).collect::>(), 0, &mut out); + visit(&mut values, 0, &mut out); out } - #[test] - fn ideal_uniform_lift_fibers() { - use std::collections::BTreeMap; + fn all_cycles(count: usize) -> Vec> { + permutations((1..count as u64).collect()) + .into_iter() + .map(|order| { + let mut map = vec![0; count]; + let mut x = 0; + for y in order { + map[x] = y; + x = y as usize; + } + map + }) + .collect() + } - // Every ideal Q_1,Q_2 combination: F_4 has equal multiplicities. - let mut counts = BTreeMap::new(); - for lower in permutations(2) { - for q in permutations(4) { - *counts.entry(lift(&lower, &q)).or_insert(0) += 1; - } - } - assert_eq!(counts.len(), 24); - assert!(counts.values().all(|&count| count == 2)); - - // Fix F_4; every extension through each intermediate n occurs 4! times - // at n=8, and equally often for n=5,6,7 after cycle projection. - let lower = vec![2, 0, 3, 1]; - let mut counts: Vec, usize>> = (5..=8).map(|_| BTreeMap::new()).collect(); - for q in permutations(8) { - let full = lift(&lower, &q); - assert_eq!(project(&full, 4), lower); - for (i, n) in (5..=8).enumerate() { - *counts[i].entry(project(&full, n)).or_insert(0) += 1; - } + struct Table<'a>(&'a [u64]); + + impl Permutation for Table<'_> { + fn forward(&self, x: u64) -> u64 { + self.0[x as usize] } - for (i, count) in counts.iter().enumerate() { - let expected_distinct: usize = (5..=i + 5).product(); - assert_eq!(count.len(), expected_distinct); - assert!(count.values().all(|&v| v == 40320 / expected_distinct)); + fn inverse(&self, x: u64) -> u64 { + self.0.iter().position(|&y| y == x).expect("bijection") as u64 } } #[test] - #[ignore = "deterministic operation-count report; not a timing or statistical CI gate"] - fn operation_count_diagnostics() { - use std::{cell::Cell, rc::Rc}; + fn exhaustive_ideal_conjugators_and_rooted_orders() { + use std::collections::BTreeMap; - struct Counted

{ - permutation: P, - calls: Rc>, + let mut conjugates = BTreeMap::new(); + for q in permutations((0..4).collect()) { + let p = SingleCycle { + ordinary: Table(&q), + mask: 3, + }; + let table: Vec<_> = (0..4).map(|x| p.forward(x)).collect(); + *conjugates.entry(table).or_insert(0) += 1; } - impl

Counted

{ - fn new(permutation: P, calls: &Rc>) -> Self { - let mut counts = calls.get(); - counts[2] += 1; - calls.set(counts); - Self { - permutation, - calls: Rc::clone(calls), + assert_eq!(conjugates.len(), 6); + assert!(conjugates.values().all(|&v| v == 4)); + + let p2 = [1, 0]; + let cycles8 = all_cycles(8); + let mut counts: Vec, usize>> = (1..=7).map(|_| BTreeMap::new()).collect(); + for p4 in all_cycles(4) { + let f4 = lift(&p2, &p4); + for p8 in &cycles8 { + let f8 = lift(&f4, p8); + let mut previous = vec![]; + for n in 1..=7 { + let order = rooted_order(&project(&f8, n + 1)); + let mut actual = vec![]; + let mut cursor = 0; + for _ in 0..n { + cursor = evaluate(n as u64 + 1, cursor, |bits| { + Table(match bits { + 1 => &p2, + 2 => &p4, + 3 => p8, + _ => unreachable!(), + }) + }); + assert_ne!(cursor, 0); + actual.push(cursor - 1); + } + assert_eq!(actual, order); + assert_eq!( + order + .iter() + .copied() + .filter(|&v| v < n as u64 - 1) + .collect::>(), + previous + ); + previous = order.clone(); + *counts[n - 1].entry(order).or_insert(0) += 1; } } } - impl Permutation for Counted

{ + for (i, counts) in counts.iter().enumerate() { + let factorial: usize = (1..=i + 1).product(); + assert_eq!(counts.len(), factorial); + assert!(counts.values().all(|&v| v == 30240 / factorial)); + } + } + + #[test] + fn unlucky_sequential_walks_are_not_truncated() { + struct ReverseCycle<'a> { + mask: u64, + calls: &'a Cell, + } + impl Permutation for ReverseCycle<'_> { fn forward(&self, x: u64) -> u64 { - let mut calls = self.calls.get(); - calls[0] += 1; - self.calls.set(calls); - self.permutation.forward(x) + self.calls.set(self.calls.get() + 1); + x.wrapping_sub(1) & self.mask } fn inverse(&self, x: u64) -> u64 { - let mut calls = self.calls.get(); - calls[1] += 1; - self.calls.set(calls); - self.permutation.inverse(x) + self.calls.set(self.calls.get() + 1); + x.wrapping_add(1) & self.mask } } + let calls = Cell::new(0); + let factory = |bits| ReverseCycle { + mask: (1u64 << bits) - 1, + calls: &calls, + }; + let first = evaluate(2049, 0, factory); + assert_eq!(first, 2048); + assert_eq!(evaluate(2049, first, factory), 2047); + assert!(calls.get() > 4096); + } - println!( - "algorithm,n,slot,mean_forward,mean_inverse,mean_levels,p50_calls,p99_calls,max_calls" - ); + #[test] + #[ignore = "deterministic sequential operation-count report, not a timing CI gate"] + fn operation_count_diagnostics() { + println!("n,k,mean_forward,mean_inverse,mean_q_per_output,mean_levels,p50_calls,p99_calls,max_calls,max_step_calls"); for n in [ 1, + 2, + 3, 7, 8, 9, @@ -759,54 +651,57 @@ mod tests { 65535, 65536, 65537, - (1 << 30) - 1, 1 << 30, - (1 << 63) + 1, - u64::MAX, + (1 << 63) - 1, + u64::MAX - 1, ] { - for algorithm in ["virtual", "balanced"] { - if algorithm == "balanced" && n > 1 << 30 { - continue; - } - for slot in [0, n / 2, n - 1] { - let calls = Rc::new(Cell::new([0; 3])); - let mut totals = [0u64; 3]; - let mut samples = vec![]; - for key in 0..10_000 { - calls.set([0; 3]); - let result = if algorithm == "virtual" { - evaluate(n, slot, |bits| { - Counted::new(WordPermutation::new(seed(key), bits), &calls) - }) - } else { - evaluate_with_stride::<2, _>(n, slot, |bits| { - Counted::new( - ExistingPermutation { - bits, - seed: seed(key), - }, - &calls, - ) - }) - }; - assert!(result < n); - let counts = calls.get(); - for i in 0..3 { - totals[i] += counts[i]; - } - samples.push(counts[0] + counts[1]); + let mut ks = vec![1, 3, 16]; + if n <= 256 { + ks.extend([n / 4, n]); + } + ks.retain(|&k| k > 0 && k <= n); + ks.sort_unstable(); + ks.dedup(); + for k in ks { + let calls = Cell::new([0; 3]); + let mut totals = [0u64; 3]; + let mut samples = vec![]; + let mut max_step = 0; + for key in 0..10_000 { + calls.set([0; 3]); + let mut cursor = 0; + for _ in 0..k { + let before = calls.get()[0] + calls.get()[1]; + cursor = evaluate(n + 1, cursor, |bits| { + let mut v = calls.get(); + v[2] += 1; + calls.set(v); + CountedCycle { + cycle: cycle(seed(key), bits), + calls: &calls, + } + }); + assert!(cursor > 0 && cursor <= n); + max_step = max_step.max(calls.get()[0] + calls.get()[1] - before); } - samples.sort_unstable(); - println!( - "{algorithm},{n},{slot},{:.4},{:.4},{:.4},{},{},{}", - totals[0] as f64 / 10_000.0, - totals[1] as f64 / 10_000.0, - totals[2] as f64 / 10_000.0, - samples[4999], - samples[9899], - samples[9999], - ); + let v = calls.get(); + for i in 0..3 { + totals[i] += v[i]; + } + samples.push(v[0] + v[1]); } + samples.sort_unstable(); + println!( + "{n},{k},{:.4},{:.4},{:.4},{:.4},{},{},{},{}", + totals[0] as f64 / 10_000.0, + totals[1] as f64 / 10_000.0, + 2.0 * (totals[0] + totals[1]) as f64 / (10_000 * k) as f64, + totals[2] as f64 / 10_000.0, + samples[4999], + samples[9899], + samples[9999], + max_step + ); } } }