From 2d17ca942848cc2a1b016bcd928e3e741e627f18 Mon Sep 17 00:00:00 2001 From: Ray Walker Date: Fri, 2 Oct 2026 05:24:59 +1000 Subject: [PATCH] perf(bench): make hot_path compile by default, lazy, representative and profile-able (LAB-7039) - required-features = ["encryption"]: a plain cargo bench skips hot_path instead of failing to compile. - [profile.bench] keeps symbols (release strips them, which also dropped the line tables), so callgrind/perf resolve function names. - Fixtures, including the 64 MiB incompressible setup, build lazily inside the routine closure, so a filtered run pays only for the ids it selects. - New byte_storage/roundtrip_msgpack and roundtrip_incompressible corpora beside the original ids; each prints its LZ4 ratio once. - bench_throughput no longer prints a hard-coded machine label. --- Cargo.toml | 10 +++ README.md | 2 + benches/hot_path.rs | 168 ++++++++++++++++++++++++++++------- examples/bench_throughput.rs | 6 +- 4 files changed, 155 insertions(+), 31 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 4d898ad..4aee570 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -144,6 +144,16 @@ panic = "abort" strip = "symbols" debug = "line-tables-only" +# Bench profile: inherits release, but keeps symbols so callgrind/perf can +# attribute cost to functions. Release's strip = "symbols" also drops the +# line tables, which would make debug = "line-tables-only" a no-op here. +[profile.bench] +strip = false +debug = "line-tables-only" + +# hot_path also benches ZeroKnowledgeEncryptor, so it needs `encryption`; +# without it a plain `cargo bench` skips this target instead of failing to compile. [[bench]] name = "hot_path" harness = false +required-features = ["encryption"] diff --git a/README.md b/README.md index 443962e..8a21ec4 100644 --- a/README.md +++ b/README.md @@ -328,6 +328,8 @@ Benchmarks on Apple M2 Max (64KB payload, compressible data): > [!TIP] > Hardware acceleration is auto-detected. ARM64 uses ARM Crypto Extensions; x86-64 uses AES-NI. +The figures above come from `examples/bench_throughput.rs` on highly compressible data. The Criterion suite in `benches/hot_path.rs` needs the `encryption` feature (`make bench`, or `cargo bench --features encryption`; a plain `cargo bench` skips it). It runs the ByteStorage roundtrip on three corpora side by side: `byte_storage/roundtrip` (synthetic ramp, kept for history), `byte_storage/roundtrip_msgpack` (realistic msgpack records, about 0.38 LZ4 ratio at 64 KB) and `byte_storage/roundtrip_incompressible`. The bench profile keeps symbols, so callgrind and perf attribute cost to functions. + --- ## Testing diff --git a/benches/hot_path.rs b/benches/hot_path.rs index 6b99ea5..f15d3cf 100644 --- a/benches/hot_path.rs +++ b/benches/hot_path.rs @@ -1,17 +1,29 @@ //! Criterion benchmark suite for cachekit-core hot paths. //! //! Run with: `cargo bench -p cachekit-core --features encryption` -//! Output: `target/criterion//report/index.html` +//! (the target declares `required-features = ["encryption"]`, so a plain +//! `cargo bench` skips it). Output: `target/criterion//report/index.html` //! //! This is the PGO training workload — extend with new groups as hot //! paths are identified. Sizes chosen to span the realistic cache-payload //! distribution (64B keys, 1KB values, 64KB large objects). +//! +//! Every fixture is built lazily inside the routine's closure. Criterion only +//! calls a closure whose id matches the filter, so a filtered run pays only +//! for the ids it selects — which keeps per-id instruction counts (callgrind, +//! cachegrind) separable from process start. + +use std::cell::OnceCell; use cachekit_core::{ByteStorage, StorageEnvelope, ZeroKnowledgeEncryptor}; use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; +use serde::Serialize; const SIZES: &[usize] = &[64, 256, 1024, 4 * 1024, 16 * 1024, 64 * 1024]; +/// Synthetic 256-byte ramp. LZ4 compresses it almost perfectly, so its +/// figures flatter real payloads; kept so the original bench ids stay +/// comparable with their history. fn make_payload(size: usize) -> Vec { (0..size).map(|i| (i % 256) as u8).collect() } @@ -22,65 +34,154 @@ fn incompressible(len: usize) -> Vec { let mut state: u64 = 0x9e3779b97f4a7c15; let mut out = Vec::with_capacity(len + 8); while out.len() < len { - state ^= state << 13; - state ^= state >> 7; - state ^= state << 17; + state = xorshift(state); out.extend_from_slice(&state.wrapping_mul(0x2545f4914f6cdd1d).to_le_bytes()); } out.truncate(len); out } +fn xorshift(mut state: u64) -> u64 { + state ^= state << 13; + state ^= state >> 7; + state ^= state << 17; + state +} + +/// A typical cached row: what the SDKs hand ByteStorage after msgpack-encoding +/// a dict/object with string keys. +#[derive(Serialize)] +struct Record { + id: u64, + user: String, + email: String, + score: f64, + active: bool, + tags: Vec<&'static str>, + created_at: String, +} + +/// Realistic payload: a stream of msgpack-named records with deterministic, +/// varied values, truncated to `size`. ByteStorage treats the payload as +/// opaque bytes, so the cut tail does not affect what is measured. +fn msgpack_payload(size: usize) -> Vec { + const TAGS: &[&str] = &["free", "pro", "trial", "eu", "us", "beta", "admin"]; + let mut state: u64 = 0x2545f4914f6cdd1d; + let mut out = Vec::with_capacity(size + 256); + let mut id = 0u64; + while out.len() < size { + state = xorshift(state); + id += 1; + let record = Record { + id: 1_000_000 + id, + user: format!("user_{:x}", state % 0xff_ffff), + email: format!("u{}@example{}.com", state % 100_000, state % 7), + score: (state % 10_000) as f64 / 100.0, + active: state % 3 != 0, + tags: (0..(state % 4) as usize) + .map(|i| TAGS[(state as usize >> (8 * i)) % TAGS.len()]) + .collect(), + created_at: format!( + "2026-{:02}-{:02}T{:02}:{:02}:{:02}Z", + 1 + state % 12, + 1 + (state >> 8) % 28, + (state >> 16) % 24, + (state >> 24) % 60, + (state >> 32) % 60 + ), + }; + out.extend_from_slice(&rmp_serde::to_vec_named(&record).unwrap()); + } + out.truncate(size); + out +} + +/// Build a corpus payload and report its LZ4 compressed/raw ratio once, so +/// the synthetic and realistic corpora can be compared side by side. +fn corpus_fixture(storage: &ByteStorage, id: &str, payload: Vec) -> Vec { + let wire = storage.store(&payload, None).unwrap(); + let envelope: StorageEnvelope = rmp_serde::from_slice(&wire).unwrap(); + eprintln!( + "{id}: lz4 compressed/raw {} / {} B (ratio {:.4})", + envelope.compressed_data.len(), + payload.len(), + envelope.compressed_data.len() as f64 / payload.len() as f64 + ); + payload +} + /// Protocol 1.1 bin-encoding proof workload (LAB-764 / LAB-866): 64 MiB /// incompressible payload, where compressed_data dominates the envelope and /// the array-of-ints vs msgpack-bin difference is fully visible. Measures the /// envelope codec in isolation (rmp encode/decode of StorageEnvelope) and the /// full store()/retrieve() e2e paths. +struct Envelope64MiB { + payload: Vec, + wire: Vec, + envelope: StorageEnvelope, +} + fn bench_envelope_codec_64mib(c: &mut Criterion) { const SIZE: usize = 64 * 1024 * 1024; let storage = ByteStorage::new(None); - let payload = incompressible(SIZE); - let wire = storage.store(&payload, None).unwrap(); - let envelope: StorageEnvelope = rmp_serde::from_slice(&wire).unwrap(); - // Self-enforcing workload check: LZ4 overhead is positive on incompressible - // input, so a generator regression toward compressible data fails here - // instead of silently benchmarking the wrong workload. - assert!( - envelope.compressed_data.len() >= SIZE, - "bench payload must be incompressible" - ); - eprintln!( - "64mib_incompressible: compressed_data {} B, envelope wire {} B (ratio {:.4})", - envelope.compressed_data.len(), - wire.len(), - wire.len() as f64 / envelope.compressed_data.len() as f64 - ); + let fixture = OnceCell::new(); + let fixture = || { + fixture.get_or_init(|| { + let payload = incompressible(SIZE); + let wire = storage.store(&payload, None).unwrap(); + let envelope: StorageEnvelope = rmp_serde::from_slice(&wire).unwrap(); + // Self-enforcing workload check: LZ4 overhead is positive on incompressible + // input, so a generator regression toward compressible data fails here + // instead of silently benchmarking the wrong workload. + assert!( + envelope.compressed_data.len() >= SIZE, + "bench payload must be incompressible" + ); + eprintln!( + "64mib_incompressible: compressed_data {} B, envelope wire {} B (ratio {:.4})", + envelope.compressed_data.len(), + wire.len(), + wire.len() as f64 / envelope.compressed_data.len() as f64 + ); + Envelope64MiB { + payload, + wire, + envelope, + } + }) + }; let mut group = c.benchmark_group("byte_storage/64mib_incompressible"); group.sample_size(10); group.throughput(Throughput::Bytes(SIZE as u64)); group.bench_function("envelope_encode", |b| { - b.iter(|| black_box(rmp_serde::to_vec(black_box(&envelope)).unwrap())); + let f = fixture(); + b.iter(|| black_box(rmp_serde::to_vec(black_box(&f.envelope)).unwrap())); }); group.bench_function("envelope_decode", |b| { - b.iter(|| black_box(rmp_serde::from_slice::(black_box(&wire)).unwrap())); + let f = fixture(); + b.iter(|| black_box(rmp_serde::from_slice::(black_box(&f.wire)).unwrap())); }); group.bench_function("store_e2e", |b| { - b.iter(|| black_box(storage.store(black_box(&payload), None).unwrap())); + let f = fixture(); + b.iter(|| black_box(storage.store(black_box(&f.payload), None).unwrap())); }); group.bench_function("retrieve_e2e", |b| { - b.iter(|| black_box(storage.retrieve(black_box(&wire)).unwrap())); + let f = fixture(); + b.iter(|| black_box(storage.retrieve(black_box(&f.wire)).unwrap())); }); group.finish(); } -fn bench_byte_storage_roundtrip(c: &mut Criterion) { +fn bench_roundtrip_corpus(c: &mut Criterion, name: &str, corpus: fn(usize) -> Vec) { let storage = ByteStorage::new(None); - let mut group = c.benchmark_group("byte_storage/roundtrip"); + let mut group = c.benchmark_group(name); for &size in SIZES { - let data = make_payload(size); group.throughput(Throughput::Bytes(size as u64)); - group.bench_with_input(BenchmarkId::from_parameter(size), &data, |b, data| { + let data = OnceCell::new(); + group.bench_function(BenchmarkId::from_parameter(size), |b| { + let data = data + .get_or_init(|| corpus_fixture(&storage, &format!("{name}/{size}"), corpus(size))); b.iter(|| { let envelope = storage.store(black_box(data), None).unwrap(); let (out, _fmt) = storage.retrieve(black_box(&envelope)).unwrap(); @@ -91,15 +192,22 @@ fn bench_byte_storage_roundtrip(c: &mut Criterion) { group.finish(); } +fn bench_byte_storage_roundtrip(c: &mut Criterion) { + bench_roundtrip_corpus(c, "byte_storage/roundtrip", make_payload); + bench_roundtrip_corpus(c, "byte_storage/roundtrip_msgpack", msgpack_payload); + bench_roundtrip_corpus(c, "byte_storage/roundtrip_incompressible", incompressible); +} + fn bench_encrypt_decrypt(c: &mut Criterion) { let encryptor = ZeroKnowledgeEncryptor::new().unwrap(); let key = [0x42u8; 32]; let aad = b"bench-aad"; let mut group = c.benchmark_group("encryption/aes_gcm_roundtrip"); for &size in SIZES { - let plaintext = make_payload(size); group.throughput(Throughput::Bytes(size as u64)); - group.bench_with_input(BenchmarkId::from_parameter(size), &plaintext, |b, pt| { + let plaintext = OnceCell::new(); + group.bench_function(BenchmarkId::from_parameter(size), |b| { + let pt = plaintext.get_or_init(|| make_payload(size)); b.iter(|| { let ct = encryptor.encrypt_aes_gcm(black_box(pt), &key, aad).unwrap(); let pt2 = encryptor diff --git a/examples/bench_throughput.rs b/examples/bench_throughput.rs index 840be7e..4d79c48 100644 --- a/examples/bench_throughput.rs +++ b/examples/bench_throughput.rs @@ -102,7 +102,11 @@ fn bench_size(size: usize, iterations: usize) { } fn main() { - println!("Throughput benchmark on Apple M2 Max\n"); + println!( + "Throughput benchmark ({}/{})\n", + std::env::consts::OS, + std::env::consts::ARCH + ); // Small data (call overhead visible) bench_size(1024, 100_000); // 1KB