Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -144,6 +144,16 @@ panic = "abort"
strip = "symbols"
debug = "line-tables-only"

# Bench profile: inherits release, but keeps symbols so callgrind/perf can
# attribute cost to functions. Release's strip = "symbols" also drops the
# line tables, which would make debug = "line-tables-only" a no-op here.
[profile.bench]
strip = false
debug = "line-tables-only"

# hot_path also benches ZeroKnowledgeEncryptor, so it needs `encryption`;
# without it a plain `cargo bench` skips this target instead of failing to compile.
[[bench]]
name = "hot_path"
harness = false
required-features = ["encryption"]
2 changes: 2 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -328,6 +328,8 @@ Benchmarks on Apple M2 Max (64KB payload, compressible data):
> [!TIP]
> Hardware acceleration is auto-detected. ARM64 uses ARM Crypto Extensions; x86-64 uses AES-NI.

The figures above come from `examples/bench_throughput.rs` on highly compressible data. The Criterion suite in `benches/hot_path.rs` needs the `encryption` feature (`make bench`, or `cargo bench --features encryption`; a plain `cargo bench` skips it). It runs the ByteStorage roundtrip on three corpora side by side: `byte_storage/roundtrip` (synthetic ramp, kept for history), `byte_storage/roundtrip_msgpack` (realistic msgpack records, about 0.38 LZ4 ratio at 64 KB) and `byte_storage/roundtrip_incompressible`. The bench profile keeps symbols, so callgrind and perf attribute cost to functions.

---

## Testing
Expand Down
168 changes: 138 additions & 30 deletions benches/hot_path.rs
Original file line number Diff line number Diff line change
@@ -1,17 +1,29 @@
//! Criterion benchmark suite for cachekit-core hot paths.
//!
//! Run with: `cargo bench -p cachekit-core --features encryption`
//! Output: `target/criterion/<bench_id>/report/index.html`
//! (the target declares `required-features = ["encryption"]`, so a plain
//! `cargo bench` skips it). Output: `target/criterion/<bench_id>/report/index.html`
//!
//! This is the PGO training workload — extend with new groups as hot
//! paths are identified. Sizes chosen to span the realistic cache-payload
//! distribution (64B keys, 1KB values, 64KB large objects).
//!
//! Every fixture is built lazily inside the routine's closure. Criterion only
//! calls a closure whose id matches the filter, so a filtered run pays only
//! for the ids it selects — which keeps per-id instruction counts (callgrind,
//! cachegrind) separable from process start.

use std::cell::OnceCell;

use cachekit_core::{ByteStorage, StorageEnvelope, ZeroKnowledgeEncryptor};
use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput};
use serde::Serialize;

const SIZES: &[usize] = &[64, 256, 1024, 4 * 1024, 16 * 1024, 64 * 1024];

/// Synthetic 256-byte ramp. LZ4 compresses it almost perfectly, so its
/// figures flatter real payloads; kept so the original bench ids stay
/// comparable with their history.
fn make_payload(size: usize) -> Vec<u8> {
(0..size).map(|i| (i % 256) as u8).collect()
}
Expand All @@ -22,65 +34,154 @@ fn incompressible(len: usize) -> Vec<u8> {
let mut state: u64 = 0x9e3779b97f4a7c15;
let mut out = Vec::with_capacity(len + 8);
while out.len() < len {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state = xorshift(state);
out.extend_from_slice(&state.wrapping_mul(0x2545f4914f6cdd1d).to_le_bytes());
}
out.truncate(len);
out
}

fn xorshift(mut state: u64) -> u64 {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
}

/// A typical cached row: what the SDKs hand ByteStorage after msgpack-encoding
/// a dict/object with string keys.
#[derive(Serialize)]
struct Record {
id: u64,
user: String,
email: String,
score: f64,
active: bool,
tags: Vec<&'static str>,
created_at: String,
}

/// Realistic payload: a stream of msgpack-named records with deterministic,
/// varied values, truncated to `size`. ByteStorage treats the payload as
/// opaque bytes, so the cut tail does not affect what is measured.
fn msgpack_payload(size: usize) -> Vec<u8> {
const TAGS: &[&str] = &["free", "pro", "trial", "eu", "us", "beta", "admin"];
let mut state: u64 = 0x2545f4914f6cdd1d;
let mut out = Vec::with_capacity(size + 256);
let mut id = 0u64;
while out.len() < size {
state = xorshift(state);
id += 1;
let record = Record {
id: 1_000_000 + id,
user: format!("user_{:x}", state % 0xff_ffff),
email: format!("u{}@example{}.com", state % 100_000, state % 7),
score: (state % 10_000) as f64 / 100.0,
active: state % 3 != 0,
tags: (0..(state % 4) as usize)
.map(|i| TAGS[(state as usize >> (8 * i)) % TAGS.len()])
.collect(),
created_at: format!(
"2026-{:02}-{:02}T{:02}:{:02}:{:02}Z",
1 + state % 12,
1 + (state >> 8) % 28,
(state >> 16) % 24,
(state >> 24) % 60,
(state >> 32) % 60
),
};
out.extend_from_slice(&rmp_serde::to_vec_named(&record).unwrap());
}
out.truncate(size);
out
}

/// Build a corpus payload and report its LZ4 compressed/raw ratio once, so
/// the synthetic and realistic corpora can be compared side by side.
fn corpus_fixture(storage: &ByteStorage, id: &str, payload: Vec<u8>) -> Vec<u8> {
let wire = storage.store(&payload, None).unwrap();
let envelope: StorageEnvelope = rmp_serde::from_slice(&wire).unwrap();
eprintln!(
"{id}: lz4 compressed/raw {} / {} B (ratio {:.4})",
envelope.compressed_data.len(),
payload.len(),
envelope.compressed_data.len() as f64 / payload.len() as f64
);
payload
}

/// Protocol 1.1 bin-encoding proof workload (LAB-764 / LAB-866): 64 MiB
/// incompressible payload, where compressed_data dominates the envelope and
/// the array-of-ints vs msgpack-bin difference is fully visible. Measures the
/// envelope codec in isolation (rmp encode/decode of StorageEnvelope) and the
/// full store()/retrieve() e2e paths.
struct Envelope64MiB {
payload: Vec<u8>,
wire: Vec<u8>,
envelope: StorageEnvelope,
}

fn bench_envelope_codec_64mib(c: &mut Criterion) {
const SIZE: usize = 64 * 1024 * 1024;
let storage = ByteStorage::new(None);
let payload = incompressible(SIZE);
let wire = storage.store(&payload, None).unwrap();
let envelope: StorageEnvelope = rmp_serde::from_slice(&wire).unwrap();
// Self-enforcing workload check: LZ4 overhead is positive on incompressible
// input, so a generator regression toward compressible data fails here
// instead of silently benchmarking the wrong workload.
assert!(
envelope.compressed_data.len() >= SIZE,
"bench payload must be incompressible"
);
eprintln!(
"64mib_incompressible: compressed_data {} B, envelope wire {} B (ratio {:.4})",
envelope.compressed_data.len(),
wire.len(),
wire.len() as f64 / envelope.compressed_data.len() as f64
);
let fixture = OnceCell::new();
let fixture = || {
fixture.get_or_init(|| {
let payload = incompressible(SIZE);
let wire = storage.store(&payload, None).unwrap();
let envelope: StorageEnvelope = rmp_serde::from_slice(&wire).unwrap();
// Self-enforcing workload check: LZ4 overhead is positive on incompressible
// input, so a generator regression toward compressible data fails here
// instead of silently benchmarking the wrong workload.
assert!(
envelope.compressed_data.len() >= SIZE,
"bench payload must be incompressible"
);
eprintln!(
"64mib_incompressible: compressed_data {} B, envelope wire {} B (ratio {:.4})",
envelope.compressed_data.len(),
wire.len(),
wire.len() as f64 / envelope.compressed_data.len() as f64
);
Envelope64MiB {
payload,
wire,
envelope,
}
})
};

let mut group = c.benchmark_group("byte_storage/64mib_incompressible");
group.sample_size(10);
group.throughput(Throughput::Bytes(SIZE as u64));
group.bench_function("envelope_encode", |b| {
b.iter(|| black_box(rmp_serde::to_vec(black_box(&envelope)).unwrap()));
let f = fixture();
b.iter(|| black_box(rmp_serde::to_vec(black_box(&f.envelope)).unwrap()));
});
group.bench_function("envelope_decode", |b| {
b.iter(|| black_box(rmp_serde::from_slice::<StorageEnvelope>(black_box(&wire)).unwrap()));
let f = fixture();
b.iter(|| black_box(rmp_serde::from_slice::<StorageEnvelope>(black_box(&f.wire)).unwrap()));
});
group.bench_function("store_e2e", |b| {
b.iter(|| black_box(storage.store(black_box(&payload), None).unwrap()));
let f = fixture();
b.iter(|| black_box(storage.store(black_box(&f.payload), None).unwrap()));
});
group.bench_function("retrieve_e2e", |b| {
b.iter(|| black_box(storage.retrieve(black_box(&wire)).unwrap()));
let f = fixture();
b.iter(|| black_box(storage.retrieve(black_box(&f.wire)).unwrap()));
});
group.finish();
}

fn bench_byte_storage_roundtrip(c: &mut Criterion) {
fn bench_roundtrip_corpus(c: &mut Criterion, name: &str, corpus: fn(usize) -> Vec<u8>) {
let storage = ByteStorage::new(None);
let mut group = c.benchmark_group("byte_storage/roundtrip");
let mut group = c.benchmark_group(name);
for &size in SIZES {
let data = make_payload(size);
group.throughput(Throughput::Bytes(size as u64));
group.bench_with_input(BenchmarkId::from_parameter(size), &data, |b, data| {
let data = OnceCell::new();
group.bench_function(BenchmarkId::from_parameter(size), |b| {
let data = data
.get_or_init(|| corpus_fixture(&storage, &format!("{name}/{size}"), corpus(size)));
b.iter(|| {
let envelope = storage.store(black_box(data), None).unwrap();
let (out, _fmt) = storage.retrieve(black_box(&envelope)).unwrap();
Expand All @@ -91,15 +192,22 @@ fn bench_byte_storage_roundtrip(c: &mut Criterion) {
group.finish();
}

fn bench_byte_storage_roundtrip(c: &mut Criterion) {
bench_roundtrip_corpus(c, "byte_storage/roundtrip", make_payload);
bench_roundtrip_corpus(c, "byte_storage/roundtrip_msgpack", msgpack_payload);
bench_roundtrip_corpus(c, "byte_storage/roundtrip_incompressible", incompressible);
}

fn bench_encrypt_decrypt(c: &mut Criterion) {
let encryptor = ZeroKnowledgeEncryptor::new().unwrap();
let key = [0x42u8; 32];
let aad = b"bench-aad";
let mut group = c.benchmark_group("encryption/aes_gcm_roundtrip");
for &size in SIZES {
let plaintext = make_payload(size);
group.throughput(Throughput::Bytes(size as u64));
group.bench_with_input(BenchmarkId::from_parameter(size), &plaintext, |b, pt| {
let plaintext = OnceCell::new();
group.bench_function(BenchmarkId::from_parameter(size), |b| {
let pt = plaintext.get_or_init(|| make_payload(size));
b.iter(|| {
let ct = encryptor.encrypt_aes_gcm(black_box(pt), &key, aad).unwrap();
let pt2 = encryptor
Expand Down
6 changes: 5 additions & 1 deletion examples/bench_throughput.rs
Original file line number Diff line number Diff line change
Expand Up @@ -102,7 +102,11 @@ fn bench_size(size: usize, iterations: usize) {
}

fn main() {
println!("Throughput benchmark on Apple M2 Max\n");
println!(
"Throughput benchmark ({}/{})\n",
std::env::consts::OS,
std::env::consts::ARCH
);

// Small data (call overhead visible)
bench_size(1024, 100_000); // 1KB
Expand Down
Loading