diff --git a/CHANGELOG.md b/CHANGELOG.md index f259c5d2..daa60a7f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -624,6 +624,23 @@ after its public API and format compatibility policies are established. ### Fixed +- Benchmark report admission refuses identical or conflicting repeated metadata + coordinates with a typed failure naming the coordinate (issue #142). Complete + canonical report admission remains in progress. Ordered metadata keys, exact + headers/catalogs, complete metric widths and canonical unsigned decimals now + refuse malformed reports while admitting the committed historical fixture. + Fixed measurement policies and consistent per-row sample counts are admitted + before publication. Ratios, throughput, percentile order and reused-chunk + bounds now refuse inconsistent evidence with typed expected/observed failures; + arithmetic and numeric parsing refuse without approximation. Byte/count + metrics admit only their portable unsigned 64-bit range; timing and throughput + retain unsigned 128-bit precision. Publication requires an immutable admitted + report; refusal preserves the prior artifact and retained recovery stage. + Direct parser admission enforces the same one-MiB input ceiling as subprocess + capture before decoding, and preserves UTF-8 error sources. A dedicated + I/O-free fuzz facade exercises the production parser with a deterministic + historical seed; seed preparation and campaign discovery include the target. + - The benchmark workload catalog describes current range authentication once before output, matching its single-pass metric definitions (#71). diff --git a/fuzz/Cargo.toml b/fuzz/Cargo.toml index 88feb486..c8779611 100644 --- a/fuzz/Cargo.toml +++ b/fuzz/Cargo.toml @@ -12,7 +12,7 @@ cargo-fuzz = true [dependencies] keep = { path = "..", version = "=0.0.0" } libfuzzer-sys = { version = "=0.4.13", default-features = false, features = ["link_libfuzzer"] } -xtask = { path = "../xtask", version = "=0.0.0", default-features = false, features = ["golden-protocol-fuzz", "repository-json-fuzz"] } +xtask = { path = "../xtask", version = "=0.0.0", default-features = false, features = ["golden-protocol-fuzz", "repository-json-fuzz", "benchmark-report-fuzz"] } [lints.rust] warnings = "deny" @@ -148,3 +148,10 @@ bench = false [workspace] members = ["."] resolver = "3" + +[[bin]] +name = "benchmark_report" +path = "fuzz_targets/benchmark_report.rs" +test = false +doc = false +bench = false diff --git a/fuzz/README.md b/fuzz/README.md index 115dc98f..fcccd99e 100644 --- a/fuzz/README.md +++ b/fuzz/README.md @@ -115,3 +115,20 @@ for the finite periods declared in `campaign.env`. A confirmed crash, timeout, or out-of-memory input must be minimized and promoted into a committed deterministic regression test; an artifact or cache alone never closes the defect. + +## Benchmark report admission + +The `benchmark_report` target calls the production report parser through the +`benchmark-report-fuzz` feature, with task execution and host capture disabled. +Its coordinates are fixed to the committed historical baseline. Seed +preparation materializes that complete canonical report under the derived, +ignored corpus; no host measurement or new performance baseline is implied. +The facade performs no filesystem, process, clock or network operations. + +The parser enforces a one-MiB input ceiling and preserves typed encoding, +canonical row, metadata, counter-width and arithmetic refusals. Unit laws +verify that the seed reaches admission and corrupted bytes refuse; these +prevent an always-refusing facade from passing solely because it never panics. +Run this target with the same pinned nightly and resource bounds as the other +targets. Campaign success is bounded exploration evidence, not proof that all +malformed reports have been enumerated. diff --git a/fuzz/fuzz_targets/benchmark_report.rs b/fuzz/fuzz_targets/benchmark_report.rs new file mode 100644 index 00000000..2d5c9560 --- /dev/null +++ b/fuzz/fuzz_targets/benchmark_report.rs @@ -0,0 +1,10 @@ +#![no_main] + +//! This target owns bounded canonical benchmark-report admission fuzzing. + +use libfuzzer_sys::fuzz_target; +use xtask::admit_benchmark_report; + +fuzz_target!(|bytes: &[u8]| { + let _ = admit_benchmark_report(bytes); +}); diff --git a/xtask/Cargo.toml b/xtask/Cargo.toml index a1089678..1c32754c 100644 --- a/xtask/Cargo.toml +++ b/xtask/Cargo.toml @@ -9,6 +9,7 @@ publish = false [features] default = ["repository-tasks"] golden-protocol-fuzz = [] +benchmark-report-fuzz = [] repository-json-fuzz = ["dep:serde", "dep:serde_json"] repository-tasks = [ "dep:blake3", diff --git a/xtask/src/benchmark_baseline/artifact.rs b/xtask/src/benchmark_baseline/artifact.rs index 445a9969..8a841427 100644 --- a/xtask/src/benchmark_baseline/artifact.rs +++ b/xtask/src/benchmark_baseline/artifact.rs @@ -3,17 +3,27 @@ use super::BenchmarkBaselineError; use super::environment::CapturedEnvironment; -pub(super) fn validate( - bytes: &[u8], +/// Immutable report bytes admitted against captured measurement coordinates. +#[must_use] +pub(super) struct AdmittedReport<'a> { + bytes: &'a [u8], +} + +impl<'a> AdmittedReport<'a> { + pub(super) const fn bytes(&self) -> &'a [u8] { + self.bytes + } +} + +pub(super) fn validate<'a>( + bytes: &'a [u8], environment: &CapturedEnvironment, -) -> Result<(), BenchmarkBaselineError> { - let report = - std::str::from_utf8(bytes).map_err(|_source| BenchmarkBaselineError::ReportViolation { - reason: "report-is-not-utf8", - })?; +) -> Result, BenchmarkBaselineError> { + let report = super::report_input::decode(bytes)?; if report.contains('\r') || !report.ends_with('\n') { return violation("report-line-framing"); } + super::metadata_uniqueness::admit(report)?; let mut lines = report.lines(); if lines.next() != Some("schema\tkeep.streaming-cas-baseline/v1") { return violation("report-schema"); @@ -86,7 +96,8 @@ pub(super) fn validate( { return violation("report-profile-count"); } - Ok(()) + super::report_grammar::admit(report)?; + Ok(AdmittedReport { bytes }) } fn require_line( diff --git a/xtask/src/benchmark_baseline/artifact_publication.rs b/xtask/src/benchmark_baseline/artifact_publication.rs index fdb3043d..87959ad3 100644 --- a/xtask/src/benchmark_baseline/artifact_publication.rs +++ b/xtask/src/benchmark_baseline/artifact_publication.rs @@ -10,7 +10,11 @@ const LOCK_NAME: &str = ".streaming-cas-baseline-v1.lock"; const OUTPUT_RELATIVE_PATH: &str = "target/benchmark/streaming-cas-baseline-v1.tsv"; const STAGE_NAME: &str = ".streaming-cas-baseline-v1.tsv.stage"; -pub(super) fn persist(repository_root: &Path, bytes: &[u8]) -> Result<(), BenchmarkBaselineError> { +pub(super) fn persist( + repository_root: &Path, + report: &super::artifact::AdmittedReport<'_>, +) -> Result<(), BenchmarkBaselineError> { + let bytes = report.bytes(); let output = repository_root.join(OUTPUT_RELATIVE_PATH); let parent = output .parent() diff --git a/xtask/src/benchmark_baseline/artifact_publication_tests.rs b/xtask/src/benchmark_baseline/artifact_publication_tests.rs index 1b2b0495..92588667 100644 --- a/xtask/src/benchmark_baseline/artifact_publication_tests.rs +++ b/xtask/src/benchmark_baseline/artifact_publication_tests.rs @@ -14,9 +14,11 @@ fn interrupted_baseline_stage_is_recovered_before_publication() -> Result<(), Bo let stage = parent.join(STAGE_NAME); fs::write(&stage, b"abandoned")?; - persist(directory.path(), b"exact report\n")?; + let (bytes, environment) = report_fixture()?; + let report = super::super::artifact::validate(bytes.as_bytes(), &environment)?; + persist(directory.path(), &report)?; - assert_eq!(fs::read(&output)?, b"exact report\n"); + assert_eq!(fs::read(&output)?, bytes.as_bytes()); assert!(!stage.exists()); directory.close()?; Ok(()) @@ -28,8 +30,10 @@ fn failed_baseline_publication_removes_its_stage() -> Result<(), Box> let (output, parent) = output_paths(&directory)?; fs::create_dir(&output)?; + let (bytes, environment) = report_fixture()?; + let report = super::super::artifact::validate(bytes.as_bytes(), &environment)?; assert!(matches!( - persist(directory.path(), b"exact report\n"), + persist(directory.path(), &report), Err(BenchmarkBaselineError::Io { action: "publish report", .. @@ -52,8 +56,10 @@ fn concurrent_baseline_publishers_are_refused() -> Result<(), Box> { .open(parent.join(LOCK_NAME))?; lock.try_lock()?; + let (bytes, environment) = report_fixture()?; + let report = super::super::artifact::validate(bytes.as_bytes(), &environment)?; assert!(matches!( - persist(directory.path(), b"exact report\n"), + persist(directory.path(), &report), Err(BenchmarkBaselineError::ReportViolation { reason: "benchmark-publication-already-active" }) @@ -71,3 +77,54 @@ fn output_paths( fs::create_dir_all(&parent)?; Ok((output, parent)) } + +#[test] +fn refused_report_preserves_prior_artifact_and_interrupted_stage() -> Result<(), Box> { + let directory = TestDirectory::create("benchmark-admission-preservation")?; + let (output, parent) = output_paths(&directory)?; + fs::write(&output, b"previous artifact\n")?; + let stage = parent.join(STAGE_NAME); + fs::write(&stage, b"interrupted stage\n")?; + let (bytes, environment) = report_fixture()?; + let malformed = + format!("{bytes}metadata\tgit-commit\tffffffffffffffffffffffffffffffffffffffff\n"); + assert!( + matches!(super::super::artifact::validate(malformed.as_bytes(), &environment) + .and_then(|report| persist(directory.path(), &report)), + Err(BenchmarkBaselineError::DuplicateReportMetadata { coordinate }) + if coordinate == "git-commit") + ); + assert_eq!(fs::read(&output)?, b"previous artifact\n"); + assert_eq!(fs::read(&stage)?, b"interrupted stage\n"); + assert!(!parent.join(LOCK_NAME).exists()); + directory.close()?; + Ok(()) +} + +fn report_fixture() -> Result< + ( + String, + crate::benchmark_baseline::environment::CapturedEnvironment, + ), + Box, +> { + use crate::benchmark_baseline::environment::CapturedEnvironment; + use crate::benchmark_baseline::host_environment::CapturedHost; + let environment = CapturedEnvironment { + commit: String::from("c529c07f385b5bcd76a4e57c1987001d496f9135"), + tree: "clean", + rustc_version: String::from("rustc 1.96.0 (ac68faa20 2026-05-25)"), + target_triple: String::from("aarch64-apple-darwin"), + host: CapturedHost { + os_description: String::from("Darwin 25.3.0 arm64"), + cpu_model: String::from("Apple M1 Pro"), + logical_cpu_count: std::num::NonZeroUsize::new(10).ok_or("invalid CPU fixture")?, + }, + }; + Ok(( + String::from(include_str!( + "../../../benchmark/baselines/c529c07-aarch64-apple-darwin.tsv" + )), + environment, + )) +} diff --git a/xtask/src/benchmark_baseline/captured_environment.rs b/xtask/src/benchmark_baseline/captured_environment.rs new file mode 100644 index 00000000..510da802 --- /dev/null +++ b/xtask/src/benchmark_baseline/captured_environment.rs @@ -0,0 +1,19 @@ +//! This module owns captured source, compiler and host measurement coordinates. + +use std::num::NonZeroUsize; + +#[derive(Eq, PartialEq)] +pub(super) struct CapturedEnvironment { + pub(super) commit: String, + pub(super) tree: &'static str, + pub(super) rustc_version: String, + pub(super) target_triple: String, + pub(super) host: CapturedHost, +} + +#[derive(Eq, PartialEq)] +pub(super) struct CapturedHost { + pub(super) os_description: String, + pub(super) cpu_model: String, + pub(super) logical_cpu_count: NonZeroUsize, +} diff --git a/xtask/src/benchmark_baseline/catalog_membership_tests.rs b/xtask/src/benchmark_baseline/catalog_membership_tests.rs new file mode 100644 index 00000000..dee507d8 --- /dev/null +++ b/xtask/src/benchmark_baseline/catalog_membership_tests.rs @@ -0,0 +1,39 @@ +//! Model laws for exact closed catalog membership without replacement aliases. + +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn unknown_and_duplicate_catalog_members_refuse_at_the_exact_slot() +-> Result<(), Box> { + let environment = environment(); + let valid = report(&environment, 13, 5); + for row in valid + .lines() + .filter(|line| line.starts_with("scenario\t") || line.starts_with("profile\t")) + { + let mut fields = row.split('\t'); + let kind = fields.next().ok_or("missing catalog kind")?; + let name = fields.next().ok_or("missing catalog name")?; + let width = if kind == "scenario" { 3 } else { 7 }; + let expected = row.split('\t').take(width).collect::>().join("\t"); + let unknown = row.replacen( + &format!("{kind}\t{name}\t"), + &format!("{kind}\tunknown\t"), + 1, + ); + let duplicate = valid + .lines() + .find(|candidate| candidate.starts_with(&format!("{kind}\t")) && *candidate != row) + .ok_or("missing duplicate fixture")?; + for observed in [unknown.as_str(), duplicate] { + let malformed = valid.replacen(&format!("{row}\n"), &format!("{observed}\n"), 1); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::InvalidReportRow { + expected: actual_expected, observed: actual_observed, + }) if actual_expected == expected && actual_observed == observed) + ); + } + } + Ok(()) +} diff --git a/xtask/src/benchmark_baseline/completeness_tests.rs b/xtask/src/benchmark_baseline/completeness_tests.rs new file mode 100644 index 00000000..5f5a4710 --- /dev/null +++ b/xtask/src/benchmark_baseline/completeness_tests.rs @@ -0,0 +1,78 @@ +//! Model laws for required metadata and complete metric row widths. + +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn missing_and_unknown_metadata_refuse_the_exact_required_coordinate() +-> Result<(), Box> { + let environment = environment(); + let valid = report(&environment, 13, 5); + for row in valid.lines().filter(|line| line.starts_with("metadata\t")) { + let key = row.split('\t').nth(1).ok_or("missing fixture key")?; + let unknown = row.replacen(&format!("metadata\t{key}\t"), "metadata\tunknown\t", 1); + let successor = valid + .lines() + .skip_while(|line| *line != row) + .nth(1) + .ok_or("missing successor fixture")?; + for (replacement, observed) in [ + (String::new(), successor), + (format!("{unknown}\n"), unknown.as_str()), + ] { + let malformed = valid.replacen(&format!("{row}\n"), &replacement, 1); + let error = artifact::validate(malformed.as_bytes(), &environment) + .err() + .ok_or("incomplete metadata was admitted")?; + exact_metadata_refusal(error, key, observed); + } + } + Ok(()) +} + +fn exact_metadata_refusal(error: BenchmarkBaselineError, key: &str, observed: &str) { + if [ + "build-profile", + "git-commit", + "git-tree", + "rustc-version", + "target-triple", + "os-description", + "cpu-model", + "logical-cpu-count", + ] + .contains(&key) + { + let expected = format!("report-{key}"); + assert!( + matches!(error, BenchmarkBaselineError::ReportViolation { reason } if reason == expected) + ); + } else { + assert!(matches!(error, BenchmarkBaselineError::InvalidReportRow { + expected, observed: actual, + } if expected == key && actual == observed)); + } +} + +#[test] +fn every_metric_row_refuses_missing_or_extra_fields_exactly() +-> Result<(), Box> { + let environment = environment(); + let valid = report(&environment, 13, 5); + for row in valid + .lines() + .filter(|line| line.starts_with("scenario\t") || line.starts_with("profile\t")) + { + let (truncated, _field) = row.rsplit_once('\t').ok_or("missing metric fixture")?; + let extended = format!("{row}\t1"); + for observed in [truncated, extended.as_str()] { + let malformed = valid.replacen(&format!("{row}\n"), &format!("{observed}\n"), 1); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::InvalidReportRow { + expected: "complete metric row", observed: actual, + }) if actual == observed) + ); + } + } + Ok(()) +} diff --git a/xtask/src/benchmark_baseline/counter_width_tests.rs b/xtask/src/benchmark_baseline/counter_width_tests.rs new file mode 100644 index 00000000..170a8f77 --- /dev/null +++ b/xtask/src/benchmark_baseline/counter_width_tests.rs @@ -0,0 +1,48 @@ +//! Laws for admission of producer-representable benchmark counters. + +use super::super::metric_error::ReportMetricError; +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn oversized_counters_refuse_with_their_exact_metric_and_bound() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for (name, metric) in [ + ("cold-ingest", "source-bytes-read"), + ("cold-ingest", "output-bytes-written"), + ("cold-ingest", "operation-count"), + ("cold-ingest", "total-allocation-count"), + ("cold-ingest", "total-allocated-bytes"), + ("cold-ingest", "peak-live-allocation-count"), + ("cold-ingest", "peak-live-heap-bytes"), + ("fixed-64", "base-materialized-bytes"), + ("fixed-64", "total-allocation-count"), + ("fixed-64", "total-allocated-bytes"), + ("fixed-64", "peak-live-heap-bytes"), + ] { + let malformed = + super::metric_relation_tests::mutate(&valid, name, metric, "18446744073709551616"); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::Metric(ReportMetricError::Bound { + metric: actual_metric, maximum: 18_446_744_073_709_551_615, + observed: 18_446_744_073_709_551_616, + })) if actual_metric == metric) + ); + } +} + +#[test] +fn the_maximum_counter_value_remains_admissible() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for (name, metric) in [ + ("cold-ingest", "source-bytes-read"), + ("cold-ingest", "total-allocation-count"), + ("fixed-64", "base-materialized-bytes"), + ] { + let maximum = + super::metric_relation_tests::mutate(&valid, name, metric, "18446744073709551615"); + assert!(artifact::validate(maximum.as_bytes(), &environment).is_ok()); + } +} diff --git a/xtask/src/benchmark_baseline/counter_widths.rs b/xtask/src/benchmark_baseline/counter_widths.rs new file mode 100644 index 00000000..8eaf6922 --- /dev/null +++ b/xtask/src/benchmark_baseline/counter_widths.rs @@ -0,0 +1,61 @@ +//! This module owns the portable u64 widths of v1 byte and count metrics. + +use super::BenchmarkBaselineError; +use super::metric_error::ReportMetricError; +use super::report_schema::{PROFILE_HEADER, SCENARIO_HEADER}; + +const COUNTERS: &[&str] = &[ + "logical-bytes", + "physical-bytes-read", + "physical-bytes-written", + "source-bytes-read", + "output-bytes-written", + "read-amplification-numerator", + "read-amplification-denominator", + "write-amplification-numerator", + "write-amplification-denominator", + "deduplication-ratio-numerator", + "deduplication-ratio-denominator", + "reused-unique-chunks", + "chunk-instances", + "operation-count", + "total-allocation-count", + "total-allocated-bytes", + "peak-live-allocation-count", + "peak-live-heap-bytes", + "base-unique-chunks", + "base-materialized-bytes", + "insertion-reused-chunks", + "deletion-reused-chunks", + "neighbor-reused-chunks", +]; + +pub(super) fn scenario(values: &str) -> Result<(), BenchmarkBaselineError> { + admit(SCENARIO_HEADER, values, 3) +} + +pub(super) fn profile(values: &str) -> Result<(), BenchmarkBaselineError> { + admit(PROFILE_HEADER, values, 7) +} + +fn admit(header: &'static str, values: &str, prefix: usize) -> Result<(), BenchmarkBaselineError> { + for (metric, value) in header.split('\t').skip(prefix).zip(values.split('\t')) { + if !COUNTERS.contains(&metric) { + continue; + } + let observed = value.parse::().map_err(|source| { + BenchmarkBaselineError::Metric(ReportMetricError::Encoding { + observed: value.to_owned(), + source, + }) + })?; + if observed > u128::from(u64::MAX) { + return Err(BenchmarkBaselineError::Metric(ReportMetricError::Bound { + metric, + maximum: u128::from(u64::MAX), + observed, + })); + } + } + Ok(()) +} diff --git a/xtask/src/benchmark_baseline/environment.rs b/xtask/src/benchmark_baseline/environment.rs index 747279d9..a8484cec 100644 --- a/xtask/src/benchmark_baseline/environment.rs +++ b/xtask/src/benchmark_baseline/environment.rs @@ -4,21 +4,14 @@ use std::path::Path; use std::process::Command; use super::BenchmarkBaselineError; -use super::host_environment::{self, CapturedHost}; +use super::host_environment; use super::process::{ProcessOutput, run}; use super::tracked_source; const DIAGNOSTIC_LIMIT: usize = 65_536; const VALUE_LIMIT: usize = 4_096; -#[derive(Eq, PartialEq)] -pub(super) struct CapturedEnvironment { - pub(super) commit: String, - pub(super) tree: &'static str, - pub(super) rustc_version: String, - pub(super) target_triple: String, - pub(super) host: CapturedHost, -} +pub(super) use super::captured_environment::CapturedEnvironment; pub(super) fn capture( repository_root: &Path, diff --git a/xtask/src/benchmark_baseline/error.rs b/xtask/src/benchmark_baseline/error.rs index 92e7570e..69a14186 100644 --- a/xtask/src/benchmark_baseline/error.rs +++ b/xtask/src/benchmark_baseline/error.rs @@ -54,6 +54,15 @@ pub(crate) enum BenchmarkBaselineError { ExternalCargoConfiguration { path: PathBuf, }, + ReportInput(super::report_input::ReportInputError), + Metric(super::metric_error::ReportMetricError), + InvalidReportRow { + expected: &'static str, + observed: String, + }, + DuplicateReportMetadata { + coordinate: String, + }, ReportViolation { reason: &'static str, }, @@ -117,6 +126,21 @@ impl fmt::Display for BenchmarkBaselineError { escaped_path(formatter, path)?; write!(formatter, "` makes benchmark evidence incomparable") } + Self::ReportInput(source) => fmt::Display::fmt(source, formatter), + Self::Metric(source) => fmt::Display::fmt(source, formatter), + Self::InvalidReportRow { expected, observed } => { + write!( + formatter, + "benchmark report expected {expected}, observed `" + )?; + escaped_controls(formatter, observed)?; + formatter.write_str("`") + } + Self::DuplicateReportMetadata { coordinate } => { + formatter.write_str("benchmark report repeats metadata `")?; + escaped_controls(formatter, coordinate)?; + formatter.write_str("`") + } Self::ReportViolation { reason } => { write!(formatter, "benchmark report violates `{reason}`") } @@ -131,6 +155,8 @@ impl Error for BenchmarkBaselineError { Self::DiagnosticEncoding { source, .. } | Self::ValueEncoding { source, .. } => { Some(source) } + Self::ReportInput(source) => Some(source), + Self::Metric(source) => Some(source), Self::MissingPipe { .. } | Self::ReaderThread { .. } | Self::OutputBound { .. } @@ -138,6 +164,8 @@ impl Error for BenchmarkBaselineError { | Self::InvalidValue { .. } | Self::AmbientBuildSetting { .. } | Self::ExternalCargoConfiguration { .. } + | Self::InvalidReportRow { .. } + | Self::DuplicateReportMetadata { .. } | Self::ReportViolation { .. } => None, } } diff --git a/xtask/src/benchmark_baseline/fixtures/single-pass-report-v1.tsv b/xtask/src/benchmark_baseline/fixtures/single-pass-report-v1.tsv new file mode 100644 index 00000000..e175ebc6 --- /dev/null +++ b/xtask/src/benchmark_baseline/fixtures/single-pass-report-v1.tsv @@ -0,0 +1,39 @@ +schema keep.streaming-cas-baseline/v1 +metadata build-profile optimized-release +metadata git-commit 30ffe90e53c01a24d8931244a7f76eaecd0da8a4 +metadata git-tree clean +metadata rustc-version rustc 1.96.0 (ac68faa20 2026-05-25) +metadata target-triple aarch64-apple-darwin +metadata os-description Darwin 27.0.0 arm64 +metadata cpu-model Apple M5 Pro +metadata cpu-clock process +metadata peak-memory incremental-live-heap +metadata verification mandatory +metadata timing-unit nanoseconds +metadata byte-unit bytes +metadata ratio-encoding exact-numerator-denominator +metadata logical-cpu-count 18 +metadata sample-count 100 +metadata warmup-count 5 +scenario-header name verification sample-count logical-bytes physical-bytes-read physical-bytes-written source-bytes-read output-bytes-written read-amplification-numerator read-amplification-denominator write-amplification-numerator write-amplification-denominator deduplication-ratio-numerator deduplication-ratio-denominator reused-unique-chunks chunk-instances operation-count logical-bytes-per-second total-wall-time-ns p50-wall-time-ns p95-wall-time-ns p99-wall-time-ns total-cpu-time-ns p50-cpu-time-ns p95-cpu-time-ns p99-cpu-time-ns total-allocation-count total-allocated-bytes peak-live-allocation-count peak-live-heap-bytes +scenario cold-ingest ingest-chunk-and-blob-identity 100 1048576 0 1048576 1048576 0 0 1048576 1048576 1048576 1048576 1048576 0 13 1 258493798 405648416 4020750 4198875 4486917 405353000 4022000 4198000 4488000 2800 132013200 20 1313464 +scenario warm-ingest ingest-chunk-and-blob-identity 100 1048576 1048576 0 1048576 0 1048576 1048576 0 1048576 1048576 0 13 13 1 260650153 402292493 4012917 4105125 4187708 401916000 4009000 4098000 4180000 600 26508400 3 263488 +scenario repeated-near-neighbor-edits ingest-chunk-and-blob-identity 100 8388608 6191702 2196906 8388608 0 6191702 8388608 2196906 8388608 8388608 2196906 74 100 4 259823431 3228580248 31947041 33969791 34018833 3225173000 31908000 33947000 33999000 6800 327809000 40 2470386 +scenario early-insertion ingest-chunk-and-blob-identity 100 4198400 1997398 2201002 4198400 0 1997398 4198400 2201002 4198400 4198400 2201002 24 50 2 255320849 1644362382 16262167 17314959 17849500 1639816000 16241000 17145000 17435000 5400 274622600 38 2472826 +scenario early-deletion ingest-chunk-and-blob-identity 100 4190208 1997398 2192810 4190208 0 1997398 4190208 2192810 4190208 4190208 2192810 24 50 2 256626666 1632803035 16261334 16784250 17255416 1630332000 16247000 16768000 17093000 5400 273803400 38 2464634 +scenario many-tiny-blobs ingest-chunk-and-blob-identity 100 32896 0 32896 32896 0 0 32896 32896 32896 32896 32896 0 256 256 57504209 57206247 568167 607833 637541 57207000 568000 601000 629000 191200 6764795200 889 538368 +scenario large-binary ingest-chunk-and-blob-identity 100 1048576 0 1048576 1048576 0 0 1048576 1048576 1048576 1048576 1048576 0 15 1 260492529 402535920 4010667 4138250 4175209 402114000 4007000 4135000 4169000 3000 132031600 22 1313464 +scenario high-deduplication ingest-chunk-and-blob-identity 100 4194304 2097152 2097152 4194304 0 2097152 4194304 2097152 4194304 4194304 2097152 25 50 2 257589265 1628291457 16194083 17003209 17375250 1625548000 16174000 16935000 17231000 5000 264075200 36 2368392 +scenario zero-deduplication ingest-chunk-and-blob-identity 100 3145728 0 3145728 3145728 0 0 3145728 3145728 3145728 3145728 3145728 0 44 2 257237578 1222888204 12180000 12527792 12819375 1221580000 12168000 12516000 12803000 7500 369104000 58 3417440 +scenario sequential-range-reads selected-complete-chunks 100 1048576 3413411 0 0 1048576 3413411 1048576 0 1048576 1048576 0 0 46 32 413525989 253569552 2525750 2596250 2626167 253340000 2525000 2596000 2621000 0 0 0 0 +scenario random-range-reads selected-complete-chunks 100 131072 2396798 0 0 131072 2396798 131072 0 131072 131072 0 0 33 32 73708940 177823748 1776500 1808333 1832375 177667000 1774000 1808000 1826000 0 0 0 0 +scenario whole-blob-verification chunks-profile-and-blob 100 1048576 1048576 0 0 1048576 1048576 1048576 0 1048576 1048576 0 0 15 1 255121665 411010173 4053250 4380958 4842791 409219000 4046000 4279000 4648000 0 0 0 0 +scenario varied-input-partitioning ingest-chunk-and-blob-identity 100 262144 0 262144 262144 0 0 262144 262144 262144 262144 262144 0 4 4 142660033 183754338 1836916 1917125 1948250 183626000 1834000 1914000 1943000 4000 132846400 6 328488 +profile-header name provenance minimum-kib target-kib maximum-kib timed-input sample-count logical-bytes-per-second total-wall-time-ns p50-wall-time-ns p95-wall-time-ns p99-wall-time-ns total-cpu-time-ns total-allocation-count total-allocated-bytes peak-live-heap-bytes base-unique-chunks base-materialized-bytes insertion-reused-chunks deletion-reused-chunks neighbor-reused-chunks +profile keep-fastcdc-4-16-64 keep.fastcdc-gear64/v1 4 16 64 large-text 100 845298580 124048002 1238042 1282291 1299334 123967000 200 364000 3640 86 2097152 81 85 85 +profile keep-fastcdc-16-64-256 keep.fastcdc-gear64/v1 16 64 256 large-text 100 856413879 122437997 1219041 1267000 1305500 122357000 200 98800 988 25 2097152 24 24 24 +profile keep-fastcdc-64-256-1024 keep.fastcdc-gear64/v1 64 256 1024 large-text 100 857980609 122214417 1212667 1275708 1288625 122139000 200 24400 244 5 2097152 4 4 4 +profile fixed-64 benchmark.fixed-size/v1 64 64 64 large-text 100 1341675729 78154205 776208 808416 834750 78145000 200 71200 712 32 2097152 0 0 31 +profile git-cas-buzhash-64-256-1024 git-cas@432c5d9effb12c9f66536f1386791bb4421f3cea 64 256 1024 large-text 100 638184508 164306088 1646125 1695125 1699542 164202000 200 24400 244 5 2097152 4 4 4 +threshold-header metric status rationale +threshold all-performance-metrics unconfigured requires-controlled-baseline-history diff --git a/xtask/src/benchmark_baseline/host_environment.rs b/xtask/src/benchmark_baseline/host_environment.rs index 307c782b..e4b39044 100644 --- a/xtask/src/benchmark_baseline/host_environment.rs +++ b/xtask/src/benchmark_baseline/host_environment.rs @@ -4,7 +4,6 @@ use std::fs::File; #[cfg(target_os = "linux")] use std::io::Read; -use std::num::NonZeroUsize; use std::path::Path; use std::process::Command; @@ -20,12 +19,7 @@ const VALUE_LIMIT: usize = 4_096; #[cfg(target_os = "linux")] const CPUINFO_READ_LIMIT: u64 = 65_536; -#[derive(Eq, PartialEq)] -pub(super) struct CapturedHost { - pub(super) os_description: String, - pub(super) cpu_model: String, - pub(super) logical_cpu_count: NonZeroUsize, -} +pub(super) use super::captured_environment::CapturedHost; pub(super) fn capture() -> Result { let logical_cpu_count = diff --git a/xtask/src/benchmark_baseline/metadata_policy.rs b/xtask/src/benchmark_baseline/metadata_policy.rs new file mode 100644 index 00000000..a8440c45 --- /dev/null +++ b/xtask/src/benchmark_baseline/metadata_policy.rs @@ -0,0 +1,34 @@ +//! This module owns the fixed optimized subprocess measurement policy. + +use super::BenchmarkBaselineError; + +pub(super) const SAMPLE_COUNT: &str = "100"; + +pub(super) fn admit(key: &str, observed: &str) -> Result<(), BenchmarkBaselineError> { + let expected = match key { + "cpu-clock" => "process", + "peak-memory" => "incremental-live-heap", + "verification" => "mandatory", + "timing-unit" => "nanoseconds", + "byte-unit" => "bytes", + "ratio-encoding" => "exact-numerator-denominator", + "sample-count" => SAMPLE_COUNT, + "warmup-count" => "5", + _ => return Ok(()), + }; + require(expected, observed) +} + +pub(super) fn require( + expected: &'static str, + observed: &str, +) -> Result<(), BenchmarkBaselineError> { + if expected == observed { + Ok(()) + } else { + Err(BenchmarkBaselineError::InvalidReportRow { + expected, + observed: observed.to_owned(), + }) + } +} diff --git a/xtask/src/benchmark_baseline/metadata_policy_tests.rs b/xtask/src/benchmark_baseline/metadata_policy_tests.rs new file mode 100644 index 00000000..ae2c3628 --- /dev/null +++ b/xtask/src/benchmark_baseline/metadata_policy_tests.rs @@ -0,0 +1,52 @@ +//! Laws for benchmark metadata policies and consistent sample evidence. + +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn optimized_metadata_policies_refuse_substitution_exactly() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for (key, expected, observed) in [ + ("cpu-clock", "process", "wall"), + ("peak-memory", "incremental-live-heap", "unknown"), + ("verification", "mandatory", "disabled"), + ("timing-unit", "nanoseconds", "milliseconds"), + ("byte-unit", "bytes", "kilobytes"), + ("ratio-encoding", "exact-numerator-denominator", "float"), + ("sample-count", "100", "0"), + ("sample-count", "100", "99"), + ("sample-count", "100", "1001"), + ("warmup-count", "5", "0"), + ("warmup-count", "5", "101"), + ] { + let malformed = valid.replace( + &format!("metadata\t{key}\t{expected}\n"), + &format!("metadata\t{key}\t{observed}\n"), + ); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::InvalidReportRow { + expected: actual_expected, observed: actual_observed, + }) if actual_expected == expected && actual_observed == observed) + ); + } +} + +#[test] +fn every_catalog_row_binds_its_sample_count_to_the_publication_policy() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for row in valid + .lines() + .filter(|line| line.starts_with("scenario\t") || line.starts_with("profile\t")) + { + let malformed_row = row.replacen("\t100\t", "\t99\t", 1); + let malformed = valid.replace(row, &malformed_row); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::InvalidReportRow { + expected: "100", observed, + }) if observed == "99") + ); + } +} diff --git a/xtask/src/benchmark_baseline/metadata_uniqueness.rs b/xtask/src/benchmark_baseline/metadata_uniqueness.rs new file mode 100644 index 00000000..19d3bcc2 --- /dev/null +++ b/xtask/src/benchmark_baseline/metadata_uniqueness.rs @@ -0,0 +1,45 @@ +//! This module owns rejection of ambiguous report metadata coordinates. + +use std::collections::BTreeSet; + +use super::BenchmarkBaselineError; + +pub(super) fn admit(report: &str) -> Result<(), BenchmarkBaselineError> { + let mut coordinates = BTreeSet::new(); + for line in report.lines() { + let mut fields = line.split('\t'); + if fields.next() != Some("metadata") { + continue; + } + if let Some(coordinate) = fields.next() + && !coordinates.insert(coordinate) + { + return Err(BenchmarkBaselineError::DuplicateReportMetadata { + coordinate: coordinate.to_owned(), + }); + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::super::BenchmarkBaselineError; + + #[test] + fn conflicting_and_identical_metadata_repetitions_both_refuse() { + for repeated in ["first", "second"] { + let report = format!("metadata\tgit-commit\tfirst\nmetadata\tgit-commit\t{repeated}\n"); + assert!(matches!( + super::admit(&report), + Err(BenchmarkBaselineError::DuplicateReportMetadata { coordinate }) + if coordinate == "git-commit" + )); + } + } + + #[test] + fn distinct_coordinates_may_share_values() { + assert!(super::admit("metadata\tfirst\tvalue\nmetadata\tsecond\tvalue\n").is_ok()); + } +} diff --git a/xtask/src/benchmark_baseline/metric_error.rs b/xtask/src/benchmark_baseline/metric_error.rs new file mode 100644 index 00000000..3446e603 --- /dev/null +++ b/xtask/src/benchmark_baseline/metric_error.rs @@ -0,0 +1,80 @@ +//! This module owns precise failures of benchmark metric relationships. + +use std::error::Error; +use std::fmt; +use std::num::ParseIntError; + +use crate::diagnostic::escaped_controls; + +pub(crate) enum ReportMetricError { + Mismatch { + metric: &'static str, + expected: u128, + observed: u128, + }, + Bound { + metric: &'static str, + maximum: u128, + observed: u128, + }, + Arithmetic { + bytes: u128, + samples: u128, + duration: u128, + }, + Encoding { + observed: String, + source: ParseIntError, + }, +} + +impl fmt::Debug for ReportMetricError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(self, formatter) + } +} + +impl fmt::Display for ReportMetricError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Mismatch { + metric, + expected, + observed, + } => write!( + formatter, + "benchmark metric `{metric}` expected {expected}, observed {observed}" + ), + Self::Bound { + metric, + maximum, + observed, + } => write!( + formatter, + "benchmark metric `{metric}` exceeds {maximum}, observed {observed}" + ), + Self::Arithmetic { + bytes, + samples, + duration, + } => write!( + formatter, + "benchmark throughput arithmetic refused: bytes {bytes}, samples {samples}, duration {duration}" + ), + Self::Encoding { observed, .. } => { + formatter.write_str("benchmark metric cannot decode `")?; + escaped_controls(formatter, observed)?; + formatter.write_str("`") + } + } + } +} + +impl Error for ReportMetricError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::Encoding { source, .. } => Some(source), + Self::Mismatch { .. } | Self::Bound { .. } | Self::Arithmetic { .. } => None, + } + } +} diff --git a/xtask/src/benchmark_baseline/metric_relation_tests.rs b/xtask/src/benchmark_baseline/metric_relation_tests.rs new file mode 100644 index 00000000..83b528a4 --- /dev/null +++ b/xtask/src/benchmark_baseline/metric_relation_tests.rs @@ -0,0 +1,196 @@ +//! Laws for admission of internally consistent benchmark evidence. + +use super::super::metric_error::ReportMetricError; +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn ratio_and_throughput_fields_bind_to_measured_counters() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for (column, metric, expected) in [ + ( + "read-amplification-numerator", + "read-amplification-numerator", + 0, + ), + ( + "read-amplification-denominator", + "read-amplification-denominator", + 1_048_576, + ), + ( + "write-amplification-numerator", + "write-amplification-numerator", + 1_048_576, + ), + ( + "write-amplification-denominator", + "write-amplification-denominator", + 1_048_576, + ), + ( + "deduplication-ratio-numerator", + "deduplication-ratio-numerator", + 1_048_576, + ), + ( + "deduplication-ratio-denominator", + "deduplication-ratio-denominator", + 1_048_576, + ), + ( + "logical-bytes-per-second", + "logical-bytes-per-second", + 166_506_928, + ), + ] { + let malformed = mutate(&valid, "cold-ingest", column, "1"); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::Metric(ReportMetricError::Mismatch { + metric: actual, expected: actual_expected, observed: 1, + })) if actual == metric && actual_expected == expected) + ); + } +} + +#[test] +fn percentiles_and_reused_chunks_are_bounded_by_their_witnesses() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for (name, column, metric, maximum) in [ + ( + "cold-ingest", + "reused-unique-chunks", + "reused-unique-chunks", + 13, + ), + ( + "cold-ingest", + "p50-wall-time-ns", + "wall-percentiles", + 6_503_458, + ), + ( + "cold-ingest", + "p50-cpu-time-ns", + "cpu-percentiles", + 6_449_000, + ), + ( + "fixed-64", + "insertion-reused-chunks", + "insertion-reused-chunks", + 32, + ), + ( + "fixed-64", + "p50-wall-time-ns", + "profile-wall-percentiles", + 1_373_875, + ), + ] { + let malformed = mutate(&valid, name, column, "999999999999"); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::Metric(ReportMetricError::Bound { + metric: actual, maximum: actual_maximum, observed: 999_999_999_999, + })) if actual == metric && actual_maximum == maximum) + ); + } +} + +#[test] +fn throughput_refuses_zero_duration_without_arithmetic_approximation() { + let environment = environment(); + let valid = report(&environment, 13, 5); + let mut malformed = valid; + for column in [ + "total-wall-time-ns", + "p50-wall-time-ns", + "p95-wall-time-ns", + "p99-wall-time-ns", + ] { + malformed = mutate(&malformed, "cold-ingest", column, "0"); + } + assert!(matches!( + artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::Metric( + ReportMetricError::Arithmetic { + bytes: 1_048_576, + samples: 100, + duration: 0, + } + )) + )); +} + +#[test] +fn historical_metric_relationships_are_admissible() -> Result<(), BenchmarkBaselineError> { + for row in super::HISTORICAL_BASELINE.lines() { + if row.starts_with("scenario\t") { + let values = row.split('\t').skip(3).collect::>().join("\t"); + super::super::metric_relations::scenario(&values)?; + } + if row.starts_with("profile\t") { + let values = row.split('\t').skip(7).collect::>().join("\t"); + super::super::metric_relations::profile(&values)?; + } + } + Ok(()) +} + +pub(super) fn mutate(report: &str, name: &str, column: &str, value: &str) -> String { + let mut output = String::new(); + let mut header = ""; + for row in report.lines() { + if row.starts_with("scenario-header\t") || row.starts_with("profile-header\t") { + header = row; + } + let mut fields = row.split('\t'); + let kind = fields.next(); + let row_name = fields.next(); + if matches!(kind, Some("scenario" | "profile")) && row_name == Some(name) { + let changed = row + .split('\t') + .zip(header.split('\t')) + .map(|(original, field)| if field == column { value } else { original }) + .collect::>() + .join("\t"); + output.push_str(&changed); + } else { + output.push_str(row); + } + output.push('\n'); + } + output +} + +#[test] +fn overflowing_metric_decoding_retains_the_original_parse_failure() +-> Result<(), Box> { + use std::error::Error; + use std::num::{IntErrorKind, ParseIntError}; + + let environment = environment(); + let valid = report(&environment, 13, 5); + let overflow = "340282366920938463463374607431768211456"; + let malformed = mutate(&valid, "cold-ingest", "logical-bytes", overflow); + let error = artifact::validate(malformed.as_bytes(), &environment) + .err() + .ok_or_else(|| std::io::Error::other("overflow was admitted"))?; + assert!( + matches!(&error, BenchmarkBaselineError::Metric(ReportMetricError::Encoding { + observed, source, + }) if observed == overflow && source.kind() == &IntErrorKind::PosOverflow) + ); + assert_eq!( + error + .source() + .and_then(Error::source) + .and_then(|source| source.downcast_ref::()) + .map(ParseIntError::kind), + Some(&IntErrorKind::PosOverflow) + ); + Ok(()) +} diff --git a/xtask/src/benchmark_baseline/metric_relations.rs b/xtask/src/benchmark_baseline/metric_relations.rs new file mode 100644 index 00000000..7cc0c80a --- /dev/null +++ b/xtask/src/benchmark_baseline/metric_relations.rs @@ -0,0 +1,186 @@ +//! This module owns arithmetic relationships between admitted metric fields. + +use super::BenchmarkBaselineError; +use super::metric_error::ReportMetricError; + +pub(super) fn scenario(values: &str) -> Result<(), BenchmarkBaselineError> { + let [ + samples, + logical, + read, + written, + _source, + _output, + read_num, + read_den, + write_num, + write_den, + dedup_num, + dedup_den, + reused, + chunks, + _operations, + rate, + wall_total, + wall50, + wall95, + wall99, + cpu_total, + cpu50, + cpu95, + cpu99, + _allocations, + _allocated, + _peak_count, + _peak_heap, + ] = numbers::<28>(values)?; + for (metric, expected, observed) in [ + ("read-amplification-numerator", read, read_num), + ("read-amplification-denominator", logical, read_den), + ("write-amplification-numerator", written, write_num), + ("write-amplification-denominator", logical, write_den), + ("deduplication-ratio-numerator", logical, dedup_num), + ("deduplication-ratio-denominator", written, dedup_den), + ] { + equal(metric, expected, observed)?; + } + at_most("reused-unique-chunks", chunks, reused)?; + percentiles("wall-percentiles", [wall50, wall95, wall99, wall_total])?; + percentiles("cpu-percentiles", [cpu50, cpu95, cpu99, cpu_total])?; + throughput(logical, samples, wall_total, rate) +} + +pub(super) fn profile(values: &str) -> Result<(), BenchmarkBaselineError> { + let [ + samples, + rate, + wall_total, + wall50, + wall95, + wall99, + _cpu_total, + _allocations, + _allocated, + _peak_heap, + base_chunks, + _base_bytes, + inserted, + deleted, + neighbor, + ] = numbers::<15>(values)?; + for (metric, reused) in [ + ("insertion-reused-chunks", inserted), + ("deletion-reused-chunks", deleted), + ("neighbor-reused-chunks", neighbor), + ] { + at_most(metric, base_chunks, reused)?; + } + percentiles( + "profile-wall-percentiles", + [wall50, wall95, wall99, wall_total], + )?; + // The frozen v1 catalog's timed large-text member is exactly one MiB. + throughput(1_048_576, samples, wall_total, rate) +} + +fn numbers(values: &str) -> Result<[u128; N], BenchmarkBaselineError> { + let mut fields = values.split('\t'); + let mut admitted = [0; N]; + for slot in &mut admitted { + let observed = fields.next().unwrap_or_default(); + *slot = observed.parse().map_err(|source| { + BenchmarkBaselineError::Metric(ReportMetricError::Encoding { + observed: observed.to_owned(), + source, + }) + })?; + } + if fields.next().is_some() { + return Err(BenchmarkBaselineError::InvalidReportRow { + expected: "complete metric row", + observed: values.to_owned(), + }); + } + Ok(admitted) +} + +const fn equal( + metric: &'static str, + expected: u128, + observed: u128, +) -> Result<(), BenchmarkBaselineError> { + if expected == observed { + Ok(()) + } else { + Err(BenchmarkBaselineError::Metric( + ReportMetricError::Mismatch { + metric, + expected, + observed, + }, + )) + } +} + +const fn at_most( + metric: &'static str, + maximum: u128, + observed: u128, +) -> Result<(), BenchmarkBaselineError> { + if observed <= maximum { + Ok(()) + } else { + Err(BenchmarkBaselineError::Metric(ReportMetricError::Bound { + metric, + maximum, + observed, + })) + } +} + +fn percentiles(metric: &'static str, values: [u128; 4]) -> Result<(), BenchmarkBaselineError> { + let [p50, p95, p99, total] = values; + for (maximum, observed) in [(p95, p50), (p99, p95), (total, p99)] { + at_most(metric, maximum, observed)?; + } + Ok(()) +} + +fn throughput( + bytes: u128, + samples: u128, + duration: u128, + observed: u128, +) -> Result<(), BenchmarkBaselineError> { + let expected = bytes + .checked_mul(samples) + .and_then(|value| value.checked_mul(1_000_000_000)) + .and_then(|value| value.checked_div(duration)) + .ok_or(BenchmarkBaselineError::Metric( + ReportMetricError::Arithmetic { + bytes, + samples, + duration, + }, + ))?; + equal("logical-bytes-per-second", expected, observed) +} + +#[cfg(test)] +mod tests { + use super::{BenchmarkBaselineError, ReportMetricError}; + + #[test] + fn throughput_product_overflow_refuses_instead_of_saturating() { + assert!(matches!( + super::throughput(u128::MAX, 100, 1, 0), + Err(BenchmarkBaselineError::Metric( + ReportMetricError::Arithmetic { + bytes: u128::MAX, + samples: 100, + duration: 1, + } + )) + )); + } +} diff --git a/xtask/src/benchmark_baseline/mod.rs b/xtask/src/benchmark_baseline/mod.rs index 4b4396ae..436dc0b9 100644 --- a/xtask/src/benchmark_baseline/mod.rs +++ b/xtask/src/benchmark_baseline/mod.rs @@ -3,10 +3,19 @@ mod artifact; mod artifact_publication; mod build_environment; +mod captured_environment; +mod counter_widths; mod environment; mod error; mod host_environment; +mod metadata_policy; +mod metadata_uniqueness; +mod metric_error; +mod metric_relations; mod process; +mod report_grammar; +mod report_input; +mod report_schema; mod tracked_source; use std::path::Path; @@ -15,7 +24,7 @@ use std::process::Command; pub(crate) use error::BenchmarkBaselineError; const DIAGNOSTIC_LIMIT: usize = 262_144; -const REPORT_LIMIT: usize = 1_048_576; +const REPORT_LIMIT: usize = report_input::MAXIMUM_REPORT_BYTES; pub(crate) fn run(repository_root: &Path) -> Result<(), BenchmarkBaselineError> { build_environment::admit(repository_root)?; @@ -60,8 +69,8 @@ pub(crate) fn run(repository_root: &Path) -> Result<(), BenchmarkBaselineError> reason: "successful-benchmark-wrote-diagnostics", }); } - artifact::validate(&output.stdout, &environment)?; - artifact_publication::persist(repository_root, &output.stdout) + let report = artifact::validate(&output.stdout, &environment)?; + artifact_publication::persist(repository_root, &report) } fn admit_clean_source( diff --git a/xtask/src/benchmark_baseline/numeric_format_tests.rs b/xtask/src/benchmark_baseline/numeric_format_tests.rs new file mode 100644 index 00000000..98815468 --- /dev/null +++ b/xtask/src/benchmark_baseline/numeric_format_tests.rs @@ -0,0 +1,19 @@ +//! Laws for canonical metric decimal encodings. + +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn noncanonical_metric_decimals_refuse_exactly() { + let environment = environment(); + let valid = report(&environment, 13, 5); + let prefix = "scenario\tcold-ingest\tingest-chunk-and-blob-identity\t"; + for observed in ["", "0100", "+100", "-100", "1.0"] { + let malformed = valid.replace(&format!("{prefix}100\t"), &format!("{prefix}{observed}\t")); + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::InvalidReportRow { + expected: "canonical unsigned decimal", observed: actual, + }) if actual == observed) + ); + } +} diff --git a/xtask/src/benchmark_baseline/rationale.md b/xtask/src/benchmark_baseline/rationale.md new file mode 100644 index 00000000..a5bfee47 --- /dev/null +++ b/xtask/src/benchmark_baseline/rationale.md @@ -0,0 +1,102 @@ +# Canonical benchmark-report admission + +This directory owns the ingress and publication boundary for optimized +streaming-CAS baseline evidence. The report is a protocol, not arbitrary +stdout: the named profile is `keep.streaming-cas-baseline/v1`, with UTF-8, +LF framing, tab-separated fields and a final LF. + +The parser and subprocess capture share a one-MiB input ceiling. The parser +checks byte length before UTF-8 decoding or allocation-heavy metadata admission. +The exact limit reaches decoding; a larger input refuses with maximum and +observed lengths, even if its encoding is invalid. UTF-8 failures retain their +original source through the typed error chain. + +A source coordinate may occur exactly once. Identical duplicates also refuse: +accepting them would admit multiple representations and leave ambiguous +interpretations available to downstream tools. Refusals retain the coordinate +and escape controls in diagnostics. + +The frozen row grammar requires all metadata keys in writer order, exact +scenario and profile headers, complete ordered catalogs, fixed row widths, +unsigned canonical decimal metrics and the explicit unconfigured threshold +policy. Decimal fields reject signs, leading zeros, empty values and u128 +overflow. Scenario identity includes its verification posture; profile identity +includes provenance, chunk bounds and timed input. Historical measurements are +not imported as expected metric values. The committed baseline is a positive +compatibility fixture rather than a performance expectation. + +The optimized subprocess has a fixed policy of 100 measured samples and five +warmups, matching `benchmark/src/main.rs`. Metadata records that exact policy; +every scenario and profile row repeats the same sample count. Policy changes +require a coordinated update to the runner and admission contract. Process CPU +clock, incremental live-heap memory, mandatory verification, nanoseconds, bytes +and exact numerator/denominator ratios are admitted as fixed semantic values. + +Metric relationships are admitted with typed expected/observed failures. Ratio +numerators and denominators must equal their named counters; reused chunks may +not exceed their base or observed chunk counts. Percentiles must be ordered +and may not exceed the aggregate total. Throughput is the exact integer quotient +of logical bytes times samples times one billion divided by wall duration; +overflow and zero duration refuse. The frozen profile timed input is one MiB. +The decoder preserves its original integer parse error through the error chain. + +Byte and count metrics use a portable unsigned 64-bit ceiling, including +profile chunk counts whose producer representation is `usize`. Ratio counters +share the width of the byte counters they repeat. Timing and throughput retain +unsigned 128-bit precision. The maximum counter is admitted; a larger value +refuses with the metric name, maximum and observed value before relationship +checks. The ingress makes no narrowing or saturating conversion. + +Admission produces an `AdmittedReport` whose fields are private to the +admission module. It borrows the exact immutable input bytes without copying +them. Publication requires a reference to this validated type; raw bytes cannot +reach the persister directly. The borrow prevents mutation of the source bytes +between admission and publication. Admission runs before artifact publication. A grammar refusal performs no +filesystem mutation. Existing publication and recovery ordering are unchanged; +these checks introduce no durable format version or public API change. + +Issue #142 remains open until final acceptance verification is complete. A refusal law verifies that a prior artifact, an +interrupted stage and the absence of a publication lock remain unchanged. Structural admission alone does not prove those semantics. + +The committed report supplies an independent row-order fixture: each of its +38 adjacent-row transpositions refuses at the first exact order violation, and +each of its 18 catalog-row deletions refuses the named incomplete catalog. A +temporary metadata-order mutation causes the transposition law to fail; the +restored production parser passes both laws in debug and release. This bounded +model evidence complements the parser fuzz campaign. + +The I/O-free benchmark-report fuzz facade reuses these production parser files +and captured-coordinate types. Its dependency-free feature admits the fixed +historical seed and refuses deterministic corruptions. Seed preparation +materializes the canonical report for the registered libFuzzer target. On the +reviewed nightly and cargo-fuzz versions, a final-source bounded campaign ran +10,000 executions with seed 142, a one-MiB input cap, five-second input timeout +and one-GiB RSS limit without a failure. This is exploration evidence, not a +claim that green coverage establishes absence of malformed states. + +Compatibility also admits the unchanged single-pass report produced at +`30ffe90e53c01a24d8931244a7f76eaecd0da8a4`, with its matching M5 Pro/Darwin +coordinates. The copied fixture is historical evidence from PR #134, not a new +measurement or a cross-host performance comparison. Thirty-six further catalog +mutations refuse unknown or duplicated members at their exact slots. + +Earlier verification found the reviewed harness registry still listed eleven +targets. It now includes `benchmark_report`; all 26 campaign-policy laws passed +in debug and release. That earlier full-workspace run was blocked by the +hardware-dependent source-identity law tracked in #139 and PR #140. The first +run also lacked the external BLAKE3 tool on PATH; its oracle law passed with +the real tool available. Those results describe the earlier source, not the +current validation state. + +After integrating main through `8d902516e682361882bc5c9902de296ce5c9de85`, +including the merged #140 correction, exact head +`e1e03e50ffc62741f3fb1c884dd414c13637a399` passed copy-isolated Docker full +workspace debug and release tests, including doctests, formatting, both +all-target/all-feature Clippy profiles with warnings denied, and source +structure checks. The external BLAKE3 oracle used the installed real tool. +These checks do not replace independent review or establish merge approval. + +Completeness models also check all sixteen metadata coordinates for deletion +and unknown-key replacement (32 mutations), and all eighteen metric rows for +a missing or extra field (36 mutations). Both laws assert exact typed failure +coordinates or complete-row observations and pass in Docker debug and release. diff --git a/xtask/src/benchmark_baseline/report_compatibility_tests.rs b/xtask/src/benchmark_baseline/report_compatibility_tests.rs new file mode 100644 index 00000000..aac68067 --- /dev/null +++ b/xtask/src/benchmark_baseline/report_compatibility_tests.rs @@ -0,0 +1,36 @@ +//! Compatibility laws for historical two-pass and optimized single-pass reports. + +use std::num::NonZeroUsize; + +use super::super::captured_environment::{CapturedEnvironment, CapturedHost}; +use super::super::{BenchmarkBaselineError, artifact}; + +#[test] +fn single_pass_report_retains_its_source_and_counter_semantics() +-> Result<(), Box> { + let bytes = include_bytes!("fixtures/single-pass-report-v1.tsv"); + let environment = CapturedEnvironment { + commit: String::from("30ffe90e53c01a24d8931244a7f76eaecd0da8a4"), + tree: "clean", + rustc_version: String::from("rustc 1.96.0 (ac68faa20 2026-05-25)"), + target_triple: String::from("aarch64-apple-darwin"), + host: CapturedHost { + os_description: String::from("Darwin 27.0.0 arm64"), + cpu_model: String::from("Apple M5 Pro"), + logical_cpu_count: NonZeroUsize::new(18).ok_or("invalid CPU fixture")?, + }, + }; + let admitted = artifact::validate(bytes, &environment)?; + assert_eq!(admitted.bytes(), bytes); + let mismatched = CapturedEnvironment { + commit: String::from("c529c07f385b5bcd76a4e57c1987001d496f9135"), + ..environment + }; + assert!(matches!( + artifact::validate(bytes, &mismatched), + Err(BenchmarkBaselineError::ReportViolation { + reason: "report-git-commit" + }) + )); + Ok(()) +} diff --git a/xtask/src/benchmark_baseline/report_grammar.rs b/xtask/src/benchmark_baseline/report_grammar.rs new file mode 100644 index 00000000..d4335628 --- /dev/null +++ b/xtask/src/benchmark_baseline/report_grammar.rs @@ -0,0 +1,126 @@ +//! This module owns complete, ordered baseline row and decimal admission. + +use std::str::Lines; + +use super::BenchmarkBaselineError; +use super::report_schema::{ + METADATA_KEYS, PROFILE_HEADER, PROFILE_PREFIXES, SCENARIO_HEADER, SCENARIO_PREFIXES, +}; + +enum MetricCatalog { + Scenario, + Profile, +} + +pub(super) fn admit(report: &str) -> Result<(), BenchmarkBaselineError> { + let mut lines = report.lines(); + exact(&mut lines, "schema\tkeep.streaming-cas-baseline/v1")?; + for key in METADATA_KEYS { + metadata(&mut lines, key)?; + } + exact(&mut lines, SCENARIO_HEADER)?; + for prefix in SCENARIO_PREFIXES { + metrics(&mut lines, prefix, MetricCatalog::Scenario)?; + } + exact(&mut lines, PROFILE_HEADER)?; + for prefix in PROFILE_PREFIXES { + metrics(&mut lines, prefix, MetricCatalog::Profile)?; + } + exact(&mut lines, "threshold-header\tmetric\tstatus\trationale")?; + exact( + &mut lines, + "threshold\tall-performance-metrics\tunconfigured\trequires-controlled-baseline-history", + )?; + lines + .next() + .map_or(Ok(()), |observed| invalid("end of report", observed)) +} + +fn exact(lines: &mut Lines<'_>, expected: &'static str) -> Result<(), BenchmarkBaselineError> { + let observed = lines.next().unwrap_or_default(); + if observed == expected { + Ok(()) + } else { + invalid(expected, observed) + } +} + +fn metadata(lines: &mut Lines<'_>, key: &'static str) -> Result<(), BenchmarkBaselineError> { + let observed = lines.next().unwrap_or_default(); + let mut fields = observed.split('\t'); + if fields.next() != Some("metadata") || fields.next() != Some(key) { + return invalid(key, observed); + } + let value = fields.next().unwrap_or_default(); + if value.is_empty() || value.chars().any(char::is_control) || fields.next().is_some() { + return invalid("one nonempty metadata value", observed); + } + if matches!(key, "logical-cpu-count" | "sample-count" | "warmup-count") { + decimal(value)?; + } + super::metadata_policy::admit(key, value) +} + +fn metrics( + lines: &mut Lines<'_>, + prefix: &'static str, + catalog: MetricCatalog, +) -> Result<(), BenchmarkBaselineError> { + let observed = lines.next().unwrap_or_default(); + let Some(values) = observed + .strip_prefix(prefix) + .and_then(|tail| tail.strip_prefix('\t')) + else { + return invalid(prefix, observed); + }; + let mut fields = values.split('\t'); + let sample_count = fields.next().unwrap_or_default(); + decimal(sample_count)?; + super::metadata_policy::require(super::metadata_policy::SAMPLE_COUNT, sample_count)?; + let mut count = 1_usize; + for value in fields { + decimal(value)?; + count = count + .checked_add(1) + .ok_or_else(|| BenchmarkBaselineError::InvalidReportRow { + expected: "bounded metric count", + observed: observed.to_owned(), + })?; + } + match (catalog, count) { + (MetricCatalog::Scenario, 28) => { + super::counter_widths::scenario(values)?; + super::metric_relations::scenario(values) + } + (MetricCatalog::Profile, 15) => { + super::counter_widths::profile(values)?; + super::metric_relations::profile(values) + } + (MetricCatalog::Scenario | MetricCatalog::Profile, _) => { + invalid("complete metric row", observed) + } + } +} + +fn decimal(value: &str) -> Result<(), BenchmarkBaselineError> { + if value.is_empty() + || !value.bytes().all(|byte| byte.is_ascii_digit()) + || (value.len() > 1 && value.starts_with('0')) + { + invalid("canonical unsigned decimal", value) + } else { + value.parse::().map(|_admitted| ()).map_err(|source| { + BenchmarkBaselineError::Metric(super::metric_error::ReportMetricError::Encoding { + observed: value.to_owned(), + source, + }) + }) + } +} + +fn invalid(expected: &'static str, observed: &str) -> Result { + Err(BenchmarkBaselineError::InvalidReportRow { + expected, + observed: observed.to_owned(), + }) +} diff --git a/xtask/src/benchmark_baseline/report_input.rs b/xtask/src/benchmark_baseline/report_input.rs new file mode 100644 index 00000000..2e02d0d5 --- /dev/null +++ b/xtask/src/benchmark_baseline/report_input.rs @@ -0,0 +1,55 @@ +//! This module owns byte-size and UTF-8 admission before report parsing. + +use std::error::Error; +use std::fmt; +use std::str::Utf8Error; + +use super::BenchmarkBaselineError; + +pub(super) const MAXIMUM_REPORT_BYTES: usize = 1_048_576; + +pub(crate) enum ReportInputError { + Bound { maximum: usize, observed: usize }, + Encoding { source: Utf8Error }, +} + +pub(super) fn decode(bytes: &[u8]) -> Result<&str, BenchmarkBaselineError> { + if bytes.len() > MAXIMUM_REPORT_BYTES { + return Err(BenchmarkBaselineError::ReportInput( + ReportInputError::Bound { + maximum: MAXIMUM_REPORT_BYTES, + observed: bytes.len(), + }, + )); + } + std::str::from_utf8(bytes).map_err(|source| { + BenchmarkBaselineError::ReportInput(ReportInputError::Encoding { source }) + }) +} + +impl fmt::Debug for ReportInputError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(self, formatter) + } +} + +impl fmt::Display for ReportInputError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Bound { maximum, observed } => write!( + formatter, + "benchmark report exceeds {maximum} bytes, observed {observed}" + ), + Self::Encoding { .. } => formatter.write_str("benchmark report is not UTF-8"), + } + } +} + +impl Error for ReportInputError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::Encoding { source } => Some(source), + Self::Bound { .. } => None, + } + } +} diff --git a/xtask/src/benchmark_baseline/report_input_tests.rs b/xtask/src/benchmark_baseline/report_input_tests.rs new file mode 100644 index 00000000..eaaeddcb --- /dev/null +++ b/xtask/src/benchmark_baseline/report_input_tests.rs @@ -0,0 +1,55 @@ +//! Laws for report input bounds before decoding or allocation-heavy admission. + +use super::super::report_input::ReportInputError; +use super::{BenchmarkBaselineError, artifact, environment}; + +#[test] +fn oversized_reports_refuse_before_encoding_admission() -> Result<(), Box> { + let maximum = super::super::REPORT_LIMIT; + let observed = maximum + .checked_add(1) + .ok_or("input fixture size overflow")?; + let oversized = vec![0xff; observed]; + assert!(matches!(artifact::validate(&oversized, &environment()), + Err(BenchmarkBaselineError::ReportInput(ReportInputError::Bound { + maximum: actual_maximum, observed: actual_observed, + })) if actual_maximum == maximum && actual_observed == observed)); + Ok(()) +} + +#[test] +fn exact_input_limit_reaches_decoding_without_size_refusal() +-> Result<(), Box> { + let maximum = super::super::REPORT_LIMIT; + let valid_encoding = vec![b'a'; maximum]; + let decoded = super::super::report_input::decode(&valid_encoding)?; + assert_eq!(decoded.len(), maximum); + let invalid_encoding = vec![0xff; maximum]; + assert!( + matches!(super::super::report_input::decode(&invalid_encoding), + Err(BenchmarkBaselineError::ReportInput(ReportInputError::Encoding { source })) + if source.valid_up_to() == 0 && source.error_len() == Some(1)) + ); + Ok(()) +} + +#[test] +fn report_encoding_failure_retains_the_original_utf8_source() +-> Result<(), Box> { + use std::error::Error; + let error = artifact::validate(&[0xff], &environment()) + .err() + .ok_or("invalid encoding was admitted")?; + assert!(matches!(&error, + BenchmarkBaselineError::ReportInput(ReportInputError::Encoding { source }) + if source.valid_up_to() == 0 && source.error_len() == Some(1))); + assert_eq!( + error + .source() + .and_then(Error::source) + .and_then(|source| source.downcast_ref::()) + .map(std::str::Utf8Error::valid_up_to), + Some(0) + ); + Ok(()) +} diff --git a/xtask/src/benchmark_baseline/report_schema.rs b/xtask/src/benchmark_baseline/report_schema.rs new file mode 100644 index 00000000..f64e5909 --- /dev/null +++ b/xtask/src/benchmark_baseline/report_schema.rs @@ -0,0 +1,43 @@ +//! This module owns the frozen streaming-CAS baseline v1 row grammar. +pub(super) const SCENARIO_HEADER: &str = "scenario-header\tname\tverification\tsample-count\tlogical-bytes\tphysical-bytes-read\tphysical-bytes-written\tsource-bytes-read\toutput-bytes-written\tread-amplification-numerator\tread-amplification-denominator\twrite-amplification-numerator\twrite-amplification-denominator\tdeduplication-ratio-numerator\tdeduplication-ratio-denominator\treused-unique-chunks\tchunk-instances\toperation-count\tlogical-bytes-per-second\ttotal-wall-time-ns\tp50-wall-time-ns\tp95-wall-time-ns\tp99-wall-time-ns\ttotal-cpu-time-ns\tp50-cpu-time-ns\tp95-cpu-time-ns\tp99-cpu-time-ns\ttotal-allocation-count\ttotal-allocated-bytes\tpeak-live-allocation-count\tpeak-live-heap-bytes"; +pub(super) const PROFILE_HEADER: &str = "profile-header\tname\tprovenance\tminimum-kib\ttarget-kib\tmaximum-kib\ttimed-input\tsample-count\tlogical-bytes-per-second\ttotal-wall-time-ns\tp50-wall-time-ns\tp95-wall-time-ns\tp99-wall-time-ns\ttotal-cpu-time-ns\ttotal-allocation-count\ttotal-allocated-bytes\tpeak-live-heap-bytes\tbase-unique-chunks\tbase-materialized-bytes\tinsertion-reused-chunks\tdeletion-reused-chunks\tneighbor-reused-chunks"; +pub(super) const METADATA_KEYS: [&str; 16] = [ + "build-profile", + "git-commit", + "git-tree", + "rustc-version", + "target-triple", + "os-description", + "cpu-model", + "cpu-clock", + "peak-memory", + "verification", + "timing-unit", + "byte-unit", + "ratio-encoding", + "logical-cpu-count", + "sample-count", + "warmup-count", +]; +pub(super) const SCENARIO_PREFIXES: [&str; 13] = [ + "scenario\tcold-ingest\tingest-chunk-and-blob-identity", + "scenario\twarm-ingest\tingest-chunk-and-blob-identity", + "scenario\trepeated-near-neighbor-edits\tingest-chunk-and-blob-identity", + "scenario\tearly-insertion\tingest-chunk-and-blob-identity", + "scenario\tearly-deletion\tingest-chunk-and-blob-identity", + "scenario\tmany-tiny-blobs\tingest-chunk-and-blob-identity", + "scenario\tlarge-binary\tingest-chunk-and-blob-identity", + "scenario\thigh-deduplication\tingest-chunk-and-blob-identity", + "scenario\tzero-deduplication\tingest-chunk-and-blob-identity", + "scenario\tsequential-range-reads\tselected-complete-chunks", + "scenario\trandom-range-reads\tselected-complete-chunks", + "scenario\twhole-blob-verification\tchunks-profile-and-blob", + "scenario\tvaried-input-partitioning\tingest-chunk-and-blob-identity", +]; +pub(super) const PROFILE_PREFIXES: [&str; 5] = [ + "profile\tkeep-fastcdc-4-16-64\tkeep.fastcdc-gear64/v1\t4\t16\t64\tlarge-text", + "profile\tkeep-fastcdc-16-64-256\tkeep.fastcdc-gear64/v1\t16\t64\t256\tlarge-text", + "profile\tkeep-fastcdc-64-256-1024\tkeep.fastcdc-gear64/v1\t64\t256\t1024\tlarge-text", + "profile\tfixed-64\tbenchmark.fixed-size/v1\t64\t64\t64\tlarge-text", + "profile\tgit-cas-buzhash-64-256-1024\tgit-cas@432c5d9effb12c9f66536f1386791bb4421f3cea\t64\t256\t1024\tlarge-text", +]; diff --git a/xtask/src/benchmark_baseline/row_mutation_tests.rs b/xtask/src/benchmark_baseline/row_mutation_tests.rs new file mode 100644 index 00000000..d3125401 --- /dev/null +++ b/xtask/src/benchmark_baseline/row_mutation_tests.rs @@ -0,0 +1,83 @@ +//! Model laws for frozen report row order and required catalogs. + +use super::{BenchmarkBaselineError, artifact, environment, report}; + +#[test] +fn every_adjacent_row_transposition_refuses_at_the_first_order_violation() +-> Result<(), Box> { + let environment = environment(); + let valid = report(&environment, 13, 5); + let rows: Vec<_> = valid.lines().collect(); + for (index, pair) in rows.windows(2).enumerate() { + let [first, second] = pair else { + return Err("invalid window fixture".into()); + }; + let successor = index.checked_add(1).ok_or("row index overflow")?; + let mut malformed = String::new(); + for (position, row) in rows.iter().enumerate() { + malformed.push_str(if position == index { + second + } else if position == successor { + first + } else { + row + }); + malformed.push('\n'); + } + let error = artifact::validate(malformed.as_bytes(), &environment) + .err() + .ok_or("transposed rows were admitted")?; + if first.starts_with("schema\t") { + assert!(matches!( + error, + BenchmarkBaselineError::ReportViolation { + reason: "report-schema" + } + )); + } else { + let expected = expected_row(first)?; + assert!(matches!(error, BenchmarkBaselineError::InvalidReportRow { + expected: actual_expected, observed, + } if actual_expected == expected && observed == *second)); + } + } + Ok(()) +} + +#[test] +fn deleting_any_catalog_row_refuses_the_exact_missing_catalog() { + let environment = environment(); + let valid = report(&environment, 13, 5); + for row in valid + .lines() + .filter(|line| line.starts_with("scenario\t") || line.starts_with("profile\t")) + { + let malformed = valid.replace(&format!("{row}\n"), ""); + let expected = if row.starts_with("scenario\t") { + "report-scenario-count" + } else { + "report-profile-count" + }; + assert!( + matches!(artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::ReportViolation { reason }) if reason == expected) + ); + } +} + +fn expected_row(row: &str) -> Result> { + if row.starts_with("metadata\t") { + return row + .split('\t') + .nth(1) + .map(str::to_owned) + .ok_or_else(|| "missing metadata key".into()); + } + if row.starts_with("scenario\t") { + return Ok(row.split('\t').take(3).collect::>().join("\t")); + } + if row.starts_with("profile\t") { + return Ok(row.split('\t').take(7).collect::>().join("\t")); + } + Ok(row.to_owned()) +} diff --git a/xtask/src/benchmark_baseline/tests.rs b/xtask/src/benchmark_baseline/tests.rs index 0d96064c..7c55c8cb 100644 --- a/xtask/src/benchmark_baseline/tests.rs +++ b/xtask/src/benchmark_baseline/tests.rs @@ -170,32 +170,131 @@ fn host() -> CapturedHost { } } +const HISTORICAL_BASELINE: &str = + include_str!("../../../benchmark/baselines/c529c07-aarch64-apple-darwin.tsv"); + fn report(environment: &CapturedEnvironment, scenarios: usize, profiles: usize) -> String { - let mut report = format!( - "schema\tkeep.streaming-cas-baseline/v1\n\ - metadata\tgit-commit\t{}\n\ - metadata\tgit-tree\t{}\n\ - metadata\trustc-version\t{}\n\ - metadata\ttarget-triple\t{}\n\ - metadata\tos-description\t{}\n\ - metadata\tcpu-model\t{}\n\ - metadata\tlogical-cpu-count\t{}\n\ - metadata\tbuild-profile\toptimized-release\n\ - threshold\tall-performance-metrics\tunconfigured\t\ - requires-controlled-baseline-history\n", - environment.commit, - environment.tree, - environment.rustc_version, - environment.target_triple, - environment.host.os_description, - environment.host.cpu_model, - environment.host.logical_cpu_count - ); - for index in 0..scenarios { - let _written = writeln!(report, "scenario\t{index}"); + let admitted_scenarios: Vec<_> = HISTORICAL_BASELINE + .lines() + .filter(|line| line.starts_with("scenario\t")) + .take(scenarios) + .collect(); + let admitted_profiles: Vec<_> = HISTORICAL_BASELINE + .lines() + .filter(|line| line.starts_with("profile\t")) + .take(profiles) + .collect(); + let mut output = String::new(); + for line in HISTORICAL_BASELINE.lines() { + if line.starts_with("scenario\t") && !admitted_scenarios.contains(&line) { + continue; + } + if line.starts_with("profile\t") && !admitted_profiles.contains(&line) { + continue; + } + let _written = writeln!(output, "{line}"); } - for index in 0..profiles { - let _written = writeln!(report, "profile\t{index}"); + output + .replace( + "c529c07f385b5bcd76a4e57c1987001d496f9135", + &environment.commit, + ) + .replace( + "rustc 1.96.0 (ac68faa20 2026-05-25)", + &environment.rustc_version, + ) + .replace( + "metadata\tlogical-cpu-count\t10", + "metadata\tlogical-cpu-count\t1", + ) +} + +#[test] +fn report_admission_refuses_conflicting_and_identical_source_duplicates() { + let environment = environment(); + for commit in [ + &environment.commit, + &String::from("ffffffffffffffffffffffffffffffffffffffff"), + ] { + let mut bytes = report(&environment, 13, 5); + let _written = writeln!(bytes, "metadata\tgit-commit\t{commit}"); + assert!(matches!( + artifact::validate(bytes.as_bytes(), &environment), + Err(BenchmarkBaselineError::DuplicateReportMetadata { coordinate }) + if coordinate == "git-commit" + )); } - report } + +#[test] +fn report_admission_requires_exact_headers_and_canonical_numbers() { + let environment = environment(); + let valid = report(&environment, 13, 5); + let header = super::report_schema::SCENARIO_HEADER; + for (malformed, expected, observed) in [ + ( + valid.replace("scenario-header\tname", "scenario-header\tnames"), + header, + header.replace("\tname\t", "\tnames\t"), + ), + ( + valid.replace( + "scenario\tcold-ingest\tingest-chunk-and-blob-identity\t100", + "scenario\tcold-ingest\tingest-chunk-and-blob-identity\t0100", + ), + "canonical unsigned decimal", + String::from("0100"), + ), + ( + format!("{valid}unexpected\trow\n"), + "end of report", + String::from("unexpected\trow"), + ), + ] { + assert!(matches!( + artifact::validate(malformed.as_bytes(), &environment), + Err(BenchmarkBaselineError::InvalidReportRow { expected: actual_expected, observed: actual_observed }) + if actual_expected == expected && actual_observed == observed + )); + } + assert!(artifact::validate(valid.as_bytes(), &environment).is_ok()); +} + +#[test] +fn malformed_report_rows_cannot_forge_diagnostic_lines() { + let error = BenchmarkBaselineError::InvalidReportRow { + expected: "canonical unsigned decimal", + observed: String::from("1\nforged\tvalue"), + }; + assert_eq!( + error.to_string(), + "benchmark report expected canonical unsigned decimal, observed `1\\nforged\\tvalue`" + ); +} + +#[path = "numeric_format_tests.rs"] +mod numeric_format_tests; + +#[path = "metadata_policy_tests.rs"] +mod metadata_policy_tests; + +#[path = "metric_relation_tests.rs"] +mod metric_relation_tests; + +#[path = "counter_width_tests.rs"] +mod counter_width_tests; + +#[path = "report_input_tests.rs"] +mod report_input_tests; + +#[path = "row_mutation_tests.rs"] +mod row_mutation_tests; + +#[path = "report_compatibility_tests.rs"] +mod report_compatibility_tests; + +#[path = "catalog_membership_tests.rs"] +mod catalog_membership_tests; + +#[path = "completeness_tests.rs"] +mod completeness_tests; diff --git a/xtask/src/benchmark_report_fuzz.rs b/xtask/src/benchmark_report_fuzz.rs new file mode 100644 index 00000000..c7bd5017 --- /dev/null +++ b/xtask/src/benchmark_report_fuzz.rs @@ -0,0 +1,52 @@ +//! This module owns the I/O-free production benchmark-report fuzz facade. + +#[path = "benchmark_baseline/artifact.rs"] +mod artifact; +#[path = "benchmark_baseline/captured_environment.rs"] +mod captured_environment; +#[path = "benchmark_baseline/counter_widths.rs"] +mod counter_widths; +#[allow( + dead_code, + reason = "pure admission reuses the complete task error boundary" +)] +#[path = "benchmark_baseline/error.rs"] +mod error; +#[path = "benchmark_baseline/metadata_policy.rs"] +mod metadata_policy; +#[path = "benchmark_baseline/metadata_uniqueness.rs"] +mod metadata_uniqueness; +#[path = "benchmark_baseline/metric_error.rs"] +mod metric_error; +#[path = "benchmark_baseline/metric_relations.rs"] +mod metric_relations; +#[path = "benchmark_baseline/report_grammar.rs"] +mod report_grammar; +#[path = "benchmark_baseline/report_input.rs"] +mod report_input; +#[path = "benchmark_baseline/report_schema.rs"] +mod report_schema; + +use captured_environment::{CapturedEnvironment, CapturedHost}; +use error::BenchmarkBaselineError; + +mod environment { + pub(super) use super::captured_environment::CapturedEnvironment; +} + +pub(super) fn admit(input: &[u8]) -> Result<(), BenchmarkBaselineError> { + let environment = CapturedEnvironment { + commit: String::from("c529c07f385b5bcd76a4e57c1987001d496f9135"), + tree: "clean", + rustc_version: String::from("rustc 1.96.0 (ac68faa20 2026-05-25)"), + target_triple: String::from("aarch64-apple-darwin"), + host: CapturedHost { + os_description: String::from("Darwin 25.3.0 arm64"), + cpu_model: String::from("Apple M1 Pro"), + logical_cpu_count: std::num::NonZeroUsize::MIN.saturating_add(9), + }, + }; + artifact::validate(input, &environment).map(|report| { + let _admitted_bytes = report.bytes(); + }) +} diff --git a/xtask/src/benchmark_report_fuzz_tests.rs b/xtask/src/benchmark_report_fuzz_tests.rs new file mode 100644 index 00000000..ec5cf18a --- /dev/null +++ b/xtask/src/benchmark_report_fuzz_tests.rs @@ -0,0 +1,35 @@ +//! Laws for the production report fuzz entry point and its fixed corpus seed. + +use crate::{BenchmarkReportAdmission, admit_benchmark_report}; + +const SEED: &[u8] = include_bytes!("../../benchmark/baselines/c529c07-aarch64-apple-darwin.tsv"); + +#[test] +fn historical_seed_reaches_full_production_admission() { + assert_eq!( + admit_benchmark_report(SEED), + BenchmarkReportAdmission::Admitted + ); + assert_eq!( + admit_benchmark_report(b"not a report\n"), + BenchmarkReportAdmission::Refused + ); +} + +#[test] +fn corpus_byte_mutations_produce_repeatable_admission_and_refusal() { + for (position, original) in SEED.iter().enumerate().step_by(17) { + let mut mutated = SEED.to_vec(); + if let Some(value) = mutated.get_mut(position) { + *value = original ^ 0x80; + } + assert_eq!( + admit_benchmark_report(&mutated), + BenchmarkReportAdmission::Refused + ); + assert_eq!( + admit_benchmark_report(&mutated), + BenchmarkReportAdmission::Refused + ); + } +} diff --git a/xtask/src/fuzz_campaign/target/tests.rs b/xtask/src/fuzz_campaign/target/tests.rs index b43744d5..65303aff 100644 --- a/xtask/src/fuzz_campaign/target/tests.rs +++ b/xtask/src/fuzz_campaign/target/tests.rs @@ -23,6 +23,7 @@ fn checked_in_harness_set_is_exact_and_sorted() -> Result<(), Box> { assert_eq!( targets.iter().map(FuzzTarget::as_str).collect::>(), [ + "benchmark_report", "blob_hasher", "blob_id_binary", "blob_id_text", diff --git a/xtask/src/fuzz_seed_corpus.rs b/xtask/src/fuzz_seed_corpus.rs index 3635322a..72d995dc 100644 --- a/xtask/src/fuzz_seed_corpus.rs +++ b/xtask/src/fuzz_seed_corpus.rs @@ -1,5 +1,6 @@ //! This module owns deterministic fuzz seed recipes and materialization. +mod benchmark_report_seeds; mod catalog_seeds; mod cdc_seeds; mod filesystem; @@ -68,6 +69,7 @@ impl Seed { pub(super) fn prepare(repository_root: &Path) -> Result<(), FuzzSeedError> { let files = RepositoryFiles::open(repository_root)?; let mut seeds = identity_seeds::seeds(&files)?; + seeds.extend(benchmark_report_seeds::seeds()?); seeds.extend(catalog_seeds::seeds(&files)?); seeds.extend(cdc_seeds::seeds()?); seeds.extend(golden_protocol_seeds_from(&files)?); diff --git a/xtask/src/fuzz_seed_corpus/benchmark_report_seeds.rs b/xtask/src/fuzz_seed_corpus/benchmark_report_seeds.rs new file mode 100644 index 00000000..44cc6f66 --- /dev/null +++ b/xtask/src/fuzz_seed_corpus/benchmark_report_seeds.rs @@ -0,0 +1,14 @@ +//! This module owns the source-bound historical benchmark-report seed. + +use super::{FuzzSeedError, Seed}; + +const REPORT: &[u8] = + include_bytes!("../../../benchmark/baselines/c529c07-aarch64-apple-darwin.tsv"); + +pub(super) fn seeds() -> Result, FuzzSeedError> { + Ok(vec![Seed::new( + "benchmark_report", + "canonical-v1", + REPORT.to_vec(), + )?]) +} diff --git a/xtask/src/fuzz_seed_corpus/tests/materialization.rs b/xtask/src/fuzz_seed_corpus/tests/materialization.rs index 37dd51dc..8ea245d1 100644 --- a/xtask/src/fuzz_seed_corpus/tests/materialization.rs +++ b/xtask/src/fuzz_seed_corpus/tests/materialization.rs @@ -47,7 +47,8 @@ fn seed_preparation_materializes_the_complete_deterministic_set() prepare(root)?; let corpus = root.join("fuzz/corpus"); let first = seed_contents(&corpus)?; - assert_eq!(first.len(), 46); + assert_eq!(first.len(), 47); + assert_eq!(target_seed_count(&first, "benchmark_report/"), 1); assert_eq!(target_seed_count(&first, "catalog_format/"), 6); assert_eq!(target_seed_count(&first, "golden_protocol/"), 9); assert_eq!(target_seed_count(&first, "layout_record/"), 4); diff --git a/xtask/src/lib.rs b/xtask/src/lib.rs index f2d9786e..dc75384e 100644 --- a/xtask/src/lib.rs +++ b/xtask/src/lib.rs @@ -6,7 +6,7 @@ #[cfg(feature = "golden-protocol-fuzz")] extern crate self as xtask; -#[cfg(feature = "golden-protocol-fuzz")] +#[cfg(any(feature = "golden-protocol-fuzz", feature = "benchmark-report-fuzz"))] mod diagnostic; #[cfg(feature = "golden-protocol-fuzz")] @@ -113,3 +113,38 @@ pub fn admit_repository_json(input: &[u8]) -> RepositoryJsonAdmission { RepositoryJsonAdmission::Refused } } + +#[cfg(feature = "benchmark-report-fuzz")] +#[allow( + clippy::redundant_pub_crate, + reason = "the facade hides the production parser implementation" +)] +mod benchmark_report_fuzz; + +/// Whether the production benchmark-report parser admitted the fuzz input. +#[cfg(feature = "benchmark-report-fuzz")] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum BenchmarkReportAdmission { + /// The input is a complete canonical report for the fixed historical seed. + Admitted, + /// The production parser refused the input. + Refused, +} + +/// Exercises bounded production report admission without host capture or I/O. +/// +/// Source coordinates are fixed to the committed historical fuzz seed. No +/// filesystem, process, clock or network operation occurs. Input exceeding one +/// MiB refuses before decoding; admission does not publish any artifact. +#[cfg(feature = "benchmark-report-fuzz")] +#[must_use] +pub fn admit_benchmark_report(input: &[u8]) -> BenchmarkReportAdmission { + if benchmark_report_fuzz::admit(input).is_ok() { + BenchmarkReportAdmission::Admitted + } else { + BenchmarkReportAdmission::Refused + } +} + +#[cfg(all(test, feature = "benchmark-report-fuzz"))] +mod benchmark_report_fuzz_tests;