diff --git a/libs/@local/graph/atlas/benches/backfill_walk.rs b/libs/@local/graph/atlas/benches/backfill_walk.rs index 0d4fd995406..8d1fc0eecfd 100644 --- a/libs/@local/graph/atlas/benches/backfill_walk.rs +++ b/libs/@local/graph/atlas/benches/backfill_walk.rs @@ -5,15 +5,22 @@ //! delivery first. The decision between them is per-tile selection time as the zoom deepens, and //! this target produces exactly that curve. //! -//! Before the timed groups, one report prints the full sweep of both variants across mask shape -//! (independent rows hidden versus whole spatial blocks), visible fraction, and zoom along the -//! fixture's densest descent path, with scan counts, per-tile medians, and the independent -//! variant's re-delivery census (the crowding the chained variant exists to remove). The timed -//! groups then pin the decision points: both variants at the root and at the deepest zoom under the -//! adversarial mask. +//! Before the timed group, the calibration reports print, served-engine tables first. Each report +//! compares fill rules or delivery engines over the fixture's densest descent path or an audit tile +//! set, printing counts, ratios, and median-of-five wall times rather than Criterion statistics. +//! The per-scope tables rebuild the corpus at a quarter, one, and four times the configured scale. +//! The last report is the sweep of both walk variants across mask shape (independent rows hidden +//! versus whole spatial blocks), visible fraction, and zoom along the descent path, with scan +//! counts, per-tile medians, and the independent variant's re-delivery census (the crowding the +//! chained variant exists to remove). The timed group then pins the decision points at the root +//! and at the deepest zoom under the adversarial mask: both walk variants, the coverage rules, the +//! rank and refined rules, and the recommended budget over the scanning and served engines. //! -//! The corpus defaults to 300,000 points so a sweep stays in seconds; set `ATLAS_BACKFILL_POINTS` -//! for other scales. Wall time depends on the host: compare numbers within one machine, not across. +//! The corpus defaults to 300,000 points so a sweep stays in seconds. `ATLAS_BACKFILL_POINTS` sets +//! other scales. `ATLAS_SCOPE_CASCADE_ONLY` narrows the run to the per-scope cascade-build table. +//! `ATLAS_DENSITY_CLOSURE_ONLY` runs the density-closure reports, which no other run prints, with +//! their interleaved cost table added when `ATLAS_DENSITY_CLOSURE_TIMED` is also set. Wall time +//! depends on the host: compare numbers within one machine, not across. #![expect( clippy::print_stdout, clippy::float_arithmetic, @@ -52,7 +59,10 @@ const REPETITIONS: usize = 5; /// The median sample's index once the five sort. const MEDIAN: usize = 2; -/// The visible fractions every table sweeps. +/// The visible fractions the broad tables sweep. +/// +/// A narrower table names its own subset, and [`sweep`] runs fractions of its own, down to one +/// percent. const FRACTIONS: [f64; 5] = [1.0, 0.75, 0.5, 0.25, 0.05]; /// The mask shapes every table sweeps: independent rows, then whole spatial blocks. @@ -122,6 +132,11 @@ fn remask(bench: &mut WalkBench, clustered: bool, visible: f64) { } } +/// Returns the corpus scale: `ATLAS_BACKFILL_POINTS`, or [`DEFAULT_POINTS`]. +/// +/// # Panics +/// +/// This panics when `ATLAS_BACKFILL_POINTS` is set to a value that is not a point count. fn points() -> usize { std::env::var("ATLAS_BACKFILL_POINTS").map_or(DEFAULT_POINTS, |value| { value @@ -214,7 +229,11 @@ impl CrossCheck { } } -/// Widens a count for signed comparison. +/// Converts a count for signed comparison. +/// +/// # Panics +/// +/// Panics if `value` is at least 2⁶³ on a target whose `usize` can represent it. fn count(value: usize) -> i64 { i64::try_from(value).expect("corpus counts fit i64") } @@ -346,6 +365,10 @@ const RULES: [FillRule; 4] = [ ]; /// Returns the cell at one tile coordinate. +/// +/// # Panics +/// +/// This panics when `z` exceeds the key width or `(x, y)` lies off the zoom's grid. const fn cell_of(z: u8, x: u32, y: u32) -> MortonCell { MortonCell::new( Depth::new(z).expect("tile zooms lie within the key width"), @@ -421,7 +444,10 @@ fn density(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { } } -/// Prints per-tile selection cost: the chained variant against the coverage variant. +/// Prints per-tile selection cost for four deliveries of one tile. +/// +/// The chained walk and today's unmasked rule deliver the same points by two code paths. The +/// coverage rule and the cell repair come after them. fn selection_cost(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { println!( "\nper-tile selection cost, median of {REPETITIONS}: {} points", @@ -567,6 +593,9 @@ fn density_summary(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { } /// Prints the saturation census: how often each rule's chain runs short. +/// +/// `zero cov` counts the opposite tiles, where the chain has already delivered more points than the +/// tile's cut holds covered cells and the coverage target is nothing. fn saturation(bench: &mut WalkBench, tiles: &[(u8, u32, u32)]) { println!( "\nsaturation census over {} tiles: chains ending below their target", @@ -663,6 +692,12 @@ fn pyramid_profile(bench: &mut WalkBench) { } /// Prints pyramid construction cost and footprint across corpus scales and mask shapes. +/// +/// The visible cascade's own build time stands beside them. +/// +/// # Panics +/// +/// Panics if a scale is zero or exceeds [`WalkBench::build`]'s row domain. fn pyramid_cost(scales: &[usize]) { println!("\npyramid construction, median of {REPETITIONS}"); println!( @@ -723,6 +758,15 @@ fn pyramid_cost(scales: &[usize]) { } /// Prints target query cost over a built pyramid. +/// +/// `query ns` times one count query against the tile's cut. `chain ns` projects a chain's cost by +/// multiplying that time by the `z + 1` levels its targets read. Only path entries at zooms +/// divisible by six contribute rows. +/// +/// # Panics +/// +/// Panics if the schedule's deepest cut exceeds the key width, or if an evaluated tile is off its +/// grid, beyond the schedule's maximum zoom or at a cut beyond the key width. fn query_cost(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { /// Queries per timed batch. const BATCH: usize = 10_000; @@ -805,7 +849,7 @@ struct Tally { dots: usize, /// Points the tiles deliver themselves. delivered: usize, - /// Tiles whose own delivery passes the budget. + /// Tiles whose own delivery passes [`BUDGET`], whatever budget the rule itself carries. over: usize, /// Refinement levels summed over the tiles. refined: usize, @@ -937,8 +981,12 @@ fn dot_count(rules: &[FillRule], rows: &[(bool, f64, Vec)], tiles: usize) /// Prints the noninterference census: delivered rows over two corpora sharing one visible view. /// -/// Corpus B is the masked fixture. Corpus A contains the same visible rows and nothing else. A rule +/// Corpus A is the masked fixture. Corpus B contains the same visible rows and nothing else. A rule /// whose delivery is a function of the visible view alone delivers equal rows over both. +/// +/// # Panics +/// +/// This panics when the two corpora's visible columns differ in length. fn noninterference(bench: &mut WalkBench, tiles: &[(u8, u32, u32)], rules: &[FillRule]) { println!( "\nnoninterference: delivered row identities, masked corpus against visible-only corpus, \ @@ -1219,10 +1267,14 @@ fn served_identity(bench: &mut WalkBench, tiles: &[(u8, u32, u32)], rules: &[Fil /// Prints the noninterference census over the served engine. /// -/// The generation derives from the visible entries alone and cascades over them alone, so a served -/// delivery must agree row for row across the two corpora exactly as the scanning form does. A row +/// The generation derives from the visible entries alone and cascades over them alone. A served +/// delivery must agree row for row across the two corpora, exactly as the scanning form does. A row /// with a nonzero `differ` under a hidden-independent rule means the artifact reintroduced a /// dependence on the hidden rows. +/// +/// # Panics +/// +/// This panics when the two corpora's generations differ in length. fn served_noninterference(bench: &mut WalkBench, tiles: &[(u8, u32, u32)], rules: &[FillRule]) { println!( "\nnoninterference, served engine: masked corpus against visible-only corpus, {} tiles \ @@ -1282,8 +1334,13 @@ fn served_noninterference(bench: &mut WalkBench, tiles: &[(u8, u32, u32)], rules /// Prints the served engine's cells-falsely-empty tally against the independent ground truth. /// /// `empty` counts cut cells holding a visible point that the served cumulative delivery does not -/// occupy, and `truth gaps` counts tiles where the served delivery's own cell set differs from +/// occupy, and `truth gaps` counts tiles where that cumulative delivery's cell set differs from /// [`WalkBench::occupied_cells`], which reads the corpus and the mask alone. +/// +/// # Panics +/// +/// Panics if a rule is outside the rank-representative family, or if a requested tile lies off its +/// grid, beyond the schedule's maximum zoom or at a cut beyond the key width. fn served_density(bench: &mut WalkBench, tiles: &[(u8, u32, u32)], rules: &[FillRule]) { println!( "\nserved delivery against the independent cell ground truth, {} tiles per row", @@ -1345,8 +1402,9 @@ fn served_density(bench: &mut WalkBench, tiles: &[(u8, u32, u32)], rules: &[Fill /// Prints the ladder over both engines at the recommended budget. /// -/// `k` and `deepened` describe the grid each engine resolved; equal columns mean the served form -/// resolved the same grid tile for tile, and the unmasked rows carry today's own ladder. +/// `k` and `deepened` describe the grid each engine resolved. Equal columns mean the served form +/// resolved the same grid tile for tile, and the `today` columns carry today's own ladder beside +/// them. fn served_ladder(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { println!("\nthe ladder over both engines, budget {}", BUDGET / 4); println!( @@ -1405,7 +1463,9 @@ fn served_ladder(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { } } -/// Prints per-tile selection cost over both engines, as ratios to today's chained walk. +/// Prints per-tile selection cost over both engines. +/// +/// The costs appear as ratios to today's chained walk and to each other. fn served_cost(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { println!( "\nper-tile selection cost over both engines, median of {REPETITIONS}: {} points", @@ -1489,11 +1549,17 @@ fn served_cost(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { } } -/// Prints where a served delivery's own cost goes. +/// Prints the components of a served delivery's own cost. /// /// `coarse` delivers at the cut depth alone, `whole` adds the refinement search, `morton` adds the /// partial refinement without a population order, and `pop` adds the population index searches. -/// `read` times one bare prefix read of the tile's own cut grid, without the chain. +/// `read` times one bare prefix read of the tile's own cut grid, without the chain. Only path +/// entries at zooms divisible by six contribute rows. +/// +/// # Panics +/// +/// Panics if an evaluated tile lies off its grid, beyond the schedule's maximum zoom or at a cut +/// beyond the key width. fn served_breakdown(bench: &mut WalkBench, path: &[(u8, u32, u32)]) { println!( "\nserved cost by component, median of {REPETITIONS}: {} points", @@ -1596,14 +1662,23 @@ enum ScopeConstruction { /// Interleaved build medians and paired ratios for one scope. struct ScopeBuildTimes { + /// The separated build's median microseconds. separated: f64, + /// The merged construction's median microseconds. merged: f64, + /// The filtered construction's median microseconds. filtered: f64, + /// The indexed construction's median microseconds. indexed: f64, + /// The radix construction's median microseconds. radix: f64, + /// The median of adjacent merged-to-separated ratios. merged_ratio: f64, + /// The median of adjacent filtered-to-separated ratios. filtered_ratio: f64, + /// The median of adjacent indexed-to-separated ratios. indexed_ratio: f64, + /// The median of adjacent radix-to-separated ratios. radix_ratio: f64, } @@ -1632,7 +1707,9 @@ fn scope_construction_micros(bench: &WalkBench, construction: ScopeConstruction) }) } -/// Returns the median of a nonempty fixed-size sample. +/// Returns the middle of a nonempty fixed-size sample once sorted. +/// +/// At even `N` this is the upper of the two middle values. /// /// # Panics /// @@ -1658,6 +1735,7 @@ fn record_sample(samples: &mut [f64; N], index: usize, value: f6 /// Measures every construction adjacent to its own separated-build baseline. fn interleaved_scope_builds(bench: &WalkBench) -> ScopeBuildTimes { + /// The constructions in rotation order. const CONSTRUCTIONS: [ScopeConstruction; 4] = [ ScopeConstruction::Merged, ScopeConstruction::Filtered, @@ -1732,6 +1810,11 @@ fn interleaved_scope_builds(bench: &WalkBench) -> ScopeBuildTimes { } /// Prints per-scope cascade build cost for the five exact constructions. +/// +/// # Panics +/// +/// This panics when a merged, filtered, indexed, or radix artifact differs from the separated +/// build's, or when masking moved a bucket deeper. fn scope_cascade_cost(scales: &[usize]) { println!("\nper-scope masked cascade build, median of {REPETITIONS}"); println!( @@ -1801,7 +1884,7 @@ fn scope_cascade_cost(scales: &[usize]) { /// Prints the space, build, and serve trade across every artifact form. /// -/// `b/row` is bytes per visible row. The scanning form serves out of the Morton column; the served +/// `b/row` is bytes per visible row. The scanning form serves out of the Morton column. The served /// form serves out of a generation, whose shared layout drops the key column and reads keys through /// the corpus base column instead. The position column alone is the leanest form that answers /// populations, and the re-cascade is the artifact the equivalence proof names. @@ -2211,6 +2294,10 @@ const fn density_rule_name(rule: DensityRule) -> &'static str { } /// Delivers the whole rendered world at one zoom under one rule. +/// +/// # Panics +/// +/// This panics when two tiles of the zoom deliver the same position. fn world_delivery( bench: &WalkBench, rule: DensityRule, @@ -2253,6 +2340,15 @@ fn world_delivery( } /// Counts delivered positions in equal-area windows. +/// +/// Allocates one count for every cell of the grid at `depth`. The grid's shift and allocation size +/// must be representable, and positions must convert losslessly to [`usize`]. +/// +/// # Panics +/// +/// Panics if a converted position lies outside `codes`, a window index does not fit the count +/// vector, or the vector's capacity overflows. With overflow checking enabled, a grid shift at or +/// above [`usize::BITS`] also panics. fn delivered_window_counts(codes: &[u64], positions: &[u32], depth: Depth) -> Vec { let mut counts = vec![0_usize; 1_usize << (2 * u32::from(depth.get()))]; for &position in positions { @@ -2268,6 +2364,16 @@ fn delivered_window_counts(codes: &[u64], positions: &[u32], depth: Depth) -> Ve } /// Counts occupied cells of one truth grid in equal-area windows. +/// +/// Allocates one count for every cell of the grid at `window_depth`. The grid's shift and +/// allocation size must be representable. +/// +/// # Panics +/// +/// Panics if `window_depth` exceeds `occupied_depth`, a window index does not fit the count vector, +/// or the vector's capacity overflows. With overflow checking enabled, a vector-length shift at or +/// above [`usize::BITS`] panics. Shifting an occupied cell by 64 also panics in that mode, which +/// occurs at occupied depth 32 and window depth zero when a cell is visited. fn occupied_window_counts( bench: &WalkBench, occupied_depth: Depth, @@ -2286,6 +2392,13 @@ fn occupied_window_counts( } /// Returns one window's Morton index. +/// +/// `x` and `y` must lie on the grid at `depth`. +/// +/// # Panics +/// +/// Panics if the Morton prefix does not fit [`usize`]. With overflow checking enabled, depth zero +/// also panics: shifting a coordinate by 32 exceeds its [`u32`] shift range. fn window_index(depth: Depth, x: u32, y: u32) -> usize { let shift = 32 - u32::from(depth.get()); usize::try_from(MortonKey::new(x << shift, y << shift).prefix(depth)) @@ -2310,6 +2423,22 @@ struct DensityFit { } /// Fits one shown distribution to one public occupancy grid. +/// +/// The windows are the cells of `window_depth`, which is at least `render_zoom`, and each rendered +/// tile spans `2^(window_depth - render_zoom)` windows per side. A rendered tile boundary's +/// contrast is the normalized difference of the sampling ratios on its two sides, +/// `|a - b| / (a + b)`. A boundary touching a window with no occupied cells contributes no +/// sample at all, and two zero ratios, whose denominator vanishes, count as zero contrast. +/// +/// Both slices must cover the complete window grid in Morton-prefix order. Their positive totals +/// must fit [`usize`]. +/// +/// # Panics +/// +/// Panics if the slice lengths differ, a computed total is zero, or a boundary comparison indexes a +/// missing window. Index conversion also has [`window_index`]'s panic conditions. With overflow +/// checking enabled, overflow in the count sums, depth arithmetic or percentile-index arithmetic +/// panics. fn density_fit( shown: &[usize], occupied: &[usize], @@ -2398,6 +2527,16 @@ fn density_fit( } /// Finds the public occupancy grid whose spatial distribution best fits the shown dots. +/// +/// `shown` supplies the complete window grid required by [`density_fit`]. The window depth must not +/// exceed the first searched grid, [`WalkBench::uniform_grid_depth`] at `render_zoom` with no +/// additional depth. +/// +/// # Panics +/// +/// Panics if `render_zoom` lies outside the bench's schedule or the window grid is finer than the +/// first searched grid. It also propagates the grid-counting and comparison panics documented by +/// [`occupied_window_counts`] and [`density_fit`]. fn best_density_fit( bench: &WalkBench, shown: &[usize], @@ -2426,7 +2565,16 @@ fn best_density_fit( best.expect("the public-grid search checks seven depths") } -/// Counts adjacent R4 tiles whose whole-grid depth differs and tiles with a mixed grid. +/// Counts the boundaries between adjacent R4 tiles whose whole-grid depth differs. +/// +/// Returns those boundaries, the boundaries compared, and the tiles holding a mixed grid. The grid +/// arithmetic requires `z < 16` and a tile count representable by [`usize`]. +/// +/// # Panics +/// +/// Panics if a grid count or index does not fit [`usize`], or if the zoom exceeds the schedule's +/// maximum or gives a cut beyond the key width. With overflow checking enabled, `z >= 16` panics in +/// the [`u32`] side or area arithmetic. fn budget_grid_seams( bench: &WalkBench, z: u8, @@ -2472,7 +2620,7 @@ fn budget_grid_seams( (different, edges, mixed) } -/// One rule's density audit over the rendered zooms. +/// One rule's density audit over the audited zooms. #[derive(Debug)] struct RuleDensityAudit { /// The worst fit any audited zoom produced. @@ -2491,10 +2639,12 @@ struct RuleDensityAudit { grid_edges: usize, } -/// Audits one rule's fit to the best public grid at every rendered zoom. +/// Audits one rule's fit to the best public grid at zooms 0 through 3. +/// +/// The audit renders the whole world of each of those zooms tile by tile. /// /// The coarse and uniform rules are also held to their own grid's density metric: each renders what -/// its public grid occupies, so the total variation against that grid is zero. +/// its public grid occupies, and the total variation against that grid is zero. /// /// # Panics /// @@ -2576,7 +2726,13 @@ fn rule_density_audit( } /// Prints the best public-grid fit and boundary seam contrast for every rule. +/// +/// # Panics +/// +/// This panics when today's rule or the per-tile budget refinement fits its best public grid +/// exactly in every sweep cell, which would leave the metric separating nothing. fn proportional_density(bench: &mut WalkBench) { + /// The rules in table order. const RULES: [DensityRule; 5] = [ DensityRule::Today, DensityRule::Floor, @@ -2684,6 +2840,11 @@ struct UniformDensityCounts { } /// Counts the dots each public-grid law delivers over `tiles`. +/// +/// # Panics +/// +/// This panics when `bench.span()` is zero, because the staircase law divides each tile's zoom by +/// the span. fn uniform_density_counts_of( bench: &WalkBench, tiles: &[(u8, u32, u32)], @@ -2771,6 +2932,11 @@ fn uniform_density_counts_of( } /// Prints dot counts and the public uniform grid's geometric response bound. +/// +/// # Panics +/// +/// This panics when the coverage-rank and cut-only public grid counts differ, or when the terminal +/// public grid omits visible rows. fn uniform_density_counts(bench: &mut WalkBench, tiles: &[(u8, u32, u32)]) { println!("\nuniform public-grid dot count over {} tiles", tiles.len()); println!( @@ -2895,6 +3061,7 @@ struct PairTimes { /// Measures a candidate adjacent to a fresh baseline with alternating arm order. fn paired_micros(mut baseline: impl FnMut(), mut candidate: impl FnMut()) -> PairTimes { + /// Calls averaged into one timed sample. const CALLS_PER_SAMPLE: usize = 100; let mut baselines = [0.0_f64; REPETITIONS]; @@ -3070,6 +3237,10 @@ fn reports(bench: &mut WalkBench, path: &[(u8, u32, u32)], tiles: &[(u8, u32, u3 sweep(bench, path); } +/// Runs the calibration reports and then the timed group. +/// +/// `ATLAS_SCOPE_CASCADE_ONLY` narrows the run to the per-scope cascade-build table, and +/// `ATLAS_DENSITY_CLOSURE_ONLY` to the density-closure reports, which a default run omits. fn benches(criterion: &mut Criterion) { if std::env::var_os("ATLAS_SCOPE_CASCADE_ONLY").is_some() { scope_cascade_cost(&[points() / 4, points(), points() * 4]); @@ -3090,8 +3261,9 @@ fn benches(criterion: &mut Criterion) { timings(criterion, &mut bench, (deep_z, deep_x, deep_y)); } -/// Times the decision points: the rules at the root and at the deepest zoom under the adversarial -/// mask. +/// Times the decision points. +/// +/// The rules run at the root and at the deepest zoom under the adversarial mask. fn timings(criterion: &mut Criterion, bench: &mut WalkBench, deep: (u8, u32, u32)) { let (deep_z, deep_x, deep_y) = deep; // The decision points are both variants at the root and at the deepest zoom, under the diff --git a/libs/@local/graph/atlas/benches/math_kernels.rs b/libs/@local/graph/atlas/benches/math_kernels.rs index b187b08f010..128fe122608 100644 --- a/libs/@local/graph/atlas/benches/math_kernels.rs +++ b/libs/@local/graph/atlas/benches/math_kernels.rs @@ -1,25 +1,42 @@ -//! Wall-time and hardware-counter benchmarks for the math kernels. +//! Math-kernel comparisons using wall time or calling-thread counters. //! -//! Every benchmark pairs a SIMD kernel with the scalar formulation it replaces, so the -//! vectorization claims in the module docs stay tied to measured numbers rather than -//! emitted-assembly inspection alone. +//! The suite compares selected vector kernels with scalar references, serial reductions with +//! parallel forms, and production transcendental wrappers with experimental table kernels. Fixtures +//! and surrounding benchmark code determine what each measurement includes. //! -//! The measurement defaults to retired instructions, and `MATH_BENCH_EVENT`, a comma-separated -//! list, selects it: every listed measurement runs the whole suite once in list order. The event -//! is suffixed into the benchmark ids (`kernel@cycles/exp_f32x8`), so each event keeps its own -//! statistics lineage and run-over-run change reports compare like with like. The events: -//! `instructions`, `cycles`, `branch-mispredictions`, `l1d-cache-misses`, `backend-stalls` (cycles -//! the scheduler issued nothing because execution was waiting, the direct view of dependency-chain -//! latency), and `simd-instructions` (retired vector ALU operations, loop scaffolding filtered -//! out). `wall-time` selects criterion's default wall-clock measurement instead, and it needs no -//! elevated privileges. A list containing any other event runs under sudo. Instruction counts are -//! stable across runs but blind to instruction-level parallelism, so confirm a winner in `cycles` -//! before acting on close calls, and weigh instruction counts higher for kernels that run fused -//! inside larger loops, where issue slots are the shared resource. +//! `MATH_BENCH_EVENT` is a comma-separated list. It defaults to `instructions`, including when the +//! list is empty. Each event runs the suite once in list order. Event suffixes such as +//! `kernel@cycles/exp_f32x8` distinguish measurement names in the benchmark IDs. //! -//! Counters attribute to the calling thread only: rayon-parallel benchmarks under-report every -//! event because the workers' counts are invisible. Read parallel entries as coordination overhead, -//! not as the work itself. +//! # Measurements +//! +//! On supported Apple Silicon macOS systems, counter names select these events: +//! +//! - `instructions`: retired instructions. +//! - `cycles`: CPU cycles. +//! - `branch-mispredictions`: retired branch mispredictions. +//! - `l1d-cache-misses`: retired L1 data-cache miss loads. +//! - `backend-stalls`: no operation issued due to the backend, available on M4 and M5. +//! - `simd-instructions`: retired non-load/store vector Advanced SIMD instructions, available on M2 +//! through M5. +//! +//! The program uses its existing privileges for counter access, which requires root privileges or +//! the kernel's kpc entitlement. `wall-time` uses Criterion's wall-clock measurement without +//! counter access. On non-macOS platforms the counter backend also measures wall time, even when an +//! ID carries a counter-event suffix. +//! +//! Instruction counts do not measure instruction-level parallelism and can vary with executed paths +//! and allocator state. Compare cycles as well as instructions on the same machine and workload. +//! Backend stalls aggregate backend causes rather than isolate one dependency chain. The +//! vector-instruction event omits loads and stores and is not a count of all SIMD work. +//! +//! Hardware counts cover the calling thread, including any work it executes inside a parallel +//! operation. Counts from other Rayon workers are omitted. Use `wall-time` to compare the +//! completion times of serial and parallel reductions. +//! +//! # Running the suite +//! +//! These shell commands select counters or wall time: //! //! ```text //! sudo MATH_BENCH_EVENT=cycles,backend-stalls cargo bench -p hash-graph-atlas --features bench --bench math_kernels @@ -39,9 +56,10 @@ use core::{hint::black_box, time::Duration}; use codspeed_criterion_compat::{Criterion, Throughput, measurement::Measurement}; use hash_graph_atlas::bench::{kernel, math}; +/// The embedding width the vector kernels run at. const EMBEDDING_DIMENSIONS: usize = 512; -/// Deterministic, sign-varying components. +/// Generates a repeating sequence of sign-varying components plus `offset`. fn scattered(offset: f32) -> [f32; N] { core::array::from_fn(|index| { let value = f32::from(u8::try_from(index % 200).expect("bounded by modulus")); @@ -50,6 +68,7 @@ fn scattered(offset: f32) -> [f32; N] { }) } +/// Measures dot product and cosine distance at the embedding width. fn bench_vecn(criterion: &mut Criterion, event: &str) { let pair = math::vecn_pair( scattered::(0.5), @@ -72,6 +91,7 @@ fn bench_vecn(criterion: &mut Criterion, event: &str) { group.finish(); } +/// Measures batched affinity gradients and the reference-point fit. fn bench_affinity(criterion: &mut Criterion, event: &str) { let state = math::affinity_state( 1.577, @@ -103,12 +123,13 @@ fn bench_affinity(criterion: &mut Criterion, event: ); } -/// Scalar-libm baselines for the gradient kernels' `pow` composition. +/// Measures scalar powers alone and inside a synthetic rational coefficient. /// -/// The affinity gradients need `d^(2b)` for four lanes with a shared exponent, and they tolerate a -/// few ulps. The production choice is the vendored `exp2(p * log2(d))` composition (measured as -/// `kernel/pow_f32x4`); these entries keep its scalar-libm alternative measured in the same -/// isolated and fused-in-coefficient forms. +/// The scalar comparison applies [`f32::powf`] to four lanes with exponent 0.895. The coefficient +/// fixture evaluates −2abP/(1 + aPρ), with ρ the squared distance, a = 1.577, b = 0.895 and P = ρᵇ. +/// The attractive gradient uses this form with P = ρᵇ⁻¹. This fixture's exponent makes it a +/// separate synthetic workload. The `kernel@{event}/pow_f32x4` entry measures the vector power +/// composition in isolation. fn bench_pow_strategies(criterion: &mut Criterion, event: &str) { use core::simd::{Simd, f32x4}; @@ -131,10 +152,8 @@ fn bench_pow_strategies(criterion: &mut Criterion, event: &st ]) }); }); - // The fused variant embeds the strategy in the attraction-coefficient - // arithmetic, measuring what the isolated calls cannot: whether the - // instruction-count gap converts to cycles once the pow competes with - // surrounding vector work for issue slots. + // embed the power in surrounding vector arithmetic to measure the combined workload. The + // benchmark name uses "fused" for this combination, not for a fused multiply-add. let curve_a = 1.577_f32; let curve_b = 0.895_f32; let coefficient = |power: f32x4, distance_squared: f32x4| { @@ -161,16 +180,15 @@ fn bench_pow_strategies(criterion: &mut Criterion, event: &st group.finish(); } -/// The production wrappers over the vendored SLEEF kernels. +/// Measures transcendental wrappers and table-exponential candidates. /// -/// Each entry measures a wrapper exactly as production calls it. The saved per-event baselines make -/// a rewrite of the vendored kernels visible as an instruction-count or cycle change run over run. +/// The gather and architecture-specific table entries are alternatives to the production +/// exponential. Each benchmark measures the fixture's complete wrapper call. fn bench_kernels(criterion: &mut Criterion, event: &str) { use core::simd::{Simd, f32x4, f32x8, f64x4}; - // Lane values spread across the interesting ranges: large-negative - // (underflow edge), moderate, near-zero, and large-positive - // (overflow edge) keep every polynomial and scaling path live. + // sample the normal-output range near its extremes and around zero. Saturating overflow, deep + // underflow and non-finite inputs need separate fixtures. let f32_inputs = f32x8::from_array([-87.3, -12.5, -1.0, -1e-4, 0.0, 0.5, 42.0, 88.7]); let f64_inputs = f64x4::from_array([-708.0, -0.5, 1e-9, 709.0]); let base = f32x4::from_array([0.25, 2.5, 117.0, 0.9]); @@ -201,6 +219,7 @@ fn bench_kernels(criterion: &mut Criterion, event: &str) { group.finish(); } +/// Measures four-lane transform application beside its scalar reference. fn bench_transforms(criterion: &mut Criterion, event: &str) { let state = math::transform_batch( [2.0, 3.0], @@ -221,6 +240,10 @@ fn bench_transforms(criterion: &mut Criterion, event: &str) { group.finish(); } +/// Measures bounds reduction over 100,000 and 1,000,000 collinear points. +/// +/// The fixture repeats points on y = 1000 − 2x. The slice kernel runs beside its scalar reference +/// at 100,000 rows and beside its parallel form at 1,000,000. fn bench_bounds(criterion: &mut Criterion, event: &str) { let small = math::scattered_points(100_000); let large = math::scattered_points(1_000_000); @@ -246,6 +269,7 @@ fn bench_bounds(criterion: &mut Criterion, event: &str) { group.finish(); } +/// Measures serial and parallel similarity fits over 100,000 rows. fn bench_similarity_fit(criterion: &mut Criterion, event: &str) { let fixture = math::similarity_fixture(100_000, [2.0, 0.8, 0.6, 10.0, -4.0]); @@ -262,6 +286,7 @@ fn bench_similarity_fit(criterion: &mut Criterion, event: &st group.finish(); } +/// Measures a 64-way double-precision softmax. fn bench_dvecn(criterion: &mut Criterion, event: &str) { let logits = math::logits(core::array::from_fn::(|index| { f64::from(u8::try_from(index).expect("bounded dimension")).mul_add(0.05, -1.6) @@ -272,6 +297,16 @@ fn bench_dvecn(criterion: &mut Criterion, event: &s }); } +/// Configures Criterion with the backend for a counter-event name. +/// +/// By default, requests a half-second warm-up and one second of measurement over 20 samples. The +/// non-macOS backend uses wall time. +/// +/// # Panics +/// +/// Panics if `event` is unknown. On macOS, also panics if sampler initialization, event +/// configuration or counter start fails, or if a previous acquisition set the process-local guard. +/// This includes unsupported events and unavailable counter privileges. fn hardware_counter(event: &str) -> Criterion { use darwin_kperf_criterion::HardwareCounter; use darwin_kperf_events::Event; @@ -299,13 +334,11 @@ fn hardware_counter(event: &str) -> Criterion(criterion: &mut Criterion, event: &str) { let mut group = criterion.benchmark_group(format!("finite_scan@{event}")); for exponent in [12_u32, 14, 16, 18, 20] { @@ -325,11 +358,11 @@ fn bench_finite_scan(criterion: &mut Criterion, event: &str) group.finish(); } -/// Runs every group under `criterion`, with `event` suffixed into each benchmark id. +/// Runs all groups with an event-qualified benchmark ID. /// -/// The suffix keeps every measurement's statistics in its own lineage: benchmark ids are -/// criterion's storage key, so without it a multi-event run would overwrite one event's samples -/// with the next's and compare quantities of different units run over run. +/// # Panics +/// +/// Propagates benchmark and measurement panics, including failures to sample a hardware counter. fn run_benches(criterion: &mut Criterion, event: &str) { bench_vecn(criterion, event); bench_affinity(criterion, event); @@ -343,6 +376,11 @@ fn run_benches(criterion: &mut Criterion, event: &s } /// Runs the suite once under a single measurement. +/// +/// # Panics +/// +/// Propagates [`hardware_counter`] and benchmark panics. Criterion argument processing also +/// applies. fn run_event(event: &str) { if event == "wall-time" { let mut criterion = Criterion::default() @@ -359,15 +397,19 @@ fn run_event(event: &str) { Criterion::default().configure_from_args().final_summary(); } -// The dispatch mirrors `criterion_group!`/`criterion_main!` expansion; -// the macros cannot express two measurement types behind one binary, -// and a measurement is a property of a whole `Criterion` instance. -// A multi-event selection re-execs this binary once per event instead -// of looping instances in-process: kpc counter configuration is -// per-process state, and a second configurable-event setup in the same -// process fails with `FailedToSetKpcConfig` (the fixed-counter events, -// instructions and cycles, mask the problem by not needing one). +/// Runs the suite once per event, defaulting to retired instructions. +/// +/// `MATH_BENCH_EVENT` supplies a comma-separated list. A multi-event list re-executes this binary +/// once per event, with each child inheriting the arguments. An empty list selects `instructions`. +/// +/// # Panics +/// +/// Propagates [`run_event`] panics for a single event. For multiple events, panics if the +/// executable path is unavailable, a child cannot be spawned, or any child exits unsuccessfully. fn main() { + // The macOS counter backend leaves its acquisition guard set after a successful acquisition, + // including after Drop. Re-executing creates a process-local guard for each event. Therefore + // each child can acquire its own measurement once. let events = std::env::var("MATH_BENCH_EVENT").unwrap_or_else(|_| "instructions".to_owned()); let events: Vec<&str> = events .split(',') diff --git a/libs/@local/graph/atlas/benches/miner_index.rs b/libs/@local/graph/atlas/benches/miner_index.rs index c4e3e209527..8a7f3c9ccb5 100644 --- a/libs/@local/graph/atlas/benches/miner_index.rs +++ b/libs/@local/graph/atlas/benches/miner_index.rs @@ -5,14 +5,12 @@ //! query sweeps. Each build starts from scratch because every point moves between ticks. The //! suite pits `kiddo` against `grid`: //! -//! - `kiddo`: `ImmutableKdTree`, the miner's index. A balanced kd-tree adapts its partition depth -//! to local density, so its query cost is immune to the cluster skew attraction exists to produce -//! - the property the timings here certify. -//! - `grid`: a uniform bucket grid over the known frame, written here (counting-sort build, -//! ring-expansion exact kNN). The natural alternative when the caller knows the frame ahead of -//! time, and the control that prices its one assumption. A query scans whole cells, so one global -//! cell size makes the sweep's cost grow with the sum of squared cell occupancies, quadratic in -//! exactly the density skew a projected map carries. +//! - `kiddo`: [`ImmutableKdTree`], the miner's kd-tree index. The clustered fixtures measure how +//! its query cost responds to the density skew produced by attraction. +//! - `grid`: a uniform bucket grid over the known frame, with counting-sort construction and +//! ring-expansion exact kNN. Each query scans its entire starting cell before expanding. A cell +//! with `n` points therefore contributes at least `n²` distance evaluations to a full per-point +//! sweep. This lower bound explains why concentrated occupancy can make a uniform grid expensive. //! //! The fixtures are synthetic point sets in the `[0, 10]^2` frame: `clustered` (Gaussian mixture //! over a uniform background - the shape a projected map takes), `uniform` (the grid's best case), @@ -21,7 +19,7 @@ //! //! - `build`: one index construction, single-threaded. //! - `sweep`: one full per-point kNN pass at k = 24, single-threaded. Sweeps parallelize -//! embarrassingly and identically for both engines, so the single-threaded number is the +//! embarrassingly and identically for both engines. The single-threaded number is therefore the //! comparative one. //! - `tick`: two builds plus two sweeps over the two lens extremes' point sets on the clustered //! shape, which is the unit the training-loop cadence actually spends. @@ -39,8 +37,8 @@ //! //! Timings default to 250K points so a full sweep stays in minutes. Set `MINER_BENCH_POINTS` (e.g. //! to `1000000`) for headline numbers at the expected map scale. Eligible-set subtraction (512-d -//! neighbours, protected pairs, self) is frame-independent and happens downstream of the index, so -//! every engine returns raw neighbours here, self included. +//! neighbours, protected pairs, self) is frame-independent and happens downstream of the index. +//! Every engine returns raw neighbours here, self included. #![expect( clippy::print_stderr, clippy::significant_drop_tightening, @@ -89,10 +87,16 @@ const DEFAULT_POINTS: usize = 250_000; /// Ground-truth queries per fixture for the recall report. const RECALL_SAMPLE: usize = 512; +/// The fixture seed every point set and corpus in this target synthesizes from. const SEED: u64 = 0x2D5A_17ED; -/// The second lens extreme's point set for the tick unit. +/// The seed of the second lens extreme's point set for the tick unit. const SEED_EXTREME: u64 = SEED ^ 0xFFFF_FFFF; +/// Returns the point count the fixtures synthesize: `MINER_BENCH_POINTS`, or [`DEFAULT_POINTS`]. +/// +/// # Panics +/// +/// This panics when `MINER_BENCH_POINTS` is set to a value that is not a point count. fn points_count() -> usize { std::env::var("MINER_BENCH_POINTS").map_or(DEFAULT_POINTS, |value| { value @@ -104,18 +108,23 @@ fn points_count() -> usize { /// How a fixture distributes points over the frame. #[derive(Debug, Copy, Clone, PartialEq, Eq)] enum Shape { - /// A Gaussian mixture over a uniform background: 256 clusters with log-uniform spreads hold 80% - /// of points. The shape a projected map takes, and the primary fixture. + /// A Gaussian mixture over a uniform background. + /// + /// 256 clusters with log-uniform spreads hold 80% of points. This is the shape a projected map + /// takes, and the primary fixture. Clustered, - /// Uniform over the frame: the grid's best case, kept as the control that shows how much the - /// mixture costs each engine. + /// Uniform over the frame: the grid's best case. + /// + /// Kept as the control that shows how much the mixture costs each engine. Uniform, - /// One near-coincident blob holds an eighth of all points: the bucket-skew stress a - /// coincident-geometry pile-up produces. + /// One near-coincident blob holding an eighth of all points. + /// + /// The bucket-skew stress a coincident-geometry pile-up produces. Pathological, } impl Shape { + /// Returns the fixture's segment of a benchmark id. const fn label(self) -> &'static str { match self { Self::Clustered => "clustered", @@ -176,13 +185,23 @@ fn synthesize(shape: Shape, count: usize, seed: u64) -> Vec<[f32; 2]> { .collect() } -/// One kNN candidate; the heap orders by distance, worst on top. +/// One kNN candidate. +/// +/// A [`BinaryHeap`] of candidates keeps the farthest on top: the worst of the k best a query holds. #[derive(Debug, Copy, Clone, PartialEq)] struct Candidate { + /// The squared Euclidean distance from the query. distance: f32, + /// The point's index in the fixture. id: u32, } +/// Reflexive on the `f32` field because no distance is `NaN`. +/// +/// Every candidate comes from [`Grid::nearest`] with a distance between two in-frame points: a +/// sum of squares of finite differences, never `NaN` and never negative zero. The derived +/// equality is therefore reflexive, and `total_cmp` in [`Ord::cmp`] agrees with it on every +/// reachable value, as the [`Ord`] laws require. impl Eq for Candidate {} impl PartialOrd for Candidate { @@ -201,19 +220,28 @@ impl Ord for Candidate { /// A uniform bucket grid over the frame, the in-bench candidate. /// -/// The build sizes cells so the average cell holds about k points. Build is one counting sort; a -/// query expands Chebyshev rings around its cell, keeping the k best in a bounded heap, and stops +/// The build sizes cells so the average cell holds about k points. The build is one counting sort. +/// A query expands Chebyshev rings around its cell, keeping the k best in a bounded heap, and stops /// once no unvisited ring can beat the current worst, so results are exact. struct Grid { + /// One cell's side length. cell: f32, + /// Cells per axis. dim: usize, + /// Each cell's first slot in `ids` and `coords`, with one trailing end offset. starts: Vec, + /// Point indices grouped by cell. ids: Vec, + /// The coordinates of `ids`, slot for slot. coords: Vec<[f32; 2]>, } impl Grid { /// Builds the grid over `points`, sizing cells for about `occupancy` points each. + /// + /// # Panics + /// + /// This panics when `occupancy` is zero. fn build(points: &[[f32; 2]], occupancy: usize) -> Self { let cells = points.len().div_ceil(occupancy).max(1); let dim = ((cells as f32).sqrt().ceil() as usize).max(1); @@ -260,7 +288,13 @@ impl Grid { .unwrap_or(0) } - /// Collects the k nearest points to `query` into `heap`, worst on top; exact, self included. + /// Collects the k nearest points to `query` into `heap`, worst on top. + /// + /// The result is exact, and a query drawn from the indexed points recovers itself among them. + /// + /// # Panics + /// + /// This panics when `k` is zero and the grid holds a point. fn nearest(&self, query: [f32; 2], k: usize, heap: &mut BinaryHeap) { heap.clear(); let center_x = ((query[0] / self.cell) as usize).min(self.dim - 1); @@ -271,8 +305,8 @@ impl Grid { .max(self.dim - 1 - center_y); for ring in 0..=reach { - // Every cell of ring r sits at least (r - 1) cells away, so - // a full heap whose worst lies inside that bound is final. + // Every cell of ring r sits at least (r - 1) cells away. A full heap whose worst lies + // inside that bound is final. if heap.len() == k && ring >= 2 { let bound = (ring - 1) as f32 * self.cell; let worst = heap.peek().expect("the heap is full").distance; @@ -345,13 +379,25 @@ impl Grid { } } +/// The miner's kd-tree over 2D `f32` points, the `kiddo` engine under test. type KdTree = ImmutableKdTree; +/// Builds the kd-tree over `points`. +/// +/// # Panics +/// +/// This panics when `points` holds more items than a `u32` index addresses: the tree numbers its +/// items by slice position in `u32`, so construction refuses a slice whose last index does not +/// fit one. fn build_kiddo(points: &[[f32; 2]]) -> KdTree { KdTree::new_from_slice(points).expect("the fixture fits the index's item domain") } -/// Sums neighbour ids over a full per-point sweep, so the pass has an observable result. +/// Sums neighbour ids over a full per-point sweep. The pass has an observable result. +/// +/// # Panics +/// +/// This panics when `k` is zero and `points` is not empty. fn sweep_grid(grid: &Grid, points: &[[f32; 2]], k: usize) -> u64 { let mut heap = BinaryHeap::with_capacity(k); let mut sum = 0_u64; @@ -365,6 +411,11 @@ fn sweep_grid(grid: &Grid, points: &[[f32; 2]], k: usize) -> u64 { sum } +/// Sums neighbour ids over a full per-point kd-tree sweep, the counterpart of [`sweep_grid`]. +/// +/// # Panics +/// +/// This panics when `k` is zero. fn sweep_kiddo(tree: &KdTree, points: &[[f32; 2]], k: usize) -> u64 { let limit = NonZero::new(k).expect("the neighbour count is positive"); points @@ -393,6 +444,11 @@ fn tick( } /// Exact k nearest ids per sampled query, by brute force. +/// +/// # Panics +/// +/// This panics when `k` is zero or exceeds the point count, or when a sample index lies outside +/// `points`. fn ground_truth(points: &[[f32; 2]], samples: &[usize], k: usize) -> Vec> { let mut distances: Vec<(f32, u32)> = Vec::with_capacity(points.len()); samples @@ -431,6 +487,10 @@ fn recall(truth: &[Vec], results: &[Vec], k: usize) -> f64 { } /// Prints each engine's recall and the grid's occupancy skew. +/// +/// # Panics +/// +/// This panics when `count` is below [`K`], because the brute-force ground truth needs k points. fn report_recall(count: usize) { eprintln!("miner index recall audit: {count} points, k = {K}, {RECALL_SAMPLE} sampled queries"); @@ -558,8 +618,9 @@ fn bench_tick(criterion: &mut Criterion) { group.finish(); } -/// One eighth of the measured live link volume; the resulting row domain times the candidate width -/// lands the sweep in the millions of probe pairs. +/// One eighth of the measured live link volume. +/// +/// The resulting row domain times the candidate width yields a sweep of millions of probe pairs. const JUDGE_LINKS: usize = 275_000; /// Times the access layouts a mined sweep's protection vetting can take, per hit rate. @@ -570,9 +631,8 @@ fn bench_judge(criterion: &mut Criterion) { let mut group = criterion.benchmark_group("miner_index/judge"); group.sample_size(10); - // Mining candidates are close 2D points: attraction pulls linked pairs together, so the - // realistic sweep is partner-rich; the uniform sweep bounds the layouts' spread from the - // other side. + // Mining candidates are close 2D points: attraction pulls linked pairs together. The realistic + // sweep is partner-rich. The uniform sweep bounds the layouts' spread from the other side. for (label, fraction) in [("uniform", 0.0), ("linked", 0.5)] { let probes = corpus.judge_probes::(per_row, fraction, SEED); group.throughput(Throughput::Elements(probes.pairs() as u64)); @@ -588,12 +648,16 @@ fn bench_judge(criterion: &mut Criterion) { group.finish(); } +/// Returns the Criterion configuration the groups share. +/// +/// Every benchmark warms up for half a second and measures for ten seconds. fn config() -> Criterion { Criterion::default() .warm_up_time(Duration::from_millis(500)) .measurement_time(Duration::from_secs(10)) } +/// Prints the recall audit, then runs every timed group. fn benches_with_report(criterion: &mut Criterion) { report_recall(points_count()); bench_build(criterion); diff --git a/libs/@local/graph/atlas/benches/projector_backend.rs b/libs/@local/graph/atlas/benches/projector_backend.rs index 9f57f7b1bb0..3921663ff65 100644 --- a/libs/@local/graph/atlas/benches/projector_backend.rs +++ b/libs/@local/graph/atlas/benches/projector_backend.rs @@ -58,10 +58,19 @@ use hash_graph_atlas::{ use rand_xoshiro::Xoshiro256PlusPlus; use rayon::ThreadPoolBuilder; +/// The fixture seed every model and batch in this target builds from. const SEED: u64 = 0x9C0E_C708; +/// The backends under comparison. +/// +/// The CPU and the host's accelerated device family, each pinned to ordinal 0. const DEVICES: &[PinnedDevice] = &[Device::Cpu.pin(0), Device::host().pin(0)]; +/// Returns the largest forward batch's row count: `PROJECTOR_BENCH_ROWS`, or 65536. +/// +/// # Panics +/// +/// This panics when `PROJECTOR_BENCH_ROWS` is set to a value that is not a row count. fn rows() -> usize { std::env::var("PROJECTOR_BENCH_ROWS").map_or(65_536, |value| { value @@ -70,6 +79,7 @@ fn rows() -> usize { }) } +/// Returns a synthetic batch of `rows` rows drawn from the fixture seed. fn synthesize(rows: usize) -> Batch { Batch::new::(rows, SEED) } @@ -115,6 +125,12 @@ fn bench_forward(criterion: &mut Criterion) { } /// One fixed CPU training step across rayon pool sizes. +/// +/// Pool sizes above the host's rayon thread count are skipped. +/// +/// # Panics +/// +/// This panics when a rayon pool of a requested size cannot be built. fn bench_thread_scaling(criterion: &mut Criterion) { let model = Model::build::(Device::Cpu.pin(0), SEED); let batch = synthesize(4_096); @@ -141,6 +157,12 @@ fn bench_thread_scaling(criterion: &mut Criterion) { } /// One real training step at the production plan, phase by phase. +/// +/// `PROJECTOR_BENCH_LIVE_ROWS` sizes the synthesized corpus, 65536 rows by default. +/// +/// # Panics +/// +/// This panics when `PROJECTOR_BENCH_LIVE_ROWS` is set to a value that is not a row count. fn bench_live_step(criterion: &mut Criterion) { let rows = std::env::var("PROJECTOR_BENCH_LIVE_ROWS").map_or(65_536, |value| { value @@ -231,6 +253,9 @@ fn bench_live_step(criterion: &mut Criterion) { group.finish(); } +/// Returns the Criterion configuration the groups share. +/// +/// Every benchmark warms up for half a second and measures for ten seconds. fn config() -> Criterion { Criterion::default() .warm_up_time(Duration::from_millis(500)) diff --git a/libs/@local/graph/atlas/benches/relation_build.rs b/libs/@local/graph/atlas/benches/relation_build.rs index f5592f862ff..bb31e0e8bef 100644 --- a/libs/@local/graph/atlas/benches/relation_build.rs +++ b/libs/@local/graph/atlas/benches/relation_build.rs @@ -1,16 +1,16 @@ //! Wall-time benchmarks for the relation-index build under skew. //! //! The build's parallel design makes claims that only hold or fail at realistic scale and volume -//! concentration; each group here measures one of them, over corpora synthesized at the live +//! concentration. Each group here measures one of them, over corpora synthesized at the live //! store's measured shape (2.2M links, 17 relation types, the base type owning half of all //! instances, Zipf-hubbed targets): //! //! - `build`: full-build wall time across the live, uniform, and single-mega-relation profiles. The -//! profiles share endpoints, volume, and policies, so a spread between them is the cost of skew -//! alone - the two-level parallelism claim is that the spread stays small. -//! - `stages` times the two whole-slice sorts, the group emission, the protection assembly, and the -//! index re-validation in isolation, each from its own pre-sorted input state - the split that -//! attributes the build's wall time and shows what caps its thread scaling. +//! profiles share endpoints, volume, and policies. A spread between them is therefore the cost of +//! skew alone - the two-level parallelism claim is that the spread stays small. +//! - `stages` times the whole-slice group sort, the group emission, the protection assembly, and +//! the index re-validation in isolation, each from the input state that stage starts at - the +//! split that attributes the build's wall time and shows what caps its thread scaling. //! - `threads` times the full build across pool sizes, on the live profile. //! - `chunk` times group emission across chunk sizes around the production emission chunk. The //! claim that the constant is only a work-splitting unit predicts a flat response. @@ -23,7 +23,7 @@ //! Before the timed groups, one report prints the pruned-edge and omitted-mass outcomes of a //! pruning-threshold sweep on the live profile. A threshold is admissible while the omitted mass //! fraction stays numerically negligible, and these are the numbers that judgement reads. The -//! fixture's masses come from its policy spread (the live store leaves confidence unscored), so the +//! fixture's masses come from its policy spread (the live store leaves confidence unscored). The //! sweep calibrates thresholds against policy mass, not against a confidence distribution. #![expect( clippy::print_stderr, @@ -43,8 +43,14 @@ use rayon::ThreadPoolBuilder; /// One eighth of the measured live link volume. const DEFAULT_LINKS: usize = 275_000; +/// The fixture seed every corpus in this target synthesizes from. const SEED: u64 = 0x5A17_A71A; +/// Returns the link count the corpora synthesize: `RELATION_BENCH_LINKS`, or [`DEFAULT_LINKS`]. +/// +/// # Panics +/// +/// This panics when `RELATION_BENCH_LINKS` is set to a value that is not a link count. fn links() -> usize { std::env::var("RELATION_BENCH_LINKS").map_or(DEFAULT_LINKS, |value| { value @@ -89,9 +95,9 @@ fn bench_stages(criterion: &mut Criterion) { BatchSize::LargeInput, ); }); - // The emission runner reads the corpus's cached group-sorted copy, so - // its iterations carry no per-round clone; the assembly reorders its - // record input, so each round takes a fresh clone outside the timing. + // The emission runner reads the corpus's cached group-sorted copy. Its iterations carry no + // per-round clone. The assembly reorders its record input. Each round takes a fresh clone + // outside the timing. group.bench_function("emit_groups", |bencher| { bencher.iter(|| corpus.emit_groups(black_box(chunk))); }); @@ -102,8 +108,8 @@ fn bench_stages(criterion: &mut Criterion) { BatchSize::LargeInput, ); }); - // Assembly constructs the invariants and then re-validates them; - // this entry splits the stage's cost between scatter and check. + // Assembly constructs the invariants and then re-validates them. + // This entry splits the stage's cost between scatter and check. group.bench_function("validate_protection", |bencher| { bencher.iter(|| corpus.validate_protection()); }); @@ -111,7 +117,13 @@ fn bench_stages(criterion: &mut Criterion) { group.finish(); } -/// Full-build scaling across worker-pool sizes. +/// Times full-build scaling across worker-pool sizes. +/// +/// The sweep skips pool sizes above the host's rayon thread count. +/// +/// # Panics +/// +/// This panics when a rayon pool of a requested size cannot be built. fn bench_thread_scaling(criterion: &mut Criterion) { let corpus = production_corpus(Profile::Live, links(), SEED); @@ -142,8 +154,8 @@ fn bench_thread_scaling(criterion: &mut Criterion) { /// Emission response to the chunk size, on the mega profile. /// -/// The mega profile gives group-level parallelism nothing to hide behind, so the emission pass -/// rides on chunking alone - the sharpest view of the flat-response claim. +/// The mega profile assigns all instances to one relation. Chunking provides the emission pass with +/// parallel work even when there are no other populated groups. fn bench_chunk_sensitivity(criterion: &mut Criterion) { let corpus = production_corpus(Profile::Mega, links(), SEED); let production = production_chunk(); @@ -184,12 +196,16 @@ fn report_pruning_sweep() { } } +/// Returns the Criterion configuration the groups share. +/// +/// Every benchmark warms up for half a second and measures for ten seconds. fn config() -> Criterion { Criterion::default() .warm_up_time(Duration::from_millis(500)) .measurement_time(Duration::from_secs(10)) } +/// Prints the pruning-threshold sweep, then runs every timed group. fn benches_with_report(criterion: &mut Criterion) { report_pruning_sweep(); bench_build_profiles(criterion); diff --git a/libs/@local/graph/atlas/src/allocator.rs b/libs/@local/graph/atlas/src/allocator.rs index 7478e00a222..e47737ac719 100644 --- a/libs/@local/graph/atlas/src/allocator.rs +++ b/libs/@local/graph/atlas/src/allocator.rs @@ -1,9 +1,9 @@ //! Byte accounting at the allocator boundary. //! -//! [`MemoryUsageAllocator`] wraps an allocator and counts the live bytes allocated through it, -//! so a resident-size reading comes from the allocations themselves rather than from a -//! hand-maintained estimate beside them. [`MemoryUsage`] is the reader's half: a cheap handle -//! onto the same counter, held by whoever prices the memory without holding the allocator. +//! [`MemoryUsageAllocator`] tallies the layout sizes requested through an allocator. Counting +//! allocation requests avoids a separate hand-maintained size estimate for those allocations. The +//! tally is a request-side figure and not a resident-size measurement. [`MemoryUsage`] is a cheap +//! reading handle to the same counter for code that does not hold the allocator. use core::{ alloc::{self, Allocator}, @@ -14,25 +14,50 @@ use std::alloc::Global; use ::alloc::sync::Arc; -/// A reading handle onto one allocator's live-byte counter. +/// A reading handle onto one allocator's byte tally. /// -/// Clones share the counter, so every handle reads the same total. +/// Clones observe the same counter. #[derive(Debug, Clone)] pub(crate) struct MemoryUsage(Arc>); impl MemoryUsage { - /// Reads the live bytes currently allocated through the counter's allocator. + /// Reads the running tally of requested bytes. + /// + /// The figure follows the requested layout sizes, including successful resizes and the sizes + /// named when blocks are released. It counts requested bytes rather than the wrapped + /// allocator's excess capacity or the process's resident pages. When no allocator operation is + /// in progress and you have synchronized earlier operations with this read, + /// [`MemoryUsageAllocator`]'s accounting conditions make it the total requested size of its + /// live counted allocations. + /// + /// A relaxed load returns a snapshot that may already be stale when used. It imposes no + /// ordering on the allocations it counts. pub(crate) fn get(&self) -> usize { self.0.load(atomic::Ordering::Relaxed) } } -/// An allocator that counts the live bytes allocated through it. +/// An allocator that tallies the layout sizes requested through it. +/// +/// [`Allocator::allocate`] and [`Allocator::allocate_zeroed`] add `layout.size()`, a grow adds +/// the difference between the two requested sizes, a shrink subtracts that difference, and +/// [`Allocator::deallocate`] subtracts the size of the layout it is handed. Every figure is a +/// size a caller supplied. The wrapped allocator's own padding, its over-allocation above the +/// requested size, and the pages the system has actually committed are all invisible here. /// -/// Every allocation adds its requested layout size and every deallocation subtracts it, so the -/// counter reads the bytes currently held. The count covers requested layout sizes alone. An -/// allocator's own padding or over-allocation stays invisible. Clones share one counter, so a -/// collection may clone its allocator freely and the total stays one number. +/// Once all allocator operations have completed, interpreting the total as the requested size of +/// live counted allocations requires consistent layout accounting. Each deallocation and each old +/// layout supplied for resizing must name that block's last requested size. Every allocation +/// intended for the total must also pass through this allocator or one of its clones. Bytes +/// obtained elsewhere are never counted. +/// +/// The tally records the supplied sizes even when a later fitting layout names more bytes than the +/// allocation requested. A resize then adjusts from that supplied old size. Deallocation subtracts +/// the supplied size and wraps the unsigned counter if that subtraction underflows. These +/// accounting conditions do not limit the wrapped allocator's excess capacity, which the tally +/// still excludes. +/// +/// All clones share one counter. #[derive(Debug, Clone)] pub(crate) struct MemoryUsageAllocator { allocator: A, @@ -61,11 +86,11 @@ impl MemoryUsageAllocator { } } -// SAFETY: every method forwards to the wrapped allocator and returns its blocks unchanged, so -// currently-allocated pointers, layout fit, and block validity are exactly the wrapped -// allocator's. Clones share the wrapped allocator's clone semantics and one counter, so blocks -// allocated through one clone deallocate through another exactly when the wrapped allocator -// permits it. The counter only observes layouts and never touches the blocks. +// SAFETY: every method forwards to the wrapped allocator and returns its blocks unchanged. +// Currently-allocated pointers, layout fit, and block validity are exactly the wrapped allocator's. +// Clones share the wrapped allocator's clone semantics and one counter, so blocks allocated through +// one clone deallocate through another exactly when the wrapped allocator permits it. The counter +// only observes layouts and never touches the blocks. unsafe impl Allocator for MemoryUsageAllocator { fn allocate_zeroed( &self, @@ -85,9 +110,9 @@ unsafe impl Allocator for MemoryUsageAllocator { old_layout: alloc::Layout, new_layout: alloc::Layout, ) -> Result, alloc::AllocError> { - // SAFETY: every block this allocator returns comes from the wrapped allocator - // unchanged, so the caller's obligations - `ptr` denotes a current allocation of it, - // and the layouts fit it - transfer verbatim. + // SAFETY: The caller guarantees that `ptr` is currently allocated, `old_layout` fits it and + // `new_layout` is at least as large. This allocator returns the wrapped allocator's blocks + // unchanged. The same preconditions therefore hold for its `grow` call. let new_ptr = unsafe { self.allocator.grow(ptr, old_layout, new_layout)? }; self.memory_usage.fetch_add( @@ -104,8 +129,8 @@ unsafe impl Allocator for MemoryUsageAllocator { old_layout: alloc::Layout, new_layout: alloc::Layout, ) -> Result, alloc::AllocError> { - // SAFETY: every block this allocator returns comes from the wrapped allocator - // unchanged, so the caller's obligations transfer verbatim. + // SAFETY: every block this allocator returns comes from the wrapped allocator unchanged. + // The caller's obligations transfer verbatim. let new_ptr = unsafe { self.allocator.grow_zeroed(ptr, old_layout, new_layout)? }; self.memory_usage.fetch_add( @@ -122,8 +147,8 @@ unsafe impl Allocator for MemoryUsageAllocator { old_layout: alloc::Layout, new_layout: alloc::Layout, ) -> Result, alloc::AllocError> { - // SAFETY: every block this allocator returns comes from the wrapped allocator - // unchanged, so the caller's obligations transfer verbatim. + // SAFETY: every block this allocator returns comes from the wrapped allocator unchanged. + // The caller's obligations transfer verbatim. let new_ptr = unsafe { self.allocator.shrink(ptr, old_layout, new_layout)? }; self.memory_usage.fetch_sub( old_layout.size().abs_diff(new_layout.size()), @@ -145,8 +170,8 @@ unsafe impl Allocator for MemoryUsageAllocator { self.memory_usage .fetch_sub(layout.size(), atomic::Ordering::Relaxed); - // SAFETY: every block this allocator returns comes from the wrapped allocator - // unchanged, so the caller's obligations transfer verbatim. + // SAFETY: every block this allocator returns comes from the wrapped allocator unchanged. + // The caller's obligations transfer verbatim. unsafe { self.allocator.deallocate(ptr, layout); } @@ -155,11 +180,6 @@ unsafe impl Allocator for MemoryUsageAllocator { #[cfg(test)] mod tests { - /// The tests the `miri` nextest profile selects. - /// - /// Each test here drives the counting allocator through allocation, growth, shrinking and - /// release, and reads the byte counter it maintains. The profile selects by module path, so - /// moving a test in or out of this module is the whole edit. mod miri { use core::alloc::{Allocator as _, Layout}; diff --git a/libs/@local/graph/atlas/src/bench.rs b/libs/@local/graph/atlas/src/bench.rs index 8ae46fc1762..9170c246883 100644 --- a/libs/@local/graph/atlas/src/bench.rs +++ b/libs/@local/graph/atlas/src/bench.rs @@ -1,13 +1,12 @@ -//! Measurement seams for the crate's benchmark targets. +//! Input builders and stage-level measurements for Atlas benchmarks. //! -//! Benchmark targets are external crates, so pipeline stages that are private implementation detail -//! everywhere else surface here behind the `bench` cargo feature. A target can synthesize realistic -//! inputs and run one stage at a time. Every result it reads is a plain number, and no internal -//! type escapes. The one deliberate exception is the Morton key vocabulary ([`Depth`], -//! [`MortonKey`], [`MortonCell`]), which crosses typed: a target addresses cells with the same -//! invariant-carrying types production uses instead of re-deriving their contracts from raw -//! integers. Nothing here is API for consumers of the crate. The feature exists for the -//! `[[bench]]` targets and is off by default. +//! The `bench` feature is off by default. Enable it to expose selected pipeline operations to +//! standalone benchmark targets. These interfaces separate input preparation from the operations +//! under measurement, without requiring a complete fit. +//! +//! The Morton types ([`Depth`], [`MortonKey`] and [`MortonCell`]) preserve key and cell invariants +//! when constructing benchmark requests. These interfaces serve crate development and follow the +//! internal pipeline. pub use crate::{ math::{bench as math, kernel::bench as kernel}, diff --git a/libs/@local/graph/atlas/src/bitset/compress.rs b/libs/@local/graph/atlas/src/bitset/compress.rs index b01ba1287ed..25103ad13de 100644 --- a/libs/@local/graph/atlas/src/bitset/compress.rs +++ b/libs/@local/graph/atlas/src/bitset/compress.rs @@ -5,23 +5,23 @@ use roaring::RoaringBitmap; /// A compressed membership set over one row domain. /// -/// Memory is proportional to what the set admits (its cardinality and the runs its rows form) -/// rather than to the size of the domain it draws from. A set admitting a few thousand rows of a -/// million-row domain costs kilobytes. A set admitting one contiguous span costs a constant. That -/// makes this the shape for a per-request or per-session row set, where one bit per domain row -/// costs the whole domain however few rows the set admits. +/// Storage is allocated only for occupied blocks of 2¹⁶ row values. Sparse blocks use sorted arrays +/// and dense blocks use bitmaps. This avoids allocating one bit per domain row when membership is +/// sparse. A contiguous span can occupy many blocks, and insertion alone does not convert them to +/// run containers. /// -/// The type parameter names the domain, so a set of node rows and a set of link rows have different -/// types and the compiler rejects either one where the other belongs. +/// The type parameter distinguishes row domains at compile time. A set of node rows cannot +/// substitute for a set of link rows. /// -/// The representable domain is `0..u32::MAX`. [`Self::contains`] answers `false` for a row above -/// it, so a query is total over the id type, while [`Self::insert`] panics rather than dropping the -/// row. +/// The representable domain is `0..=u32::MAX`. [`Self::contains`] answers `false` for a row above +/// it, while [`Self::insert`] panics rather than dropping the row. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the types are crate-private. /// /// ```ignore -/// use crate::identity::NodeRowId; +/// use crate::{bitset::CompressedBitSet, identity::NodeRowId}; /// /// let mut visible = CompressedBitSet::new(); /// visible.insert(NodeRowId::new(3)); @@ -32,10 +32,12 @@ use roaring::RoaringBitmap; /// assert_eq!(visible.count(), 2); /// ``` /// -/// Iteration ascends by row, whatever the insertion order: +/// # Example: iterating in row order +/// +/// This in-crate example is ignored because the types are crate-private. /// /// ```ignore -/// use crate::identity::EdgeRowId; +/// use crate::{bitset::CompressedBitSet, identity::EdgeRowId}; /// /// let mut links = CompressedBitSet::new(); /// for row in [4, 1, 2].map(EdgeRowId::new) { @@ -49,17 +51,17 @@ use roaring::RoaringBitmap; /// ``` #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct CompressedBitSet { + /// The admitted rows, as `u32` values. rows: RoaringBitmap, + /// The row domain, carried in the type without owning a `T`. marker: PhantomData T>, } impl CompressedBitSet { /// Per-container bookkeeping allowance for [`Self::heap_bytes`], in bytes. /// - /// The store keeps each container behind its own entry - key, variant tag and the - /// container's inline vector or box - which the payload statistics do not report. The - /// allowance is a deliberate overestimate of that entry, so a heavily containerized set - /// never reads as cheaper than it is. + /// The estimate adds this fixed charge to the reported payload for each occupied container. It + /// does not measure the container vector's capacity or allocator overhead. const CONTAINER_ALLOWANCE: u64 = 64; /// Creates a set admitting no rows. @@ -85,10 +87,10 @@ impl CompressedBitSet { /// Returns the set's retained container bytes. /// - /// The figure sums the bitmap's array, run and bitset container payloads plus - /// [`Self::CONTAINER_ALLOWANCE`] per container, so it moves with the compression the row - /// distribution earns rather than with the row count. The wrapper's own inline size and - /// allocator slack are not counted. + /// The figure sums the array, run and bitmap byte counts reported by + /// [`RoaringBitmap::statistics`] plus [`Self::CONTAINER_ALLOWANCE`] per container. It is an + /// estimate, not a measurement or a guaranteed upper bound on allocated bytes. The value + /// excludes this type's inline size. #[must_use] pub(crate) fn heap_bytes(&self) -> u64 { let statistics = self.rows.statistics(); diff --git a/libs/@local/graph/atlas/src/bitset/dense.rs b/libs/@local/graph/atlas/src/bitset/dense.rs index f56ac399b9c..5fc4718a6d2 100644 --- a/libs/@local/graph/atlas/src/bitset/dense.rs +++ b/libs/@local/graph/atlas/src/bitset/dense.rs @@ -34,8 +34,8 @@ const fn num_words(domain_size: u64) -> u64 { reason = "the quotient names the row's word and the remainder its bit within that word" )] const fn word_index_and_mask(row: u64) -> (usize, u64) { - // Every caller bounds `row` by a domain whose words are in memory, so the word index fits - // `usize`. + // An in-memory slice has a `usize`-representable word count. Every call bounds `row` to the + // domain stored in that slice. The row's word index fits `usize`. #[expect(clippy::cast_possible_truncation)] let index = (row / WORD_BITS as u64) as usize; (index, 1 << (row % WORD_BITS as u64)) @@ -81,16 +81,10 @@ impl fmt::Display for ParseDenseBitSliceError { impl core::error::Error for ParseDenseBitSliceError {} -/// [`DenseBitSlice`]'s fields with no frame invariant coupling them. +/// Unvalidated storage for constructing and checking a [`DenseBitSlice`]. /// -/// Every zerocopy claim is true here: any header beside any whole words is a value of this type. -/// That freedom is the twin's purpose. [`DenseBitSlice`]'s hand-written [`zerocopy::TryFromBytes`] -/// delegates field validity to the derive on these fields. [`DenseBitSlice::new_empty`] builds -/// its zeroed allocation here, where a zeroed header beside a nonzero word count breaks nothing. -/// -/// The fields mirror [`DenseBitSlice`]'s exactly. The layout half of that claim is asserted at -/// compile time by the cast in `is_bit_valid`. The bit-validity half rests on the field types -/// being identical. A field type change in either twin therefore re-derives that proof. +/// The field types and their order must match `DenseBitSlice`. Any initialized header and words are +/// valid here, including a zeroed allocation whose header does not yet describe its word count. #[derive( zerocopy::FromBytes, zerocopy::IntoBytes, @@ -102,39 +96,41 @@ impl core::error::Error for ParseDenseBitSliceError {} struct RawDenseBitSlice { /// [`DenseBitSlice`]'s header, not yet coupled to the word count. domain_size: U64, + /// The row domain the words index. marker: PhantomData, - /// [`DenseBitSlice`]'s words, not yet policed for excess bits. + /// The storage words, with no restriction on excess bits. words: [U64], } /// A dense membership set over one row domain, stored as transportable bytes. /// -/// The set spends one bit per domain row. Memory is proportional to the domain rather than to -/// what the set admits, which is the right price where membership is dense or the domain is -/// small. The type parameter names the domain, so a set of node rows and a set of link rows have -/// different types and the compiler rejects either one where the other belongs. +/// The set uses one bit per domain row, rounded up to a whole word, plus its header. Use it when +/// membership is dense or the domain is small. The type parameter distinguishes row domains at +/// compile time. /// /// The set is its own byte format. A frame is the domain size as an 8-byte little-endian count, /// then the member bits packed 64 to a little-endian word in ascending row order, with every bit /// above the domain zero. [`DenseBitSlice::try_from_prefix`] reads a frame in place off the front /// of a buffer, without copying and at any byte offset, refuses one whose header, word count, or -/// excess bits break that layout, and returns the bytes after the frame. [`zerocopy::IntoBytes`] -/// carries the write side, so `as_bytes` on a live set is the frame. +/// excess bits break that layout, and returns the bytes after the frame. +/// [`zerocopy::IntoBytes::as_bytes`] exposes a live set as that frame. /// -/// The frame invariant is the type's bit validity, so every [`zerocopy::TryFromBytes`] door -/// validates it inside the cast and no door mints a set whose header and words disagree. A -/// plain prefix read is greedy - it hands validation the largest word count that fits rather -/// than the one the header claims. Reading a frame from a longer buffer therefore takes the -/// `_with_elems` door with the header's own count, which is the split -/// [`DenseBitSlice::try_from_prefix`] performs itself. +/// Every [`zerocopy::TryFromBytes`] conversion checks the word count and excess bits as part of +/// validation. Use [`Self::try_from_prefix`] to read a frame followed by other data. It derives the +/// frame length from the header. /// /// The set is unsized. Create one in place behind a box with [`DenseBitSlice::new_empty`], or /// borrow one from existing bytes with [`DenseBitSlice::try_from_prefix`]. The domain is fixed at /// creation, and mutation never moves the storage. /// -/// Sets are equal when they draw from the same domain and admit the same rows. +/// Sets are equal when they draw from the same domain and admit the same rows. The [`BitRelations`] +/// implementations modify the set by union, subtraction or intersection and return whether +/// membership changed. They panic if the domains differ, for either a [`DenseBitSet`] or another +/// `DenseBitSlice` operand. +/// +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the types are crate-private. /// /// ```ignore /// use zerocopy::IntoBytes as _; @@ -151,55 +147,53 @@ struct RawDenseBitSlice { /// assert!(read.contains(NodeRowId::new(3))); /// assert_eq!(read.count(), 2); /// assert!(rest.is_empty()); +/// # Ok::<(), crate::bitset::ParseDenseBitSliceError>(()) /// ``` #[derive(zerocopy::IntoBytes, zerocopy::Immutable, zerocopy::KnownLayout, zerocopy::Unaligned)] #[repr(C)] pub(crate) struct DenseBitSlice { /// The number of admissible rows, `0..domain_size`. domain_size: U64, + /// The row domain the words index. marker: PhantomData, /// The member bits, one word per 64 domain rows. /// /// Bits at positions at or beyond `domain_size` in the final word are zero. [`Self::insert`] - /// refuses the - /// rows that would set one, and bit validity refuses the frames that carry one. + /// refuses the rows that would set one, and bit validity refuses the frames that carry one. words: [U64], } impl DenseBitSlice { /// Creates a set admitting no rows of a `domain_size`-row domain. + /// + /// # Panics + /// + /// Panics if the zeroed allocation fails. #[must_use] pub(crate) fn new_empty(domain_size: usize) -> Box { - // A domain held in memory occupies at most `isize::MAX` bytes, so its word count fits - // `usize`. + // rounding a usize domain up to words produces no more words than domain rows #[expect(clippy::cast_possible_truncation)] let words = num_words(domain_size as u64) as usize; let mut raw = RawDenseBitSlice::::new_box_zeroed_with_elems(words) .expect("the allocation for the set's words succeeds"); raw.domain_size = U64::new(domain_size as u64); - // SAFETY: Both types are `#[repr(C)]` structs with the same fields in the same order, so - // for every word count they share size, alignment, and slice-length metadata: the cast - // preserves the allocation's layout for the deallocation as well as for the view. The - // value also satisfies the frame invariant at the cast: the header was just written, the - // allocation carries exactly `num_words(domain_size)` words, and every word is zero, so - // no bit above the domain is set. + // SAFETY: Identical repr(C) fields give both types the same allocation layout and + // trailing-word metadata. The header now describes exactly the allocated word count, and + // the zeroed words have no excess bits set. Transferring the Box's unique ownership through + // this cast therefore yields a valid frame and preserves its deallocation layout. unsafe { Box::from_raw(Box::into_raw(raw) as *mut Self) } } /// Reads one frame off the front of `bytes`, returning the set and the remaining bytes. /// - /// The header's own word count frames the cast, and the frame invariant is checked inside it - /// as the type's bit validity. This door adds nothing to that validation - it splits the - /// buffer where the header says the frame ends, and it names which clause a refused frame - /// broke, which the [`zerocopy::TryFromBytes`] doors do not. + /// Reads at any byte alignment without copying. The header determines the frame length, and + /// validation rejects excess bits in its final word. /// /// # Errors /// - /// - [`ParseDenseBitSliceError::Header`]: the bytes end before the 8-byte domain header. - /// - [`ParseDenseBitSliceError::WordCount`]: the buffer carries fewer whole words than the - /// header's domain occupies. - /// - [`ParseDenseBitSliceError::ExcessBits`]: a bit above the domain is set in the final word. + /// Returns [`ParseDenseBitSliceError`] for a missing header, insufficient words or nonzero + /// excess bits, checked in that order. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -220,14 +214,17 @@ impl DenseBitSlice { } })?; + // zerocopy's plain prefix conversion chooses the largest word count that fits the buffer. + // Supplying the header's count preserves any following data as the remainder. Self::try_ref_from_prefix_with_elems(bytes, words).map_err(|error| match error { ConvertError::Alignment(_) => unreachable!("the set reads at any alignment"), ConvertError::Size(_) => ParseDenseBitSliceError::WordCount { domain_size, words: trailing.len() / WORD_BYTES, }, - // The cast's word count comes from the header, so the count clause of the frame - // invariant is true by construction and only excess bits can refuse validity. + // Frame validity requires the header's word count and zero excess bits. Supplying that + // word count to the cast satisfies the first condition. Only excess bits can cause a + // validity error. ConvertError::Validity(_) => ParseDenseBitSliceError::ExcessBits, }) } @@ -245,10 +242,10 @@ impl DenseBitSlice { )] const unsafe fn from_frame_unchecked(bytes: &[u8]) -> &Self { let words = (bytes.len() - WORD_BYTES) / WORD_BYTES; - // SAFETY: The type is a `repr(C)` DST of one 8-byte header and `words` trailing words at - // alignment 1. The frame's data pointer with the trailing word count as its metadata - // therefore denotes exactly `bytes`, and every byte of `bytes` is initialized. The frame - // invariant the caller guarantees is the type's bit validity. + // SAFETY: repr(C) places the alignment-one header before the trailing words, whose count is + // the DST metadata. The caller guarantees an exact valid frame, and the shared byte slice + // supplies initialized memory and its borrow lifetime. The reconstructed pointer therefore + // covers exactly that frame and may be borrowed for the same lifetime. unsafe { &*ptr::from_raw_parts(bytes.as_ptr(), words) } } @@ -264,16 +261,16 @@ impl DenseBitSlice { )] unsafe fn from_frame_unchecked_mut(bytes: &mut [u8]) -> &mut Self { let words = (bytes.len() - WORD_BYTES) / WORD_BYTES; - // SAFETY: As in `from_frame_unchecked`, and the borrow is exclusive because `bytes` is. + // SAFETY: The layout and valid-frame reasoning is the same as in `from_frame_unchecked`. + // The mutable byte slice supplies exclusive access for the returned borrow's lifetime. It + // is therefore sound to borrow this exact frame mutably. unsafe { &mut *ptr::from_raw_parts_mut(bytes.as_mut_ptr(), words) } } - /// Returns the length in bytes of the whole set over a `domain_size`-row domain: the 8-byte - /// header plus one word per 64 rows. + /// Returns the frame length in bytes for a `domain_size`-row domain. /// - /// This is what a file format reserves for the set, so a header's offset chain derives region - /// geometry from the domain alone. The arithmetic cannot overflow: the largest domain's word - /// count is far below `u64::MAX / 8`. + /// The length is `8 · (1 + ⌈domain_size / 64⌉)`, including the header. It fits in `u64` for + /// every `u64` domain size. #[must_use] pub(crate) const fn total_byte_len(domain_size: u64) -> u64 { (num_words(domain_size) + 1) * WORD_BYTES as u64 @@ -287,8 +284,8 @@ impl DenseBitSlice { /// Views the member bits as whole storage words, without copying. /// - /// One little-endian word per 64 rows of domain, rows LSB-first within the word. The frame - /// invariant zeroes every bit at or past the domain, so the padding bits read zero. + /// One little-endian word per 64 rows of domain, rows LSB-first within the word. Every bit at + /// or past the domain is zero under the frame invariant. #[must_use] pub(crate) const fn words(&self) -> &[U64] { &self.words @@ -305,7 +302,12 @@ impl DenseBitSlice { /// Sets `self = op(self, rhs)` word by word, reporting whether any word changed. /// - /// `domain` is the right-hand set's domain, asserted equal so the zip covers every word. + /// `rhs` must contain exactly the words of a set over `domain`. The operation must preserve + /// zero excess bits, including if it panics after modifying earlier words. + /// + /// # Panics + /// + /// Panics if `domain` differs from this set's domain. Propagates panics from `rhs` and `op`. fn apply u64>( &mut self, domain: u64, @@ -324,7 +326,7 @@ impl DenseBitSlice { let new = op(old, rhs); word.set(new); - // Accumulating the difference keeps the loop branch-free, so it vectorizes. + // loop-free means that LLVM has a change to vectorize here changed |= old ^ new; } @@ -359,7 +361,7 @@ impl DenseBitSlice { pub(crate) fn count_below(&self, index: T) -> u64 { let row = index.as_u64().min(self.domain_size.get()); - // The clamp bounds the split index by the word count, so both slices are in bounds. + // clamping to the domain keeps the word index at or below the stored word count #[expect(clippy::cast_possible_truncation)] let index = (row / WORD_BITS as u64) as usize; let full: u64 = self.words[..index] @@ -369,7 +371,8 @@ impl DenseBitSlice { let bit = row % WORD_BITS as u64; if bit == 0 { - // The row sits on a word boundary, where its word may lie past the final one. + // a word boundary needs no partial-word read, including the boundary after the final + // word return full; } @@ -378,9 +381,11 @@ impl DenseBitSlice { /// Sets `index` to `value`, returning whether the set changed. /// + /// Clearing a row outside the domain returns `false`. + /// /// # Panics /// - /// This panics when `index` lies outside the domain. + /// Panics if `value` is `true` and `index` lies outside the domain. #[inline] pub(crate) fn set(&mut self, index: T, value: bool) -> bool { if value { @@ -408,7 +413,7 @@ impl DenseBitSlice { /// Removes `index`, returning whether the set changed. /// - /// A row outside the domain was never admitted, so removing one reports no change. + /// Removing a row outside the domain never changes the set. pub(crate) fn remove(&mut self, index: T) -> bool { let index = index.as_u64(); if index >= self.domain_size.get() { @@ -421,6 +426,13 @@ impl DenseBitSlice { } /// Iterates the rows the set admits, in ascending order. + /// + /// Every admitted row must be representable by both `usize` and `T`. Byte-frame validation + /// checks the stored domain and padding, not these iteration bounds. + /// + /// # Panics + /// + /// Advancing the iterator panics if a row is outside `T`'s range. pub(crate) fn iter(&self) -> impl Iterator + '_ { self.words.iter().enumerate().flat_map(|(index, word)| { let mut bits = word.get(); @@ -509,10 +521,7 @@ impl Iterator for RowsIn<'_, T> { } } -/// Word-wise set relations against an in-memory set over the same domain. -/// -/// Every operation panics when the two sets draw from different domains. Both sides keep their -/// bits above the domain zero, so the word loops preserve the frame's final-word invariant. +// Both operands have zero excess bits. OR, AND-NOT and AND preserve those zeros. impl BitRelations> for DenseBitSlice { fn union(&mut self, other: &DenseBitSet) -> bool { self.apply( @@ -539,10 +548,6 @@ impl BitRelations> for DenseBitSlice { } } -/// Word-wise set relations against another set of the same shape over the same domain. -/// -/// Every operation panics when the two sets draw from different domains. Both sides keep their -/// bits above the domain zero, so the word loops preserve the frame's final-word invariant. impl BitRelations for DenseBitSlice { fn union(&mut self, other: &Self) -> bool { self.apply( @@ -583,21 +588,15 @@ impl fmt::Debug for DenseBitSlice { } } -/// Bit validity is the frame invariant: a memory range is a set only when its word count is what -/// its header's domain occupies and no bit above the domain is set in the final word. -/// -/// Zerocopy reserves this trait for its derive. The derived check is field validity alone and -/// cannot carry a cross-field predicate (google/zerocopy#1330 tracks that feature and its arrival -/// retires this impl). The impl therefore delegates the field-validity half to the derive where -/// it lives - on the invariant-free twin [`RawDenseBitSlice`] - and adds the frame predicate on -/// top. The delegation rides `#[doc(hidden)]` machinery the crate exempts from semver. Any -/// zerocopy upgrade therefore re-reviews this impl. -/// -/// SAFETY: `is_bit_valid` returns true only when the twin's derived `is_bit_valid` accepts the -/// bytes and the frame predicate holds on them. A valid `RawDenseBitSlice` is a valid -/// `DenseBitSlice` because the twins' field types are identical. The layout half of that claim is -/// compile-time-asserted by the cast in the body. Refusing valid-but-incoherent frames on top is -/// sound, because `is_bit_valid` may always be conservative. +// zerocopy reserves TryFromBytes for its derive and excludes the hidden validation APIs from +// compatibility guarantees. The manual implementation adds a cross-field predicate to derived field +// validation. Dependency upgrades must recheck this use of the hidden APIs. +// +// SAFETY: TryFromBytes requires every accepted candidate to be a valid value and permits +// conservative rejection. CastUnsized preserves the referent bytes, metadata and alignment, and +// RawDenseBitSlice has exactly the same field types. Its derived validation establishes field +// validity before the word-count and excess-bit checks. Every accepted candidate therefore has both +// valid fields and the complete frame invariant. unsafe impl zerocopy::TryFromBytes for DenseBitSlice { #[expect( dead_code, @@ -613,27 +612,28 @@ unsafe impl zerocopy::TryFromBytes for DenseBitSlice { where A: zerocopy::invariant::Alignment, { - // `CastUnsized` asserts at compile time that both types are slice DSTs with one - // alignment, one trailing-slice offset, and one element size, and it preserves the - // pointer metadata, so `raw` addresses exactly the candidate's bytes. + // `CastUnsized` checks at compile time that both slice DSTs have the same alignment, + // trailing-slice offset and element size. Casting this candidate preserves its pointer + // metadata. The resulting `raw` addresses exactly the candidate's bytes. let raw = candidate.cast::<_, zerocopy::pointer::cast::CastUnsized, _>(); if ! as zerocopy::TryFromBytes>::is_bit_valid(raw) { return false; } - // SAFETY: The twin's derived `is_bit_valid` accepted exactly these bytes. + // SAFETY: assume_valid requires a bit-valid raw representation. Its derived is_bit_valid + // predicate just accepted this candidate without changing its bytes or metadata. Therefore + // the same candidate may now be treated as a valid RawDenseBitSlice. let raw = unsafe { raw.assume_valid() }.unaligned_as_ref(); - // The generic doors hand this any word count that fits their bytes, so the count is - // checked before the excess arithmetic that assumes it. + // generic conversions can supply any word count, including one the header does not describe let domain_size = raw.domain_size.get(); let words = raw.words.len(); if num_words(domain_size) != words as u64 { return false; } - // The count matches the domain, so 0 ≤ excess < 64 and a nonzero excess leaves the - // shift below in `1..=63`. + // the mathematical excess after padding the domain to whole words is in [0, 63]. A nonzero + // excess leaves the shift count in [1, 63]. let excess = (words as u64) * (WORD_BITS as u64) - domain_size; excess == 0 || raw.words[words - 1].get() >> (WORD_BITS as u64 - excess) == 0 } @@ -719,60 +719,52 @@ impl core::error::Error for ParseDenseBitSliceArrayError { } } -/// [`DenseBitSliceArray`]'s fields with no region invariant coupling them. -/// -/// Every zerocopy claim is true here: any header beside any frame bytes is a value of this type. -/// That freedom is the twin's purpose. [`DenseBitSliceArray`]'s hand-written -/// [`zerocopy::TryFromBytes`] delegates field validity to the derive on these fields. -/// [`DenseBitSliceArray::new_empty`] builds its zeroed allocation here - a zeroed header beside -/// any byte count breaks nothing - and casts once the region invariant is in place. +/// Unvalidated storage for constructing and checking a [`DenseBitSliceArray`]. /// -/// The fields mirror [`DenseBitSliceArray`]'s exactly. The layout half of that claim is asserted -/// at compile time by the cast in `is_bit_valid`. The bit-validity half rests on the field types -/// being identical. A field type change in either twin therefore re-derives that proof. +/// The field types and their order must match `DenseBitSliceArray`. Any initialized header and +/// trailing bytes are valid here, including a zeroed allocation whose frame headers have not been +/// written. #[derive(zerocopy::FromBytes, zerocopy::Immutable, zerocopy::KnownLayout, zerocopy::Unaligned)] #[repr(C)] struct RawDenseBitSliceArray { /// [`DenseBitSliceArray`]'s region header, not yet coupled to the frame bytes. domain_size: U64, + /// The row domain every frame indexes. marker: PhantomData, - /// [`DenseBitSliceArray`]'s frames, not yet policed for shape. + /// The trailing bytes, not yet validated as frames. frames: [u8], } -/// An array of same-domain [`DenseBitSlice`] frames behind one domain header, in one contiguous -/// byte region. +/// A contiguous byte array of same-domain [`DenseBitSlice`] frames. /// -/// A file's dense region is `count` membership sets over one shared domain. This type is that -/// region in memory, and it has the frame's own shape one level up: an 8-byte domain header, then -/// the frames back to back at the shared stride of [`DenseBitSlice::total_byte_len`] bytes. -/// [`DenseBitSliceArray::new_empty`] makes one allocation and writes every header, indexing -/// borrows one frame as a real [`DenseBitSlice`], and the [`zerocopy::IntoBytes`] bytes are the -/// region exactly as a file stores it, so a file write emits the array's bytes verbatim. +/// The byte format is an 8-byte little-endian domain header followed by the frames, each +/// [`DenseBitSlice::total_byte_len`] bytes long. The array header records the domain even when +/// there are no frames. Each frame repeats its domain header, allowing indexing to borrow a +/// self-describing `DenseBitSlice`. [`zerocopy::IntoBytes`] exposes the complete region without +/// encoding or copying. /// -/// The array's own header keeps every accessor total - an array of no frames still states its -/// domain, so the geometry never depends on a first frame existing. Every frame restates that -/// domain in its own header. The repetition keeps each element a self-describing frame, so -/// indexing returns a borrow of the element type itself. +/// Every [`zerocopy::TryFromBytes`] conversion validates that the trailing bytes contain a whole +/// number of valid frames over exactly the array's domain. [`Self::try_from_bytes`] additionally +/// checks the expected domain and count. Indexing borrows an already-validated frame and panics +/// when its index is outside the frame count. /// -/// The invariant - the domain header, then a whole number of valid frames over exactly that -/// domain - is the type's bit validity, so every [`zerocopy::TryFromBytes`] door validates it -/// inside the cast and no door can mint an incoherent region. The array's own doors add to that: -/// [`DenseBitSliceArray::new_empty`] builds the invariant, -/// [`DenseBitSliceArray::try_from_bytes`] checks the region against the caller's expected domain -/// and count and names which clause a refused region broke, and the unsafe -/// [`DenseBitSliceArray::from_bytes_unchecked`] re-borrows bytes a previous validation accepted. -/// Indexing trusts the doors and revalidates nothing. +/// # Platform behavior +/// +/// On 32-bit targets, a header-only array can describe a frame stride larger than `usize::MAX`. +/// Parsing accepts that array, but [`Self::len`] and indexing panic when converting its stride. /// /// Arrays are equal when they cover one domain and carry the same frames. Canonical frames make /// that byte equality: equal domains fix the word count, and bits above the domain are zero on /// both sides. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the types are crate-private. /// /// ```ignore /// use zerocopy::IntoBytes as _; /// +/// use hashql_core::id::Id as _; /// use crate::bitset::DenseBitSliceArray; /// use crate::identity::BasePosition; /// @@ -784,16 +776,16 @@ struct RawDenseBitSliceArray { /// let read = DenseBitSliceArray::::try_from_bytes(bytes, 1_000, 2)?; /// assert!(read[0].contains(BasePosition::from_u32(3))); /// assert!(read[1].contains(BasePosition::from_u32(64))); +/// # Ok::<(), crate::bitset::ParseDenseBitSliceArrayError>(()) /// ``` -// No `FromZeros`: its zeroed constructors take any frame byte count. They would therefore mint -// regions whose frame bytes are not a whole number of frames in safe code, bypassing the doors -// whose validation indexing trusts. `RawDenseBitSliceArray` carries the zeroed allocation -// instead. +// FromZeros constructors accept any trailing byte count, including incomplete frames. Allocate +// through RawDenseBitSliceArray until every header and the region geometry are valid. #[derive(zerocopy::IntoBytes, zerocopy::Immutable, zerocopy::KnownLayout, zerocopy::Unaligned)] #[repr(C)] pub(crate) struct DenseBitSliceArray { /// The domain every frame draws from. domain_size: U64, + /// The row domain every frame indexes. marker: PhantomData, /// The frames, back to back at one stride. frames: [u8], @@ -802,8 +794,12 @@ pub(crate) struct DenseBitSliceArray { impl DenseBitSliceArray { /// Creates `count` sets each admitting no rows of a `domain_size`-row domain. /// - /// One zeroed allocation of the region, with the domain header and each frame's restatement - /// of it written in place: the dense region of a file whose sets hold nothing yet. + /// Allocates one region with the array header and every frame header initialized. + /// + /// # Panics + /// + /// Panics if the region size is not representable as an allocation layout or the zeroed + /// allocation fails. #[must_use] pub(crate) fn new_empty(domain_size: usize, count: usize) -> Box { let stride = usize::try_from(DenseBitSlice::::total_byte_len(domain_size as u64)) @@ -821,32 +817,23 @@ impl DenseBitSliceArray { .expect("every frame holds at least its 8-byte domain header"); } - // SAFETY: Both types are `#[repr(C)]` structs with the same fields in the same order. For - // every frame byte count they therefore share size, alignment, and slice-length metadata - - // the cast preserves the allocation's layout for the deallocation as well as for the view. - // The value also satisfies the array invariant at the cast. The region header was written - // above. The frame bytes are exactly `count` whole strides. Each stride is the valid empty - // frame - its own copy of the domain header, then zero words. + // SAFETY: Identical repr(C) fields give both types the same allocation layout and + // trailing-byte metadata. The region header is initialized, and every one of the count + // whole strides contains a matching header followed by zero words. Transferring the Box's + // unique ownership through this cast therefore yields a valid array and preserves its + // deallocation layout. unsafe { Box::from_raw(Box::into_raw(raw) as *mut Self) } } /// Borrows exactly `count` frames over a `domain_size`-row domain from `bytes`. /// - /// This is the validating door. It checks the byte length against the geometry, then the - /// region's domain header against the caller's, then every frame against that domain. An - /// array borrowed from a file region therefore upholds the type's invariant with no state - /// beside the bytes. + /// Checks the byte length, the array's domain header, then each frame and its domain in rank + /// order. Borrows the validated bytes at any alignment, without copying. /// /// # Errors /// - /// - [`ParseDenseBitSliceArrayError::Length`]: the region's byte length is not the header plus - /// `count` strides. - /// - [`ParseDenseBitSliceArrayError::Header`]: the region's domain header claims a domain other - /// than `domain_size`. - /// - [`ParseDenseBitSliceArrayError::Frame`]: a frame breaks the frame layout, and the wrapped - /// [`ParseDenseBitSliceError`] names the broken clause. - /// - [`ParseDenseBitSliceArrayError::Domain`]: a frame claims a domain other than - /// `domain_size`. + /// Returns [`ParseDenseBitSliceArrayError`] if the region disagrees with the expected geometry + /// or contains an invalid or differently domained frame. pub(crate) fn try_from_bytes( bytes: &[u8], domain_size: u64, @@ -880,8 +867,12 @@ impl DenseBitSliceArray { /// Checks that `frames` holds whole frames over exactly a `domain_size`-row domain. /// - /// The caller has already checked that `frames` is a whole number of strides, which is what - /// bounds the stride by the region length below. + /// `frames` must be a whole number of strides, established by the calling length check. + /// + /// # Errors + /// + /// Returns [`ParseDenseBitSliceArrayError`] at the first invalid or differently domained frame, + /// in rank order. fn validate_frames( domain_size: u64, frames: &[u8], @@ -890,8 +881,8 @@ impl DenseBitSliceArray { return Ok(()); } - // A nonempty whole-stride region is at least one stride long, so the stride fits the - // length the region already occupies in memory. + // A nonempty whole-stride region contains at least one stride. This region occupies a slice + // with a `usize` length. Its stride is bounded by that length and fits `usize`. let stride = usize::try_from(DenseBitSlice::::total_byte_len(domain_size)) .expect("the region's length bounds its stride"); for (rank, frame) in frames.chunks_exact(stride).enumerate() { @@ -918,23 +909,21 @@ impl DenseBitSliceArray { /// /// # Safety /// - /// `bytes` must uphold the array invariant: the 8-byte domain header, then a whole number of - /// valid frames over exactly that domain. Bytes a previous - /// [`DenseBitSliceArray::try_from_bytes`] or [`zerocopy::TryFromBytes`] door accepted uphold - /// it. + /// `bytes` must be exactly the array's 8-byte little-endian domain header followed by a whole + /// number of valid frames over that same domain. Bytes accepted by [`Self::try_from_bytes`] + /// satisfy this invariant. #[must_use] pub(crate) const unsafe fn from_bytes_unchecked(bytes: &[u8]) -> &Self { - // SAFETY: The type is a `repr(C)` DST of one 8-byte header and a trailing byte slice at - // alignment 1. The region's data pointer with the trailing byte count as its metadata - // therefore denotes exactly `bytes`, and every byte of `bytes` is initialized. + // SAFETY: repr(C) places the alignment-one header before the trailing byte slice, whose + // length is the DST metadata. The caller guarantees an exact valid array, and the shared + // byte slice supplies initialized memory and its borrow lifetime. The reconstructed pointer + // therefore covers exactly that array and may be borrowed for the same lifetime. unsafe { &*ptr::from_raw_parts(bytes.as_ptr(), bytes.len() - WORD_BYTES) } } - /// Returns the length in bytes of a whole array: the 8-byte domain header plus `count` - /// frames over the domain. + /// Returns the byte length of the array header and `count` frames. /// - /// This is what a file format reserves for the region. Returns `None` when the geometry - /// overflows `u64`, in which case no real region matches it. + /// Returns [`None`] when the geometry overflows `u64`. #[must_use] pub(crate) const fn total_byte_len(domain_size: u64, count: u64) -> Option { let Some(frames) = count.checked_mul(DenseBitSlice::::total_byte_len(domain_size)) @@ -963,6 +952,10 @@ impl DenseBitSliceArray { } /// Returns the byte stride of one frame. + /// + /// # Panics + /// + /// Panics if the frame stride exceeds `usize::MAX`. fn stride(&self) -> usize { usize::try_from(DenseBitSlice::::total_byte_len(self.domain_size.get())) .expect("a resident frame fits the address space") @@ -972,7 +965,7 @@ impl DenseBitSliceArray { /// /// # Panics /// - /// This panics when `rank` lies at or beyond the frame count. + /// Panics if `rank` lies at or beyond the frame count or the frame stride exceeds `usize::MAX`. fn frame_range(&self, rank: usize) -> Range { assert!( rank < self.len(), @@ -987,30 +980,23 @@ impl DenseBitSliceArray { impl Index for DenseBitSliceArray { type Output = DenseBitSlice; - /// Views the frame at rank `index`. - /// - /// # Panics - /// - /// This panics when `index` lies at or beyond the frame count. fn index(&self, index: usize) -> &DenseBitSlice { let frame = &self.frames[self.frame_range(index)]; - // SAFETY: Every door of the array validated its frames, and `frame_range` carves exactly - // one whole frame out of the frame region. + // SAFETY: from_frame_unchecked requires exactly one valid frame. The array invariant + // establishes frame validity, and frame_range selects one complete stride. The subslice + // therefore satisfies the constructor's contract. unsafe { DenseBitSlice::from_frame_unchecked(frame) } } } impl IndexMut for DenseBitSliceArray { - /// Views the frame at rank `index` mutably. - /// - /// # Panics - /// - /// This panics when `index` lies at or beyond the frame count. fn index_mut(&mut self, index: usize) -> &mut DenseBitSlice { let range = self.frame_range(index); let frame = &mut self.frames[range]; - // SAFETY: As in `index`. Mutation through a `DenseBitSlice` preserves the frame - // invariant, so the exclusive borrow keeps the array invariant too. + // SAFETY: The array invariant and frame_range establish one complete valid frame, as in + // index. The subslice is exclusively borrowed, and DenseBitSlice mutation preserves its + // header, word count and zero excess bits. The mutable view therefore preserves the array + // invariant. unsafe { DenseBitSlice::from_frame_unchecked_mut(frame) } } } @@ -1031,21 +1017,14 @@ impl fmt::Debug for DenseBitSliceArray { } } -/// Bit validity is the region invariant: a memory range is an array only when its frame bytes -/// are a whole number of valid frames over exactly the domain its own header claims. -/// -/// Zerocopy reserves this trait for its derive. The derived check is field validity alone and -/// cannot carry a cross-field predicate (google/zerocopy#1330 tracks that feature and its arrival -/// retires this impl). The impl therefore delegates the field-validity half to the derive where -/// it lives - on the invariant-free twin [`RawDenseBitSliceArray`] - and adds the region -/// predicate on top. The delegation rides `#[doc(hidden)]` machinery the crate exempts from -/// semver. Any zerocopy upgrade therefore re-reviews this impl. -/// -/// SAFETY: `is_bit_valid` returns true only when the twin's derived `is_bit_valid` accepts the -/// bytes and the region predicate holds on them. A valid `RawDenseBitSliceArray` is a valid -/// `DenseBitSliceArray` because the twins' field types are identical. The layout half of that -/// claim is compile-time-asserted by the cast in the body. Refusing valid-but-incoherent regions -/// on top is sound, because `is_bit_valid` may always be conservative. +// This manual TryFromBytes implementation has the same hidden-API dependency as the DenseBitSlice +// implementation above. +// +// SAFETY: TryFromBytes requires every accepted candidate to be a valid value and permits +// conservative rejection. CastUnsized preserves the referent bytes, metadata and alignment, and +// RawDenseBitSliceArray has exactly the same field types. Derived field validation precedes the +// whole-stride check and validation of every frame against the array domain. Every +// accepted candidate therefore has both valid fields and the complete array invariant. unsafe impl zerocopy::TryFromBytes for DenseBitSliceArray { #[expect( dead_code, @@ -1061,19 +1040,20 @@ unsafe impl zerocopy::TryFromBytes for DenseBitSliceArray { where A: zerocopy::invariant::Alignment, { - // `CastUnsized` asserts at compile time that both types are slice DSTs with one - // alignment, one trailing-slice offset, and one element size, and it preserves the - // pointer metadata, so `raw` addresses exactly the candidate's bytes. + // `CastUnsized` checks at compile time that both slice DSTs have the same alignment, + // trailing-slice offset and element size. Casting this candidate preserves its pointer + // metadata. The resulting `raw` addresses exactly the candidate's bytes. let raw = candidate.cast::<_, zerocopy::pointer::cast::CastUnsized, _>(); if ! as zerocopy::TryFromBytes>::is_bit_valid(raw) { return false; } - // SAFETY: The twin's derived `is_bit_valid` accepted exactly these bytes. + // SAFETY: assume_valid requires a bit-valid raw representation. Its derived is_bit_valid + // predicate just accepted this candidate without changing its bytes or metadata. Therefore + // the same candidate may now be treated as a valid RawDenseBitSliceArray. let raw = unsafe { raw.assume_valid() }.unaligned_as_ref(); - // The generic doors hand this any byte count, so whole strides are checked before the - // frame walk that assumes them. + // generic conversions can supply any byte count, including an incomplete final frame let domain_size = raw.domain_size.get(); let stride = DenseBitSlice::::total_byte_len(domain_size); if !(raw.frames.len() as u64).is_multiple_of(stride) { diff --git a/libs/@local/graph/atlas/src/bitset/tests.rs b/libs/@local/graph/atlas/src/bitset/tests.rs index aad37ca9691..4257f5210ca 100644 --- a/libs/@local/graph/atlas/src/bitset/tests.rs +++ b/libs/@local/graph/atlas/src/bitset/tests.rs @@ -20,6 +20,8 @@ fn starts_empty() { assert!(!set.contains(NodeRowId::new(0))); } +/// Insertion reports whether the set changed, membership follows insertion, and iteration is +/// ascending across roaring's `2^16` container boundary regardless of insertion order. #[test] fn inserted_rows_are_contained_and_iterated_in_order() { // The rows straddle roaring's container boundary at 2^16, so the @@ -119,6 +121,7 @@ fn rows_above_the_representable_domain_read_absent() { assert_eq!(set.count(), 1); } +/// Inserting a row at or above `2^32` panics with the domain message. #[test] #[should_panic(expected = "the row lies in the representable domain")] fn insert_rejects_rows_above_the_representable_domain() { @@ -182,6 +185,8 @@ fn dense_bit_slice_starts_empty() { assert!(!set.contains(NodeRowId::new(0))); } +/// Insertion reports whether the frame changed, membership follows insertion, and iteration is +/// ascending across the 64-bit word boundary regardless of insertion order. #[test] fn dense_bit_slice_inserted_rows_are_contained_and_iterated_in_order() { // The rows straddle the 64-bit word boundary, so the iteration order crosses words. @@ -203,6 +208,8 @@ fn dense_bit_slice_inserted_rows_are_contained_and_iterated_in_order() { assert!(!set.contains(NodeRowId::new(128))); } +/// Removing a member returns true and drops it. Removing it again returns false and changes +/// nothing. #[test] fn dense_bit_slice_removal_reports_whether_the_set_changed() { let mut set = DenseBitSlice::new_empty(130); @@ -225,6 +232,7 @@ fn dense_bit_slice_rows_outside_the_domain_read_absent() { assert_eq!(set.count(), 1); } +/// Inserting the row equal to the domain size panics with the domain message. #[test] #[should_panic(expected = "the row lies in the set's domain")] fn dense_bit_slice_insert_rejects_rows_outside_the_domain() { @@ -285,6 +293,7 @@ fn dense_bit_slice_words_cross_the_word_boundary() { assert_eq!(set.words().as_bytes(), expected); } +/// The empty domain occupies no words at all. #[test] fn dense_bit_slice_zero_domain_packs_to_no_words() { let set = DenseBitSlice::::new_empty(0); @@ -370,6 +379,7 @@ fn dense_bit_slice_relations_against_an_in_memory_set() { ); } +/// A relation against an in-memory set over a different domain panics before touching a word. #[test] #[should_panic(expected = "the sets draw from the same domain")] fn dense_bit_slice_relations_reject_mismatched_domains() { @@ -378,6 +388,8 @@ fn dense_bit_slice_relations_reject_mismatched_domains() { slice.union(&other); } +/// `total_byte_len` is the 8-byte header plus 8 bytes per word the domain occupies, and a built +/// frame's byte length equals it. #[test] fn dense_bit_slice_total_byte_len_counts_the_header_and_the_words() { // The empty domain still carries its 8-byte header; 64 rows fill exactly one word; 65 spill @@ -451,6 +463,7 @@ fn dense_bit_slice_relations_apply_between_slices() { assert_eq!(target.count(), 0); } +/// A relation between two frames over different domains panics before touching a word. #[test] #[should_panic(expected = "the sets draw from the same domain")] fn dense_bit_slice_relations_reject_mismatched_slice_domains() { @@ -459,6 +472,7 @@ fn dense_bit_slice_relations_reject_mismatched_slice_domains() { target.union(&*other); } +/// Indexing an array at its frame count panics with the rank message. #[test] #[should_panic(expected = "the rank names one of the array's frames")] fn dense_bit_slice_array_rejects_ranks_beyond_the_frames() { @@ -497,6 +511,8 @@ mod miri { assert!(rest.is_empty()); } + /// A zero-domain frame is exactly its 8-byte header and parses back to an empty set over + /// domain zero. #[test] fn dense_bit_slice_zero_domain_frames_parse() { let set = DenseBitSlice::::new_empty(0); @@ -587,6 +603,8 @@ mod miri { assert_eq!(rest, [0xAB; 8]); } + /// A fresh array reports its frame count and domain, occupies exactly `total_byte_len`, and + /// every frame is empty over the array's domain. #[test] fn dense_bit_slice_array_starts_as_empty_frames() { let sets = DenseBitSliceArray::::new_empty(130, 3); @@ -604,6 +622,8 @@ mod miri { } } + /// Insertions through `IndexMut` reach the addressed frame alone, and the other frames keep + /// their own members. #[test] fn dense_bit_slice_array_indexes_independent_frames() { let mut sets = DenseBitSliceArray::::new_empty(130, 3); @@ -751,6 +771,7 @@ mod miri { assert_eq!(read.domain_size(), 64); } + /// An array of no frames panics on rank zero with the rank message. #[test] #[should_panic(expected = "the rank names one of the array's frames")] fn dense_bit_slice_array_of_no_frames_rejects_every_rank() { diff --git a/libs/@local/graph/atlas/src/cli/dump.rs b/libs/@local/graph/atlas/src/cli/dump.rs index 931c2602d19..16165f883ce 100644 --- a/libs/@local/graph/atlas/src/cli/dump.rs +++ b/libs/@local/graph/atlas/src/cli/dump.rs @@ -1,10 +1,10 @@ //! The dump command that writes an offline dataset from the live store. //! //! One invocation drains the store's snapshot into a directory an -//! [`OfflineDataset`](crate::dataset::offline::OfflineDataset) accepts, embeddings included, so a -//! fit can run on a machine that reaches neither Postgres nor the embedding provider. The command -//! embeds through the same fingerprinted provider contract the fit records, and the manifest -//! seals the directory last, so an interrupted dump leaves a directory the reader refuses. +//! [`OfflineDataset`](crate::dataset::offline::OfflineDataset) accepts, embeddings included. A fit +//! can run on a machine that reaches neither Postgres nor the embedding provider. The command +//! embeds through the same fingerprinted provider contract the fit records, and the manifest seals +//! the directory last. An interrupted dump leaves a directory the reader refuses. use core::{error::Error, fmt, num::NonZero, time::Duration}; use std::time::Instant; @@ -40,16 +40,22 @@ pub struct DumpArgs { /// The fit seed the canonical sample derives from. /// - /// Under probe coverage an offline fit replays the sample from its own seed, so the fit's - /// seed must equal this one. + /// Under probe coverage an offline fit replays the sample from its own seed. The fit's seed + /// must equal this one. + /// + /// Defaults to `0` when neither the flag nor `HASH_GRAPH_ATLAS_SEED` supplies one. #[arg(long, env = "HASH_GRAPH_ATLAS_SEED", default_value_t = 0)] seed: u64, /// Sampled anchor rows of the admission probe. + /// + /// Defaults to `1024`. #[arg(long, default_value = "1024")] anchors: NonZero, /// Sampled comparison rows of the admission probe. + /// + /// Defaults to `4096`. #[arg(long, default_value = "4096")] comparisons: NonZero, @@ -61,9 +67,9 @@ pub struct DumpArgs { /// Path of the annotation-corpus document the offline fit will run with. /// - /// The dump assembles the corpus and merges its card embeddings into the dump, so the offline - /// fit resolves every text it renders. A fit supplied with a corpus the dump never assembled - /// would request embeddings the dump does not hold. + /// The dump assembles the corpus and merges its card embeddings into the dump. The offline fit + /// resolves every text it renders. A fit supplied with a corpus the dump never assembled would + /// request embeddings the dump does not hold. #[arg(long, env = "HASH_GRAPH_ATLAS_ANNOTATIONS", value_hint = ValueHint::FilePath)] annotations: Option, @@ -177,11 +183,11 @@ impl DumpCommand { /// Writes one dump of the live store and returns its verdict. /// - /// The hosting binary supplies the dialed store connection. This call pins the snapshot, so - /// the dump reads the store as of the moment the command starts, and the manifest records the - /// snapshot's temporal axes for the offline fit to replay. The snapshot closes once the - /// store's streams are drained, so no transaction stays open while the embedding pass - /// round-trips to the provider. + /// The hosting binary supplies the dialed store connection. This call pins the snapshot. The + /// dump reads the store as of the moment the command starts, and the manifest records the + /// snapshot's temporal axes for the offline fit to replay. The snapshot closes once the store's + /// streams are drained. No transaction stays open while the embedding pass round-trips to the + /// provider. /// /// # Errors /// @@ -191,6 +197,11 @@ impl DumpCommand { /// - [`DumpError::Embedder`] when producing the embedding provider fails. /// - [`DumpError::Snapshot`] when the store cannot open the snapshot transaction. /// - [`DumpError::Dump`] when writing the dump directory fails. + /// + /// # Panics + /// + /// [`verify_cpu_baseline`](crate::math::kernel::verify_cpu_baseline) runs first and rejects a + /// CPU below the compiled baseline, on the conditions it documents. pub async fn run(self, client: &mut Client) -> Result { // Embedders reach this entry without passing through the shell's main. crate::math::kernel::verify_cpu_baseline(); @@ -226,10 +237,10 @@ impl DumpCommand { comparisons: self.args.comparisons, all_canonicals: self.args.all_canonicals, annotations: supplied.as_ref().map(SuppliedAnnotations::document), - // The fit's own configuration takes this same crate default, so the dump embeds - // exactly the texts the offline fit's assembly will render. A fit run with a - // different assembly requests texts the dump never embedded, and the offline - // embedder refuses them by hash rather than serving stale vectors. + // The fit's own configuration takes this same crate default. The dump embeds exactly + // the texts the offline fit's assembly will render. A fit run with a different assembly + // requests texts the dump never embedded, and the offline embedder refuses them by hash + // rather than serving stale vectors. assembly: AssemblyConfig { .. }, }; @@ -239,8 +250,8 @@ impl DumpCommand { let reading = read(&dataset, &self.args.output, &options) .await .map_err(DumpError::Dump)?; - // The reading borrows nothing from the dataset, so the snapshot transaction ends here - // rather than spanning the provider round-trips below. + // The reading borrows nothing from the dataset. The snapshot transaction ends here rather + // than spanning the provider round-trips below. drop(dataset); let finished = embed(&embedder, reading, &self.args.output, &options, &NoProgress) diff --git a/libs/@local/graph/atlas/src/cli/fit.rs b/libs/@local/graph/atlas/src/cli/fit.rs index 1973482822a..ce346918445 100644 --- a/libs/@local/graph/atlas/src/cli/fit.rs +++ b/libs/@local/graph/atlas/src/cli/fit.rs @@ -31,11 +31,19 @@ use crate::{ reason = "the flags are independent operator switches" )] pub struct FitArgs { - /// The run seed; equal seeds replay every draw, the admission probe's included. + /// The run seed. + /// + /// Every draw in the run derives from it, the admission probe's included. A second run + /// repeats a draw when it also presents the same inputs to the same backend and consumes the + /// stream in the same order. + /// + /// Defaults to `0` when neither the flag nor `HASH_GRAPH_ATLAS_SEED` supplies one. #[arg(long, env = "HASH_GRAPH_ATLAS_SEED", default_value_t = 0)] seed: u64, /// The landmark capacity. + /// + /// Defaults to `4096`. #[arg(long, default_value = "4096")] landmarks: NonZero, @@ -44,10 +52,14 @@ pub struct FitArgs { fresh: bool, /// Sampled anchor rows of the admission probe. + /// + /// Defaults to `1024`. #[arg(long, default_value = "1024")] anchors: NonZero, /// Sampled comparison rows of the admission probe. + /// + /// Defaults to `4096`. #[arg(long, default_value = "4096")] comparisons: NonZero, @@ -107,7 +119,9 @@ pub struct FitArgs { #[arg(long)] nn_descent: bool, - /// Where the admission report JSON lands. + /// Destination of the admission report JSON. + /// + /// Defaults to `admission-report.json` in the working directory. #[arg(long, default_value = "admission-report.json", value_hint = ValueHint::FilePath)] report: Utf8PathBuf, } @@ -275,8 +289,12 @@ where /// /// # Errors /// - /// Returns a [`FitError`] naming the step that failed: producing the embedding provider, the - /// run itself, or writing the admission report. + /// Returns [`FitError`] on embedding-provider preparation, fitting or report-write failure. + /// + /// # Panics + /// + /// [`verify_cpu_baseline`](crate::math::kernel::verify_cpu_baseline) runs first and rejects a + /// CPU below the compiled baseline, on the conditions it documents. /// /// [`PostgresArgs::connect`]: super::PostgresArgs::connect /// [`connect`]: super::connect diff --git a/libs/@local/graph/atlas/src/cli/mod.rs b/libs/@local/graph/atlas/src/cli/mod.rs index 09027b0472c..c28037052c0 100644 --- a/libs/@local/graph/atlas/src/cli/mod.rs +++ b/libs/@local/graph/atlas/src/cli/mod.rs @@ -1,7 +1,8 @@ //! The operator commands that fit a generation. //! //! The `hash-graph atlas` subcommand is one entry point. [`FitArgs`] and [`FitCommand`] run one -//! production generation over the live store. +//! production generation over the live store. `ServeArgs` and `ServeCommand` construct the +//! read-API router (`crate::api`) and generation maintenance that the graph binary retains. //! //! The standalone `hash-graph-atlas` binary is the other entry point, and the `cli` feature gates //! its shell. Its command line carries the fit command over its own store flags ([`PostgresArgs`]) @@ -26,7 +27,7 @@ //! [`PostgresArgs::connect`] dials the shell's own flags field by field, [`connect`] dials a //! rendered connection string. //! -//! The run entry points the fit command drives live with the runner; this module re-exports their +//! The fit command drives the run entry points live with the runner. This module re-exports their //! vocabulary ([`Options`], [`Placement`], [`ClassifierSource`], [`Summary`], [`RunError`]) as the //! crate's operator API. //! @@ -76,6 +77,11 @@ pub struct RootArgs { )] root: GenerationRoot, + /// The device this invocation's tensor work runs on. + /// + /// A command that computes no tensors reads the root without it, as the quality report does. + /// Defaults to [`PinnedDevice::host`], the platform's default family at ordinal 0, when + /// neither the flag nor `HASH_GRAPH_ATLAS_DEVICE` names one. #[arg( long, env = "HASH_GRAPH_ATLAS_DEVICE", @@ -85,6 +91,11 @@ pub struct RootArgs { } /// Parses a generation-root argument: opens the root, creating the directory when absent. +/// +/// # Errors +/// +/// Returns an [`io::Error`] when the directory cannot be created or opened, which clap renders +/// as the flag's refusal. fn parse_root(value: &str) -> io::Result { GenerationRoot::new(value) } diff --git a/libs/@local/graph/atlas/src/cli/postgres.rs b/libs/@local/graph/atlas/src/cli/postgres.rs index c16f26c1e90..b9b45d707cc 100644 --- a/libs/@local/graph/atlas/src/cli/postgres.rs +++ b/libs/@local/graph/atlas/src/cli/postgres.rs @@ -10,15 +10,20 @@ use crate::integrity::SecretString; /// The store connection flags, mirroring the graph binary's `HASH_GRAPH_PG_*` environment. /// -/// [`connect`](Self::connect) dials what the flags name; one deployment configuration drives the +/// [`connect`](Self::connect) dials what the flags name. One deployment configuration drives the /// graph binary and the standalone binary alike. #[derive(Debug, Args)] pub struct PostgresArgs { /// The store username. + /// + /// Defaults to `postgres` when neither the flag nor `HASH_GRAPH_PG_USER` supplies one. #[arg(long, default_value = "postgres", env = "HASH_GRAPH_PG_USER")] user: String, /// The store password. + /// + /// Defaults to `postgres` when neither the flag nor `HASH_GRAPH_PG_PASSWORD` supplies one. + /// The environment value stays out of `--help`. #[arg( long, default_value = "postgres", @@ -28,14 +33,20 @@ pub struct PostgresArgs { password: SecretString, /// The store host. + /// + /// Defaults to `localhost` when neither the flag nor `HASH_GRAPH_PG_HOST` supplies one. #[arg(long, default_value = "localhost", env = "HASH_GRAPH_PG_HOST")] host: String, /// The store port. + /// + /// Defaults to `5432` when neither the flag nor `HASH_GRAPH_PG_PORT` supplies one. #[arg(long, default_value_t = 5432, env = "HASH_GRAPH_PG_PORT")] port: u16, /// The database name. + /// + /// Defaults to `graph` when neither the flag nor `HASH_GRAPH_PG_DATABASE` supplies one. #[arg(long, default_value = "graph", env = "HASH_GRAPH_PG_DATABASE")] database: String, } @@ -43,7 +54,7 @@ pub struct PostgresArgs { impl PostgresArgs { /// Dials the store the flags name and drives the connection on a background task. /// - /// The flags configure the connection field by field, so the password never rides a rendered + /// The flags configure the connection field by field. The password never rides a rendered /// connection string and one containing URL-reserved characters needs no escaping. /// /// # Errors @@ -51,7 +62,7 @@ impl PostgresArgs { /// Returns a [`ConnectError`] when the store refuses the connection or handshake. pub async fn connect(self) -> Result { // The guard owns the password buffer and zeroizes it when this scope ends. The store - // config copies the bytes it is shown, and that copy is the library's own. + // config copies the bytes it receives, and that copy is the library's own. let password = self.password.expose(); let mut config = Config::new(); config @@ -112,6 +123,11 @@ pub async fn connect(dsn: &str) -> Result { } /// Dials the store the configuration names and drives the connection on a background task. +/// +/// # Errors +/// +/// Returns a [`ConnectError`] when the configuration names no TCP host, when the socket refuses +/// the connection, or when the store rejects the handshake. async fn dial(config: Config) -> Result { let host = config .get_hosts() diff --git a/libs/@local/graph/atlas/src/cli/report/classifier.rs b/libs/@local/graph/atlas/src/cli/report/classifier.rs index ffb90c599c9..991d6bb28ca 100644 --- a/libs/@local/graph/atlas/src/cli/report/classifier.rs +++ b/libs/@local/graph/atlas/src/cli/report/classifier.rs @@ -19,6 +19,8 @@ pub(crate) struct ClassifierArgs { generation: GenerationId, /// Where the report bundle JSON lands. + /// + /// Defaults to `classifier-report.json` in the working directory. #[arg(long, default_value = "classifier-report.json", value_hint = ValueHint::FilePath)] output: Utf8PathBuf, } @@ -54,9 +56,10 @@ impl Display for ClassifierVerdict { } impl ClassifierArgs { - /// Refits the generation's classifier from its staged corpus and certifies the bytes against - /// the deployed artifact, then writes the report bundle. A digest mismatch is the bundle's - /// content, so it lands in the report and the verdict instead of failing the run. + /// Refits the generation's classifier from its staged corpus and writes the report bundle. + /// + /// The refit certifies the bytes against the deployed artifact. A digest mismatch is the + /// bundle's content. The bundle and the verdict carry it instead of the run failing. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/cli/report/clumps.rs b/libs/@local/graph/atlas/src/cli/report/clumps.rs index d70443f6f89..80c01849917 100644 --- a/libs/@local/graph/atlas/src/cli/report/clumps.rs +++ b/libs/@local/graph/atlas/src/cli/report/clumps.rs @@ -17,6 +17,8 @@ pub(crate) struct ClumpArgs { table: Utf8PathBuf, /// Candidate distance thresholds, in the order the report lists them. + /// + /// Defaults to [`DEFAULT_EPSILONS`], the published calibration grid. #[arg(long = "epsilon", value_delimiter = ',', default_values_t = DEFAULT_EPSILONS.to_vec())] epsilons: Vec, } diff --git a/libs/@local/graph/atlas/src/cli/report/knn.rs b/libs/@local/graph/atlas/src/cli/report/knn.rs index 136478e8c19..6863fefc9a2 100644 --- a/libs/@local/graph/atlas/src/cli/report/knn.rs +++ b/libs/@local/graph/atlas/src/cli/report/knn.rs @@ -1,7 +1,7 @@ //! The backend sweep and the NN-Descent audit over neighbour constructions. //! -//! Each command reads the root's active generation and prints its readings. The grid arguments -//! default to the settings the deployment pinned, so an invocation without flags re-derives the +//! Each command reads the root's active generation and returns its readings. The grid arguments +//! default to the settings the deployment pinned. An invocation without flags re-derives the //! calibration evidence rather than an arbitrary sample of it. use clap::Args; @@ -20,12 +20,17 @@ pub(crate) struct BackendArgs { #[command(flatten)] root: crate::cli::RootArgs, - /// Fit seeds whose build and sample streams the sweep replays. A repeated seed measures build - /// nondeterminism. + /// Fit seeds whose build and sample streams the sweep replays. + /// + /// A repeated seed measures build nondeterminism. + /// + /// Defaults to [`backend::DEFAULT_SEEDS`], the pinned calibration grid. #[arg(long = "seed", value_delimiter = ',', default_values_t = backend::DEFAULT_SEEDS.to_vec())] seeds: Vec, /// `ef_construction` values; one index build per (seed, value). + /// + /// Defaults to [`backend::DEFAULT_CONSTRUCTIONS`]. #[arg( long = "ef-construction", value_delimiter = ',', @@ -34,6 +39,8 @@ pub(crate) struct BackendArgs { constructions: Vec, /// `ef_search` values, swept per built index. + /// + /// Defaults to [`backend::DEFAULT_SEARCHES`]. #[arg( long = "ef-search", value_delimiter = ',', @@ -67,12 +74,17 @@ pub(crate) struct DescentArgs { #[command(flatten)] root: crate::cli::RootArgs, - /// Fit seeds whose `knn-link` streams the audit replays. A repeated seed measures construction - /// nondeterminism. + /// Fit seeds whose `knn-link` streams the audit replays. + /// + /// A repeated seed measures construction nondeterminism. + /// + /// Defaults to [`descent::DEFAULT_SEEDS`], the pinned calibration grid. #[arg(long = "seed", value_delimiter = ',', default_values_t = descent::DEFAULT_SEEDS.to_vec())] seeds: Vec, /// Candidate caps the construction runs at. + /// + /// Defaults to [`descent::DEFAULT_CANDIDATES`]. #[arg( long = "candidates", value_delimiter = ',', diff --git a/libs/@local/graph/atlas/src/cli/report/ladder.rs b/libs/@local/graph/atlas/src/cli/report/ladder.rs index 27605ad3585..55ca6f836b7 100644 --- a/libs/@local/graph/atlas/src/cli/report/ladder.rs +++ b/libs/@local/graph/atlas/src/cli/report/ladder.rs @@ -19,6 +19,8 @@ pub(crate) struct LadderArgs { generation: GenerationId, /// Where the report bundle JSON lands. + /// + /// Defaults to `ladder-report.json` in the working directory. #[arg(long, default_value = "ladder-report.json", value_hint = ValueHint::FilePath)] output: Utf8PathBuf, } @@ -128,7 +130,7 @@ fn percent(part: usize, whole: usize) -> f64 { if whole == 0 { return 0.0; } - // Counts sit far below 2⁵³, so the quotient is exact enough for display. + // Counts sit far below 2⁵³. The quotient is exact enough for display. #[expect( clippy::cast_precision_loss, reason = "display quotient of small counts" diff --git a/libs/@local/graph/atlas/src/cli/report/mod.rs b/libs/@local/graph/atlas/src/cli/report/mod.rs index 94aead1e44c..ad01ed9cd32 100644 --- a/libs/@local/graph/atlas/src/cli/report/mod.rs +++ b/libs/@local/graph/atlas/src/cli/report/mod.rs @@ -1,8 +1,10 @@ -//! Analysis instruments over published generations, one submodule per report. +//! Analyses over published generations, one submodule per report. //! -//! Every instrument reads artifacts a fit already published and returns its readings. The host -//! renders them. The certified refit and the live assessment also write their evidence record, -//! because a bundle outlives the terminal that shows it. +//! An analysis reads artifacts a fit already published and returns its readings for the host to +//! render. The probe stands outside both halves of that: it solves a corpus a published generation +//! or supplied artifacts carry, and prints each solve's record as it goes. The certified refit and +//! the live assessment also write their evidence record, because a bundle outlives the terminal +//! that shows it. use core::{ error::Error, @@ -95,8 +97,8 @@ impl Display for ReportError { match self { Self::Io(_) => fmt.write_str("the report bundle could not be written"), Self::Connect(_) => fmt.write_str("the store connection could not be dialed"), - // Each instrument's own chain names the step that failed; - // this level adds no step of its own. + // Each analysis's own chain names the step that failed. This level adds no step of its + // own. Self::Assess(error) => Display::fmt(error, fmt), Self::Clumps(error) => Display::fmt(error, fmt), Self::KnnBackend(error) => Display::fmt(error, fmt), @@ -120,11 +122,12 @@ impl Error for ReportError { } } -/// The report subcommands, one per instrument. +/// The report subcommands, one per analysis. #[derive(Debug, clap::Subcommand)] pub(crate) enum ReportCommand { - /// Refits a published generation's classifier from its staged corpus and certifies the bytes - /// against the deployed artifact, then writes the report bundle. + /// Refits a published generation's classifier and writes the report bundle. + /// + /// The refit reads the staged corpus and certifies the bytes against the deployed artifact. Classifier(ClassifierArgs), /// Reads the clump grouping's shape at every candidate ε over a published k-NN table. @@ -136,14 +139,16 @@ pub(crate) enum ReportCommand { /// Audits NN-Descent neighbour constructions over the active generation. KnnDescent(DescentArgs), - /// Reads the relation effect of the condition ladder in world units over a published - /// generation - endpoint-distance contraction of the engaged pairs against the zero-condition - /// step - and writes the report bundle. + /// Reads the condition ladder's relation effect in world units over a published generation. + /// + /// The effect is the endpoint-distance contraction of the engaged pairs against the + /// zero-condition step. The run writes the report bundle. Ladder(LadderArgs), - /// Solves one fold subset from a frozen classifier corpus - a published generation's or - /// supplied artifacts' - and dumps every receipt; a budget-refused solve additionally traces - /// its stalling inner recurrence. + /// Solves one fold subset from a frozen classifier corpus and dumps every solve record. + /// + /// The corpus is a published generation's or supplied artifacts'. A budget-refused solve + /// additionally traces its stalling inner recurrence. Probe(ProbeArgs), /// Assesses the active generation's map fidelity over the live store and writes the report. @@ -151,20 +156,20 @@ pub(crate) enum ReportCommand { /// Certifies a generation's recorded target readings against the padded pass's realization. /// - /// No published generation records target evidence, so every invocation resolves its - /// generation and refuses. + /// No published generation records target evidence. Every invocation resolves its generation + /// and refuses. Realization(RealizationArgs), } impl ReportCommand { /// Runs the selected report. /// - /// The probe dumps its receipts as it solves, so it is the one instrument whose product is not - /// a verdict. + /// The probe dumps its records as it solves and answers `Ok(None)`, the one analysis that + /// returns no verdict. Realization answers with its refusal instead. /// /// # Errors /// - /// Returns a [`ReportError`] when the instrument fails or the process cannot write its record. + /// Returns a [`ReportError`] when the analysis fails or the process cannot write its record. pub(crate) async fn run(self) -> Result, ReportError> { match self { Self::Classifier(args) => args.run().await.map(ReportVerdict::Classifier).map(Some), diff --git a/libs/@local/graph/atlas/src/cli/report/probe.rs b/libs/@local/graph/atlas/src/cli/report/probe.rs index 78359e96c3d..dfff54d81d8 100644 --- a/libs/@local/graph/atlas/src/cli/report/probe.rs +++ b/libs/@local/graph/atlas/src/cli/report/probe.rs @@ -1,4 +1,4 @@ -//! One receipt-dumping solve of a fold subset from a frozen corpus. +//! One record-dumping solve of a fold subset from a frozen corpus. use camino::Utf8PathBuf; use clap::{Args, ValueHint}; @@ -29,15 +29,21 @@ pub(crate) struct ProbeArgs { #[arg(long, requires = "root")] generation: Option, - /// Directory of supplied annotation artifacts under their staged names - /// (annotation-corpus.json, annotation-embeddings.arr, annotation-hashes.arr), probed under - /// the compiled deployment defaults; the corpus of a fit that never published probes through - /// this form. + /// Directory of supplied annotation artifacts under their staged names. + /// + /// The staged names are annotation-corpus.json, annotation-embeddings.arr and + /// annotation-hashes.arr, probed under the compiled deployment defaults. The corpus of a fit + /// that never published probes through this form. #[arg(long, value_hint = ValueHint::DirPath)] inputs: Option, - /// The fold-assignment seed; the configured seed probes the production assignment, any other - /// seed probes an alternative. + /// The fold-assignment seed. + /// + /// The configured seed probes the production assignment, and any other seed probes an + /// alternative. + /// + /// Defaults to `0`. This flag reads no environment variable. A deployment whose seed is not + /// zero passes it here to probe the production assignment. #[arg(long, default_value_t = 0)] seed: u64, @@ -55,8 +61,9 @@ pub(crate) struct ProbeArgs { } impl ProbeArgs { - /// Reconstructs the frozen corpus and solves the fold subset solo, then dumps every receipt - /// with its curvature censuses. + /// Reconstructs the frozen corpus and solves the fold subset solo. + /// + /// The run then dumps every record with its curvature censuses. /// /// # Panics /// diff --git a/libs/@local/graph/atlas/src/cli/report/quality.rs b/libs/@local/graph/atlas/src/cli/report/quality.rs index e6d459562a8..d3ef9e1b5a7 100644 --- a/libs/@local/graph/atlas/src/cli/report/quality.rs +++ b/libs/@local/graph/atlas/src/cli/report/quality.rs @@ -20,19 +20,29 @@ pub(crate) struct QualityArgs { #[command(flatten)] store: crate::cli::PostgresArgs, - /// The probe seed. Equal seeds replay the sampling. + /// The probe seed, which its sampling derives from. + /// + /// Defaults to the seed [`live::Options`] carries, a compiled-in value rather than the fit's: + /// the assessment builds its generator from this option and reads no seed from the + /// generation's metadata. #[arg(long, default_value_t = live::Options::default().seed)] seed: u64, /// Sampled anchor rows. + /// + /// Defaults to the anchor count [`live::Options`] carries. #[arg(long, default_value_t = live::Options::default().anchors)] anchors: NonZero, /// Sampled comparison rows. + /// + /// Defaults to the comparison count [`live::Options`] carries. #[arg(long, default_value_t = live::Options::default().comparisons)] comparisons: NonZero, /// Where the report JSON lands. + /// + /// Defaults to `quality-report.json` in the working directory. #[arg(long, default_value = "quality-report.json", value_hint = ValueHint::FilePath)] output: Utf8PathBuf, } diff --git a/libs/@local/graph/atlas/src/cli/shell.rs b/libs/@local/graph/atlas/src/cli/shell.rs index 9ee22866c8d..49736e2192b 100644 --- a/libs/@local/graph/atlas/src/cli/shell.rs +++ b/libs/@local/graph/atlas/src/cli/shell.rs @@ -30,7 +30,7 @@ enum Command { #[command(flatten)] store: PostgresArgs, - // The fit flags dwarf the other variants, so the box keeps the enum small. + // The fit flags dwarf the other variants. The box keeps the enum small. #[command(flatten)] args: Box, @@ -45,8 +45,8 @@ enum Command { /// Fit from the dump directory instead of the live store. /// - /// The dump supplies the snapshot and every embedding, so the run reaches neither the - /// store nor the embedding provider, and the store flags and the provider key are read by + /// The dump supplies the snapshot and every embedding. The run reaches neither the store + /// nor the embedding provider, and the store flags and the provider key are read by /// nothing. The generation's metadata records the dump as the fit's source. #[arg(long, value_name = "DUMP", value_hint = ValueHint::DirPath)] offline: Option, @@ -60,7 +60,7 @@ enum Command { tui: bool, }, - /// Compiles an analysis instrument over a published generation. + /// Runs one analysis over a published generation. Report { #[command(subcommand)] command: ReportCommand, @@ -91,6 +91,10 @@ enum FitSource { } /// Resolves the fit flags into the run's source. +/// +/// # Panics +/// +/// Panics if both the offline directory and provider key are absent. #[cfg(feature = "cli")] fn fit_source( store: PostgresArgs, @@ -109,8 +113,7 @@ fn fit_source( /// One dashboard-hosted fit's failure, by step. /// -/// The dashboard path owns three failures the logged path does not have to distinguish, and an -/// operator reading a restored terminal needs to know which one they hit. +/// The restored terminal reports the step that failed. #[cfg(feature = "cli")] #[derive(Debug)] enum DashboardError { @@ -128,7 +131,7 @@ impl core::fmt::Display for DashboardError { match self { Self::Terminal(_) => fmt.write_str("the live dashboard could not use the terminal"), Self::Connect(_) => fmt.write_str("the store connection could not be dialed"), - // The fit's own chain is the diagnosis; this variant adds no + // The fit's own chain is the diagnosis. This variant adds no // step of its own. Self::Fit(error) => core::fmt::Display::fmt(error, fmt), } @@ -184,8 +187,12 @@ fn log_filter() -> tracing_subscriber::EnvFilter { /// /// # Errors /// -/// Returns the step that failed - terminal, connection, or the fit itself. The run's failure wins -/// over a terminal failure, since it is the one an operator is trying to read. +/// Returns the step that failed: terminal, connection or fit. A run failure takes +/// precedence over a terminal-restoration failure. +/// +/// # Panics +/// +/// Panics if a global tracing subscriber is already installed. #[cfg(feature = "cli")] async fn fit_on_dashboard( root: RootArgs, @@ -194,8 +201,8 @@ async fn fit_on_dashboard( ) -> Result { let dashboard = super::tui::Dashboard::start().map_err(DashboardError::Terminal)?; - // The dashboard owns the terminal from here, so the records the run - // emits belong in its pane, not on the screen it is drawing. + // The dashboard owns the terminal from here. The records the run emits belong in its pane, not + // on the screen it is drawing. tracing_subscriber::fmt() .with_env_filter(log_filter()) .with_writer(dashboard.log_sink()) @@ -241,7 +248,8 @@ async fn fit_on_dashboard( /// # Panics /// /// This panics when the tokio runtime cannot start or a global log subscriber is already -/// installed. +/// installed. After parsing, [`verify_cpu_baseline`](crate::math::kernel::verify_cpu_baseline) +/// rejects a CPU below the compiled baseline, on the conditions it documents. #[cfg(feature = "cli")] #[must_use] #[tokio::main] @@ -308,7 +316,7 @@ pub async fn main() -> std::process::ExitCode { } Command::Report { command } => match command.run().await { - // The probe dumps its receipts as it solves and hands back no + // The probe dumps its records as it solves and hands back no // verdict to render. Ok(None) => std::process::ExitCode::SUCCESS, Ok(Some(verdict)) => { @@ -342,20 +350,24 @@ mod tests { use super::{Cli, Command}; - /// A scratch generation root for one parse, keyed so libtest's shared process cannot collide. + /// Returns a temporary path keyed by the process and test name. + /// + /// # Panics /// - /// Parsing creates the root directory, so each test names its own and removes it afterwards. + /// Panics if the temporary directory path is not UTF-8. fn scratch_root(name: &str) -> Utf8PathBuf { Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") .join(format!("atlas-shell-{}-{name}", std::process::id())) } + /// The shell's command tree satisfies clap's own structural requirements. #[test] fn cli_consistency() { ::command().debug_assert(); } + /// A fit reading an offline dump parses, so long as it names no live-store flag. #[test] fn offline_without_live_flags() { let root = scratch_root("offline_without_live_flags"); diff --git a/libs/@local/graph/atlas/src/cli/tui/mod.rs b/libs/@local/graph/atlas/src/cli/tui/mod.rs index 3fa640291f8..9bba459813e 100644 --- a/libs/@local/graph/atlas/src/cli/tui/mod.rs +++ b/libs/@local/graph/atlas/src/cli/tui/mod.rs @@ -7,11 +7,11 @@ //! //! Reporting is one-way traffic. Observations and log lines travel down a channel as //! [`Observation`]s, and the renderer owns the only [`RunState`], folding what has arrived into it -//! before each frame. A reporting thread parts with its observation and carries on, so a hot loop -//! never waits on the terminal. +//! before each frame. A reporting thread parts with its observation and carries on, and the hot +//! loop never waits on the terminal. //! -//! The dashboard observes and never steers, so nothing here can change what a run publishes. The -//! channel takes every observation as it comes. A closed channel means the dashboard has already +//! The dashboard observes and never steers: what a run publishes is the run's alone. The channel +//! takes every observation as it comes. A closed channel means the dashboard has already //! finished. No observation can fail a fit. //! //! One deliberate exception to that, because raw mode swallows the interrupt: `q` and `Ctrl-C` @@ -53,7 +53,7 @@ use crate::{ /// How long the renderer waits for a key before drawing the next frame. /// /// The spinner's cadence and the terminal's responsiveness are the same number: a keypress -/// short-circuits the wait, so the dashboard reacts at once and idles at ten frames a second. +/// short-circuits the wait. The dashboard reacts at once and idles at ten frames a second. const TICK: Duration = Duration::from_millis(100); /// The exit code of an interrupted run, as a shell reports `SIGINT`. @@ -61,7 +61,7 @@ const INTERRUPTED: u8 = 130; /// Placement rows the dashboard asks the run to sample for its map. /// -/// Exactly what the widest map can hold apart, so the appetite is the frame's own resolution rather +/// Exactly what the widest map can hold apart. The appetite is the frame's own resolution rather /// than a number chosen to feel large enough. const SNAPSHOT_ROWS: usize = render::MAP_CAPACITY; @@ -76,7 +76,7 @@ pub(super) struct Dashboard { /// Raised to bring the rendering thread home. /// /// A sending half outlives every run, since the shell installs the log subscriber globally and - /// never drops it, so this flag is what ends the loop. + /// never drops it. This flag is what ends the loop. stop: Arc, /// The rendering thread, which owns the terminal and restores it as it leaves. renderer: JoinHandle>, @@ -88,7 +88,7 @@ impl Dashboard { /// # Errors /// /// Returns an [`io::Error`] when this cannot take the terminal or cannot spawn the rendering - /// thread. Both happen before the run begins, so a failure here costs nothing. + /// thread. Both happen before the run begins. A failure here costs nothing. pub(super) fn start() -> io::Result { let terminal = ratatui::try_init()?; let (observations, arrived) = mpsc::channel(); @@ -133,7 +133,7 @@ impl Dashboard { self.stop.store(true, Ordering::Release); self.renderer.join().unwrap_or_else(|_panicked| { - // The hook `ratatui::try_init` installed has already restored the terminal; restoring + // The hook `ratatui::try_init` installed has already restored the terminal. Restoring // twice costs nothing and guarantees the shell prints onto a sane screen. ratatui::restore(); Ok(()) @@ -159,7 +159,6 @@ impl Observer { } impl Progress for Observer { - /// A detached half reports into the same dashboard through the same channel. type Detached = Self; fn detach(&self) -> Self { @@ -182,6 +181,8 @@ impl Progress for Observer { self.report(Observation::ClassifierStarted(folds)); } + // The model counts folds done rather than which fold it was, and this observer drops the index + // before sending the fieldless observation. fn classifier_fold_completed(&self, _fold: usize) { self.report(Observation::ClassifierFoldCompleted); } @@ -218,10 +219,17 @@ impl Progress for Observer { }); } + /// Reports the number of positions the dashboard wants sampled. + /// + /// This is the widest map's own capacity, which is what turns snapshot gathering on at all. fn projector_sample_size(&self) -> usize { SNAPSHOT_ROWS } + /// Reports the sampled positions to the renderer. + /// + /// The observer copies the positions out of the run's buffer because the observation outlives + /// the call, and carries the landmark prefix length alongside. fn projector_snapshot(&self, positions: &[Vec2], landmarks: usize) { self.report(Observation::ProjectorSnapshot { positions: positions.to_vec(), @@ -274,6 +282,10 @@ impl io::Write for LogWriter { Ok(buf.len()) } + /// Sends every whole line buffered so far to the renderer. + /// + /// A trailing partial line remains buffered for the next write. Here a closed channel is not + /// an error, because a log line may not fail a fit. fn flush(&mut self) -> io::Result<()> { // The subscriber writes one record as a sequence of calls and ends it with a newline. Only // whole lines become rows of the pane. @@ -288,14 +300,21 @@ impl io::Write for LogWriter { } impl Drop for LogWriter { + /// Flushes the record's last line when the writer is dropped. + /// + /// The subscriber drops the writer at the end of every record, which is what makes the record + /// appear in the pane. fn drop(&mut self) { - // The subscriber drops the writer at the end of every record, - // which is what makes the record appear. drop(io::Write::flush(self)); } } /// Draws one frame of the model as it currently stands. +/// +/// # Errors +/// +/// Returns an [`io::Error`] when the terminal refuses the frame. The renderer gives up on the +/// terminal at that point rather than redrawing into it. fn draw(terminal: &mut DefaultTerminal, state: &RunState, tick: usize) -> io::Result<()> { terminal.draw(|frame| render::frame(frame, state, tick))?; @@ -337,6 +356,11 @@ fn interrupted(event: &Event) -> bool { /// /// The model lives here, on the thread that draws it: each frame folds in what has arrived since /// the last one. +/// +/// # Errors +/// +/// Returns an [`io::Error`] when a frame, a key poll or the terminal's restoration fails. The +/// dashboard's owner reports it after the run, since the fit itself is unaffected. fn render( mut terminal: DefaultTerminal, arrived: &Receiver, @@ -420,7 +444,7 @@ mod tests { observer.classifier_fold_completed(0); observer.classifier_fold_completed(1); - // The model drops a fold that arrives before its announcement, so the counter reads the + // The model drops a fold that arrives before its announcement. The counter reads the // arrival order rather than the set. let folds = absorbed(&arrived) .classifier() @@ -429,6 +453,9 @@ mod tests { assert_eq!(folds.done, 2); } + /// Quality readings followed by the stage's completion all reach the model in one fold. + /// + /// The readings outlive the stage that reported them. #[test] fn the_admission_batterys_readings_reach_the_state_the_renderer_draws() { let (observations, arrived) = channel(); @@ -459,10 +486,13 @@ mod tests { let observer = Observer { observations }; drop(arrived); - // The renderer has left; the run has not, and an observation may not fail it. + // The renderer has left. The run has not, and an observation may not fail it. observer.stage_completed(Stage::Seal); } + /// The dashboard's sample appetite is greater than the trait's silent default. + /// + /// The snapshot returned for it reaches the model with its rows and landmark prefix intact. #[test] fn the_dashboard_asks_for_a_sample_and_draws_what_comes_back() { let (observations, arrived) = channel(); @@ -481,6 +511,9 @@ mod tests { assert_eq!(placement.landmarks, 1); } + /// A record emitted through a subscriber writing into the sink becomes exactly one pane line. + /// + /// The line carries its level, its message and its fields. #[test] fn log_records_become_pane_lines() { let (observations, arrived) = channel(); @@ -490,8 +523,8 @@ mod tests { .without_time() .finish(); - // The dispatcher is thread-local here on purpose, so the test exercises the writer rather - // than the shell's global installation. + // The dispatcher is thread-local here on purpose. The test exercises the writer rather than + // the shell's global installation. tracing::subscriber::with_default(subscriber, || { tracing::info!(rows = 49, "staged the annotation corpus"); }); diff --git a/libs/@local/graph/atlas/src/cli/tui/render/loss.rs b/libs/@local/graph/atlas/src/cli/tui/render/loss.rs index cbb4efe250e..f445f1b3a57 100644 --- a/libs/@local/graph/atlas/src/cli/tui/render/loss.rs +++ b/libs/@local/graph/atlas/src/cli/tui/render/loss.rs @@ -1,8 +1,8 @@ -//! The placement's composite objective against the schedule it is descending. +//! The placement's composite objective against its training steps. //! -//! The chart is the shape of the descent rather than a table of it - braille resolution, no step -//! labels, and the exact current value on the frame's own title. The chart reads its axes off the -//! same points it plots, so the frame cannot claim a range the curve does not occupy. +//! Braille resolution shows the shape of the retained loss curve without step labels. The frame +//! title gives the current total at four decimal places. The value axis begins at zero and ends at +//! the greatest positive finite retained loss, or at one when none exists. use ratatui::{ Frame, @@ -18,13 +18,13 @@ use crate::cli::tui::state::ProjectorTraining; /// Draws the placement's descent: the composite objective against the schedule's step axis. /// -/// The chart draws the curve at braille resolution: two steps per column of the plotting area, -/// which is the pane inside its border and padding, less the gutter of the value labels. A schedule -/// is normally longer than that, so the chart is the shape of the descent, and the frame's title -/// shows the exact current value. +/// The chart draws the curve at braille resolution: two horizontal dot positions per character +/// column of the plotting area, inside the border and padding and beside the value labels. A +/// schedule can contain more steps than the plot has horizontal dot positions. The curve shows the +/// loss trend, while the frame title reports the current total at four decimal places. pub(super) fn render_loss(frame: &mut Frame, area: Rect, training: &ProjectorTraining) { - // `Dataset::data` borrows a slice, so this builds the curve once per frame and reads it twice: - // for the plot, and for its value axis. + // `Dataset::data` borrows a slice. This builds the curve once per frame and reads it twice: for + // the plot, and for its value axis. let points: Vec<(f64, f64)> = curve(training).into_iter().collect(); let [low, high] = value_bounds(points.iter().map(|&(_, loss)| loss)); let [first, last] = step_bounds(training); @@ -58,7 +58,7 @@ pub(super) fn render_loss(frame: &mut Frame, area: Rect, training: &ProjectorTra frame.render_widget(chart, area); } -/// The retained losses as the chart's own coordinates. +/// Returns the retained losses as the chart's own coordinates. /// /// A point is `(step, loss)` in the widget's coordinate type. The step axis counts offsets into the /// retained window, which the axis draws unlabelled. The loss is the `f32` the run reported, @@ -69,12 +69,10 @@ pub(super) fn curve(training: &ProjectorTraining) -> impl IntoIterator) -> [f64; 2] { let high = values .into_iter() @@ -84,10 +82,10 @@ pub(super) fn value_bounds(values: impl IntoIterator) -> [f64; 2] { if high > 0.0 { [0.0, high] } else { [0.0, 1.0] } } -/// The chart's step axis, which spans the retained window and is never narrower than one step. +/// Returns the chart's step axis, spanning the retained window and never narrower than one step. /// -/// The right edge is the last point's own coordinate, so the axis and the curve cannot disagree -/// about where the descent ends. +/// For two or more retained points, the right edge is the last point's coordinate. An empty or +/// single-point window uses the unit interval `[0, 1]`. pub(super) fn step_bounds(training: &ProjectorTraining) -> [f64; 2] { let last = curve(training) .into_iter() @@ -98,10 +96,16 @@ pub(super) fn step_bounds(training: &ProjectorTraining) -> [f64; 2] { [0.0, last.max(1.0)] } -/// The last step's objective, family by family, for the chart's footer. +/// Formats the last step's objective by family for the chart's footer. /// -/// The footer is all or nothing. The widget draws a title wider than its frame over the corner, so -/// a pane too narrow for the whole breakdown shows the plot and its total alone. +/// Semantic attraction, ordinary repulsion, mined hard-negative repulsion and relation attraction +/// read as themselves. The footer adds the temporal anchors to the landmarks under one support +/// heading, the term they are both evaluations of. It shows those five at three decimal places, +/// while the title shows the total at four. The target objective is not among the five, and the +/// title's total covers it along with them at that different precision. +/// +/// The footer is all or nothing. The widget draws a title wider than its frame over the corner. A +/// pane too narrow for the whole breakdown shows the plot and its total alone. fn breakdown(training: &ProjectorTraining, width: u16) -> String { let loss = training.last; diff --git a/libs/@local/graph/atlas/src/cli/tui/render/map.rs b/libs/@local/graph/atlas/src/cli/tui/render/map.rs index e17d6949f75..cf44e0cc601 100644 --- a/libs/@local/graph/atlas/src/cli/tui/render/map.rs +++ b/libs/@local/graph/atlas/src/cli/tui/render/map.rs @@ -1,8 +1,8 @@ //! The sampled rows as braille dots, with the landmark skeleton picked out. //! -//! The map draws the sample the observer asked for, not the corpus, and it keeps the placement's -//! own shape: equal data units per dot on both axes, so the atlas is never stretched to fill the -//! frame. The map draws nothing it cannot place. +//! The map draws sampled positions with a viewport fitted to the dot grid. Both axes use +//! approximately equal data units per dot, subject to the corner rounding of +//! [`Bounds2::with_aspect_ratio`]. An unrepresentable viewport leaves the frame empty. use ratatui::{ Frame, @@ -27,7 +27,7 @@ pub(super) const SKELETON: Color = Color::Magenta; /// How much wider than the placement the map draws its viewport. /// -/// The map carries no axis labels, so widening states nothing untrue - it only keeps the outermost +/// The map carries no axis labels. Widening states nothing untrue - it only keeps the outermost /// rows a dot inside the frame rather than against its wall. const MAP_MARGIN: Positive = Positive::new(1.04).unwrap(); diff --git a/libs/@local/graph/atlas/src/cli/tui/render/mod.rs b/libs/@local/graph/atlas/src/cli/tui/render/mod.rs index 4db05ab2283..377ddd84c8b 100644 --- a/libs/@local/graph/atlas/src/cli/tui/render/mod.rs +++ b/libs/@local/graph/atlas/src/cli/tui/render/mod.rs @@ -1,10 +1,11 @@ -//! Drawing one frame of the dashboard: the stage rail, the loss chart beside the placement map, and -//! the log pane. +//! Drawing one frame of the dashboard. //! -//! [`frame`] is a pure function of the model plus the tick that animates the running stage, so a -//! frame is reproducible: the same [`RunState`] and tick draw the same cells. Every glyph choice, -//! label, and color lives in this module tree - the model carries no presentation, and the -//! pipeline's [`Stage`] carries no prose. +//! The frame holds the stage rail, the loss chart beside the placement map, and the log pane. +//! +//! [`frame`] is a pure function of the model plus the tick that animates the running stage. A frame +//! is reproducible: the same [`RunState`] and tick draw the same cells. Every glyph choice, label, +//! and color lives in this module tree - the model carries no presentation, and the pipeline's +//! [`Stage`] carries no prose. //! //! One module per pane - [`rail`], [`loss`], [`map`], [`log`] - each owning the vocabulary it draws //! with. This module owns the composition: the geometry it lays the panes out in, and which of them @@ -58,8 +59,9 @@ const DOTS_ACROSS: u16 = 2; /// Dots one braille cell is tall. const DOTS_DOWN: u16 = 4; -/// Points the widest map can hold apart: its braille dot grid, two dots per column and four per row -/// inside the frame. +/// Points the widest map can hold apart. +/// +/// This is its braille dot grid, two dots per column and four per row inside the frame. /// /// This is what the dashboard asks a run to sample. A larger sample would cost the run copies of /// coordinates the map draws into dots already lit. @@ -103,10 +105,12 @@ pub(super) fn frame(frame: &mut Frame, state: &RunState, tick: usize) { render_log(frame, log, state); } -/// The rows the stage rail claims, one per pipeline stage plus one per admission reading the probe -/// has reported. +/// Returns the rows the stage rail claims. +/// +/// The rail claims one row per pipeline stage plus one per admission reading the probe has +/// reported. /// -/// The rail earns the readings whole or not at all. Half a battery under the admission stage would +/// The rail draws the readings whole or not at all. Half a battery under the admission stage would /// read as evidence the probe could not measure. Only the report says that. A terminal too short /// for a row per reading therefore draws no readings at all, and the numbers stay in the report. fn rail_height(state: &RunState, available: u16) -> u16 { @@ -127,13 +131,12 @@ fn rail_height(state: &RunState, available: u16) -> u16 { RAIL_HEIGHT + readings } -/// The rows the placement's band may claim, beneath a rail of `rail` rows. +/// Returns the rows the placement's band may claim, beneath a rail of `rail` rows. /// /// The band takes rows only once the placement has something to show, and it never crowds out the -/// rail or the log. The rail is the run's shape and the log is its voice, so the band takes what -/// those two leave and stays away entirely below the height where a plot says anything. Readings -/// arriving at the end of a run therefore cost the band its rows before they cost the log any. By -/// then the placement's curve has told its story, and the battery's numbers have not. +/// rail or the log. It takes what those two leave, and stays away entirely below the height where a +/// plot says anything. Readings arriving at the end of a run therefore cost the band its rows +/// before they cost the log any. const fn band_height(state: &RunState, available: u16, rail: u16) -> u16 { if state.projector().is_none() && state.placement().is_none() { return 0; @@ -147,10 +150,10 @@ const fn band_height(state: &RunState, available: u16, rail: u16) -> u16 { spare.min(LOSS_HEIGHT) } -/// The columns the map takes out of the band, beside the curve. +/// Returns the columns the map takes out of the band, beside the curve. /// -/// The curve is the reading an operator acts on, so it keeps its width first and the map takes the -/// remainder; below the width where dots resolve anything the map stays away entirely, and past the +/// The curve is the reading an operator acts on. It keeps its width first and the map takes the +/// remainder. Below the width where dots resolve anything the map stays away entirely, and past the /// width where it stops gaining detail the curve takes the rest of the growth. const fn map_width(state: &RunState, available: u16) -> u16 { if state.placement().is_none() { diff --git a/libs/@local/graph/atlas/src/cli/tui/render/rail.rs b/libs/@local/graph/atlas/src/cli/tui/render/rail.rs index 1436bc2cc09..25f64a3bffa 100644 --- a/libs/@local/graph/atlas/src/cli/tui/render/rail.rs +++ b/libs/@local/graph/atlas/src/cli/tui/render/rail.rs @@ -1,14 +1,15 @@ //! One row per pipeline stage, with the run's clock and its completion bar. //! //! The rail is the run's whole shape from the first frame - it lists every stage before any of them -//! has happened, so the pane reads as remaining work rather than as a growing log. A running stage -//! carries the counter of whatever it is counting. A finished one trades that counter for its span. +//! has happened. The pane reads as remaining work rather than as a growing log. A running stage +//! carries the counter of whatever it is counting, and a finished one trades that counter for its +//! span. //! //! The admission probe's readings hang under the rail as its one detail block. A running stage's //! counter goes away the moment that stage finishes, while the battery's readings are the numbers -//! the run answers for, so they stay for the frames that follow, the last frame of the run -//! included. Posterity is still the report the run writes. These rows are the operator's live copy -//! of the numbers behind its verdict. +//! the run answers for. They stay for the frames that follow, the last frame of the run included. +//! Posterity is still the report the run writes. These rows are the operator's live copy of the +//! numbers behind its verdict. #![expect( clippy::non_ascii_literal, reason = "the dashboard's glyphs are its rendering vocabulary" @@ -36,8 +37,9 @@ use crate::{ /// Frames of the running stage's spinner, in braille. const SPINNER: [&str; 10] = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"]; -/// The rail's label column, wide enough for the widest stage label plus the space that separates it -/// from what follows. +/// The rail's label column. +/// +/// It is wide enough for the widest stage label plus the space that separates it from what follows. const LABEL_WIDTH: usize = { let mut widest = 0; let mut index = 0; @@ -52,7 +54,7 @@ const LABEL_WIDTH: usize = { widest + 1 }; -/// Cells of the rail's completion bar: one per stage, so the bar needs no scaling. +/// Cells of the rail's completion bar: one per stage. The bar needs no scaling. const BAR_WIDTH: usize = Stage::ALL.len(); /// Cells of a stage counter's bar, narrow enough to leave the row its leader dots. @@ -109,9 +111,9 @@ pub(super) fn render_rail( } // The readings the admission probe reported, under the stage that reported them. The probe - // reports the whole battery in one burst as its report reduces the steps, so the rows arrive - // together. The composition decides whether the rail has the room for them, and a frame - // shorter than the rail asked for drops the lines it cannot hold from the bottom. + // reports the whole battery in one burst as its report reduces the steps. The rows arrive + // together. The composition decides whether the rail has the room for them, and a frame shorter + // than the rail asked for drops the lines it cannot hold from the bottom. rows.extend( state .quality() @@ -148,8 +150,8 @@ fn stage_row<'row>( span: &str, width: usize, ) -> Line<'row> { - // A counter takes the room it needs plus its leading space; the - // glyphs are single-width, so counting characters counts columns. + // A counter takes the room it needs plus its leading space. The glyphs are single-width. + // Counting characters counts columns. let (counter, counter_width) = counter.map_or((None, 0), |text| { ( Some(Span::from(format!(" {text}")).fg(ACCENT)), @@ -200,9 +202,11 @@ fn counter(run: &RunState, stage: Stage) -> Option { /// The neighbour-table counter, showing whichever part of the construction reported last. /// -/// The stage runs a batched loop, then a phase the backend names, then the descent's convergence -/// readings, then a loop again, then its verdict, so the row carries whichever of those the -/// construction is inside - one counter for every part the stage reports. +/// The configuration picks one of two constructions, and each reports its own parts. An +/// index-backed construction fills the backend row by row, hands it a linking it names its own +/// phases through, then reads every row's list back out. NN-Descent fills no backend and reports +/// its iterations alone. The recall verdict closes either one. The row carries whichever part +/// reported last rather than a sequence every run passes through. fn knn_counter(activity: &KnnActivity) -> String { match activity { KnnActivity::Inserting(batch) => batch_counter(*batch, "inserted"), @@ -216,9 +220,9 @@ fn knn_counter(activity: &KnnActivity) -> String { } } -/// One batched loop's counter, showing its position as a bar and what the covered rows did. +/// Renders one batched loop's counter, its position as a bar beside what the covered rows did. /// -/// Both loops of a construction count rows to the same total, so each says which one it is. +/// Both loops of a construction count rows to the same total. Each says which one it is. fn batch_counter(batch: Batch, covered: &str) -> String { let Some(total) = NonZero::new(batch.total) else { return String::new(); @@ -244,7 +248,7 @@ fn projector_counter(training: &ProjectorTraining) -> String { ) } -/// The classifier counter, showing completed folds as a bar and the selected strength once chosen. +/// Renders the classifier counter, completed folds as a bar beside the strength once chosen. fn classifier_counter(folds: ClassifierFolds, regularization: Option) -> String { let Some(total) = NonZero::new(folds.total) else { return String::new(); @@ -258,8 +262,9 @@ fn classifier_counter(folds: ClassifierFolds, regularization: Option) -> St ) } -/// The card-embedding counter, showing either the provider's share as a bar or the reuse that -/// avoided it. +/// Renders the card-embedding counter. +/// +/// The counter shows either the provider's share as a bar or the reuse that avoided it. fn embedding_counter(workload: EmbeddingWorkload) -> String { let Some(embedded) = NonZero::new(workload.embedded) else { return format!("{} reused", workload.reused); @@ -272,7 +277,7 @@ fn embedding_counter(workload: EmbeddingWorkload) -> String { ) } -/// A fixed-width bar of one workload's completion. +/// Renders a fixed-width bar of one workload's completion. /// /// Cells of a text row rather than a [`ratatui::widgets::Gauge`], because a gauge owns a whole /// rectangle while one counter shares its row with the stage's name, its numbers and the leader @@ -293,11 +298,11 @@ fn counter_bar(done: usize, total: NonZero) -> String { ) } -/// The rail's completion bar, filled blocks over the stages still to come. +/// Renders the rail's completion bar, filled blocks over the stages still to come. /// /// A [`Line`], because the rail frame draws it as its own bottom title. fn progress_bar(completed: usize, total: usize) -> Line<'static> { - // One cell per stage, so the bar is the rail's own index rather than a rescaling of it. + // One cell per stage. The bar is the rail's own index rather than a rescaling of it. let filled = completed.min(BAR_WIDTH); let color = if completed == total { Color::Green diff --git a/libs/@local/graph/atlas/src/cli/tui/render/tests.rs b/libs/@local/graph/atlas/src/cli/tui/render/tests.rs index a8264ce7c88..f7360b853c6 100644 --- a/libs/@local/graph/atlas/src/cli/tui/render/tests.rs +++ b/libs/@local/graph/atlas/src/cli/tui/render/tests.rs @@ -21,7 +21,15 @@ use crate::{ }, }; -/// The drawn frame as one string per row, trailing blanks trimmed. +/// Renders the drawn frame as one string per row, trailing blanks trimmed. +/// +/// Reads coordinates from zero up to the buffer's width and height. Zero width produces an empty +/// string per row, and zero height produces no rows. +/// +/// # Panics +/// +/// Panics for a nonempty area whose origin is not `(0, 0)`, or whose cell storage does not cover +/// the requested coordinates. fn rows(buffer: &Buffer) -> Vec { (0..buffer.area.height) .map(|y| { @@ -43,7 +51,7 @@ fn draw_on(state: &RunState, tick: usize, width: u16, height: u16) -> Vec Buffer { let mut terminal = Terminal::new(TestBackend::new(width, height)).expect("should open a terminal"); @@ -54,7 +62,7 @@ fn buffer_on(state: &RunState, tick: usize, width: u16, height: u16) -> Buffer { terminal.backend().buffer().clone() } -/// A grid of interior rows around the origin, and the two landmarks that lead the sample. +/// Builds a grid of interior rows around the origin, led by the sample's two landmarks. fn placement() -> Vec { let mut positions = vec![Vec2::new(-2.0, 0.0), Vec2::new(2.0, 0.0)]; for column in 0..8_u8 { @@ -69,7 +77,7 @@ fn placement() -> Vec { positions } -/// A run standing in the placement stage with a snapshot reported. +/// Builds a run standing in the placement stage with a snapshot reported. fn placed() -> RunState { let mut state = training(150, 600); state.place_projector(placement(), 2); @@ -77,7 +85,7 @@ fn placed() -> RunState { state } -/// A run standing in the placement stage, `steps` of `total` reported. +/// Builds a run standing in the placement stage, `steps` of `total` reported. #[expect( clippy::cast_precision_loss, reason = "the fixture runs a few hundred steps, exactly representable" @@ -88,7 +96,7 @@ fn training(steps: usize, total: usize) -> RunState { state.complete_at(stage, Duration::from_secs(index as u64 + 1)); } for step in 0..steps { - // A decaying curve, so the chart has a real range to label. + // A decaying curve gives the chart a real range to label. let decay = 0.5_f32.powf(step as f32 / 16.0); state.advance_projector( step, @@ -142,6 +150,9 @@ fn a_completed_stage_carries_its_glyph_span_and_leader_dots() { assert!(drawn[13].contains("stages 1/12"), "{drawn:#?}"); } +/// A half-finished embedding batch lights half the ingest row's bar. +/// +/// The done-of-total counter prints between label and leader dots, on the running row alone. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -171,6 +182,9 @@ fn the_running_ingest_row_carries_the_embedding_counter() { assert!(!drawn[2].contains('░'), "{drawn:#?}"); } +/// An ingest that embedded nothing because the cache held everything reports the reused count. +/// +/// A bar in its place would read as no progress. #[test] fn a_wholly_reused_workload_says_so_instead_of_drawing_an_empty_bar() { let mut state = RunState::new(); @@ -184,6 +198,9 @@ fn a_wholly_reused_workload_says_so_instead_of_drawing_an_empty_bar() { assert!(drawn[1].contains("4096 reused"), "{drawn:#?}"); } +/// Once ingest completes, its row trades the batch counter for the span it took. +/// +/// The counter belongs to work in flight. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -207,6 +224,9 @@ fn a_completed_ingest_stage_drops_its_counter_for_its_span() { assert!(drawn[1].ends_with("12.4s │"), "{drawn:#?}"); } +/// Three of four folds done draws six of the classifier row's eight bar cells. +/// +/// The fold counter stands beside them. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -254,6 +274,9 @@ fn the_classifier_row_carries_the_derived_boundary_until_the_folds_start() { ); } +/// With every fold in and a strength selected, the classifier row carries every reading. +/// +/// The row shows the full bar, the fold counter and the chosen regularization strength. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -307,7 +330,7 @@ fn the_running_knn_row_carries_whichever_part_of_the_construction_reported() { ); // A phase the backend named replaces the bar: the linking counts - // nothing this side of the seam. + // nothing on this side of the handoff. state.report_knn(KnnActivity::Building("building the graph".to_owned())); assert!( draw(&state, 0)[6].starts_with("│ ⠋ knn building the graph"), @@ -343,6 +366,9 @@ fn the_running_knn_row_carries_whichever_part_of_the_construction_reported() { ); } +/// A quarter of the descent schedule reported lights two of the projector row's eight bar cells. +/// +/// The step counter prints beside them. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -351,8 +377,7 @@ fn the_running_knn_row_carries_whichever_part_of_the_construction_reported() { fn the_running_projector_row_carries_its_step_counter() { let drawn = draw(&training(150, 600), 0); - // A quarter of the schedule is behind it, so two of the eight - // cells are lit. + // A quarter of the schedule is behind it. Two of the eight cells are lit. assert!( drawn[9].starts_with("│ ⠋ projector ██░░░░░░ 150/600"), "{drawn:#?}" @@ -380,13 +405,16 @@ fn the_loss_chart_labels_the_floor_and_the_peak_the_run_reached() { assert!(chart.contains('⠉') || chart.contains('⣀'), "{chart}"); } +/// A chart pane too narrow for the per-family footer keeps the chart. +/// +/// The footer row keeps an unbroken border, rather than a title clipped mid-word. #[test] fn a_pane_too_narrow_for_the_breakdown_drops_it_rather_than_the_corner() { let drawn = draw_on(&training(150, 600), 0, 60, 30); assert!(drawn[14].contains(" loss "), "{drawn:#?}"); - // The widget draws a title wider than its frame from the left corner outward, so the footer row - // is unbroken border or it is a truncated sentence starting mid-word. Absence of the whole + // The widget draws a title wider than its frame from the left corner outward. The footer row is + // unbroken border or it is a truncated sentence starting mid-word. Absence of the whole // breakdown is not enough to tell those apart. assert!( drawn[22] @@ -425,8 +453,8 @@ fn the_map_draws_beside_the_curve_on_a_wide_pane() { fn a_pane_too_narrow_for_both_keeps_the_curve_alone() { let drawn = draw_on(&placed(), 0, 70, 30); - // The curve is the reading an operator acts on, so a narrow - // pane gives up the map rather than the descent. + // The curve is the reading an operator acts on. A narrow pane gives up the map rather than the + // descent. assert!(drawn[14].contains(" loss "), "{drawn:#?}"); assert!(!drawn[14].contains(" map "), "{drawn:#?}"); } @@ -437,8 +465,7 @@ fn a_run_with_no_snapshot_leaves_the_curve_the_whole_band() { assert!(drawn[14].contains(" loss "), "{drawn:#?}"); assert!(!drawn[14].contains(" map "), "{drawn:#?}"); - // The chart has the room the map would have taken, so its - // per-family footer fits. + // The chart has the room the map would have taken. Its per-family footer fits. assert!(drawn[22].contains("semantic 0.013"), "{drawn:#?}"); } @@ -452,7 +479,7 @@ fn a_landmark_colors_the_cell_it_lands_in() { .map(|cell| cell.fg) .collect(); - // The map draws the skeleton after the interior, so a cell holding a landmark reads as skeleton + // The map draws the skeleton after the interior. A cell holding a landmark reads as skeleton // and the rest as the sample. Both colors are present: the map distinguishes them. assert!(colors.contains(&SKELETON), "{colors:?}"); assert!(colors.contains(&ACCENT), "{colors:?}"); @@ -462,12 +489,10 @@ fn a_landmark_colors_the_cell_it_lands_in() { fn the_map_keeps_the_placement_square() { let inner = Rect::new(0, 0, 40, 7); - // A braille dot is as tall as it is wide, so equal data units - // per dot on both axes is what keeps the atlas its own shape - // instead of one stretched to fill the frame. Whichever axis - // needs the most units per dot sets the scale, so the other one - // gets slack and the placement always fits: asserted from both - // sides, because a scale read off one axis alone is square too + // A braille dot is as tall as it is wide, so equal data units per dot on both axes is what + // keeps the atlas its own shape instead of one stretched to fill the frame. Whichever axis + // needs the most units per dot sets the scale. The other one gets slack and the placement + // always fits: asserted from both sides, because a scale read off one axis alone is square too // and lets the other axis run off the frame. for placement in [ [Vec2::new(-8.0, -1.0), Vec2::new(8.0, 1.0)], @@ -477,8 +502,8 @@ fn the_map_keeps_the_placement_square() { let across = (horizontal[1] - horizontal[0]) / (f64::from(inner.width) * 2.0); let down = (vertical[1] - vertical[0]) / (f64::from(inner.height) * 4.0); - // The map builds the viewport in `f32` and widens it for the canvas, so the two readings - // agree to within a rounding of the extent they were rebuilt from. + // The map builds the viewport in `f32` and widens it for the canvas. Both readings agree to + // within a rounding of the extent they were rebuilt from. let tolerance = 4.0 * f64::from(f32::EPSILON) * across; assert!( (across - down).abs() <= tolerance, @@ -497,6 +522,9 @@ fn the_map_keeps_the_placement_square() { } } +/// A row with a non-finite coordinate leaves the drawn band identical to the placement without it. +/// +/// The canvas would otherwise clamp it into a corner it does not occupy. #[test] fn a_row_the_canvas_cannot_place_is_dropped_rather_than_drawn() { let mut state = training(150, 600); @@ -550,6 +578,10 @@ fn probed(readings: usize) -> RunState { state } +/// A full admission battery draws one indented row per reading under the admission stage. +/// +/// The rows follow the battery's own order, each with leader dots out to its value and spreads +/// rendered as spreads. The rail's footer and the log keep their places below. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -576,6 +608,9 @@ fn the_admission_readings_hang_under_the_stage_that_measured_them() { assert!(drawn.join("\n").contains(" log "), "{drawn:#?}"); } +/// A partial battery draws only the readings taken. +/// +/// The rail invents neither a row nor a zero for evidence the probe does not have. #[expect( clippy::non_ascii_literal, reason = "the assertions read the dashboard's own glyphs" @@ -592,26 +627,32 @@ fn a_battery_missing_evidence_draws_only_the_readings_it_has() { assert!(!drawn.join("\n").contains("continuity"), "{drawn:#?}"); } +/// A pane that cannot spare a row per reading shows none of them rather than half a battery. +/// +/// Half a battery would read as evidence the probe could not measure. The log keeps its place +/// below. #[test] fn readings_a_short_pane_cannot_hold_whole_stay_out_of_the_rail() { let drawn = draw_on(&probed(6), 0, 60, 20); let pane = drawn.join("\n"); - // Half a battery would read as evidence the probe could not - // measure, so a pane that cannot spare a row per reading shows - // none of them - and the log keeps its voice either way. + // Half a battery would read as evidence the probe could not measure. A pane that cannot spare a + // row per reading shows none of them - and the log keeps its voice either way. assert!(drawn[13].contains("stages 12/12"), "{drawn:#?}"); assert!(!pane.contains("recall"), "{pane}"); assert!(!pane.contains("triplet agreement"), "{pane}"); assert!(pane.contains(" log "), "{pane}"); } +/// On a short pane the chart is the first thing dropped. +/// +/// The rail and the log, the run's shape and its voice, are both still drawn. #[test] fn a_pane_too_short_for_a_chart_keeps_the_rail_and_the_log() { let drawn = draw(&training(150, 600), 0); let pane = drawn.join("\n"); - // The rail is the run's shape and the log is its voice; the + // The rail is the run's shape and the log is its voice. The // chart is the first thing to go. assert!(!pane.contains(" loss "), "{pane}"); assert!(pane.contains("admission"), "{pane}"); @@ -658,12 +699,16 @@ fn the_log_pane_shows_the_newest_lines_that_fit() { let drawn = draw(&state, 0); let pane = drawn[15..].join("\n"); - // The pane is a tail rather than a scrollback, so the newest line is always visible and the - // pane drops the oldest off the top. + // The pane is a tail rather than a scrollback. The newest line is always visible and the pane + // drops the oldest off the top. assert!(pane.contains("line 39"), "{pane}"); assert!(!pane.contains("line 0 "), "{pane}"); } +/// Stage spans render as seconds to a tenth below a minute. +/// +/// At or above a minute they render as minutes and zero-padded seconds, including the minute +/// boundary itself. #[test] fn spans_read_as_a_clock() { assert_eq!(duration(Duration::from_millis(400)), "0.4s"); diff --git a/libs/@local/graph/atlas/src/cli/tui/state/mod.rs b/libs/@local/graph/atlas/src/cli/tui/state/mod.rs index 57e1930aa1f..a2ee9a742a6 100644 --- a/libs/@local/graph/atlas/src/cli/tui/state/mod.rs +++ b/libs/@local/graph/atlas/src/cli/tui/state/mod.rs @@ -4,8 +4,8 @@ //! placement, the admission probe's readings, and the log tail. It absorbs [`Observation`]s and //! answers the questions the renderer asks - what is each stage doing, how far the paid embedding //! has come, how the placement is descending, where its rows currently sit, and what the run has -//! said lately. The model holds no terminal and no channel, and its only clock is the run's start, -//! so the whole reduction is exercisable without drawing anything. +//! said lately. The model holds no terminal and no channel, and its only clock is the run's start. +//! The whole reduction is exercisable without drawing anything. use alloc::collections::VecDeque; use core::time::Duration; @@ -22,7 +22,7 @@ use crate::{ /// Log lines the dashboard keeps behind the visible tail. /// -/// A tall terminal shows a few dozen; the rest are scrollback the pane does not offer yet, kept +/// A tall terminal shows a few dozen. The rest are scrollback the pane does not offer yet, kept /// bounded so a long run cannot grow the model without limit. const LOG_CAPACITY: usize = 256; @@ -39,8 +39,8 @@ pub(super) enum StageStatus { /// One card-embedding workload, with the split that sized it and what the provider has returned. /// -/// `reused` and `embedded` partition the run's distinct card texts; `done` counts the `embedded` -/// share that has come back, so a workload served entirely from the prior generation is complete at +/// `reused` and `embedded` partition the run's distinct card texts. `done` counts the `embedded` +/// share that has come back. A workload served entirely from the prior generation is complete at /// `embedded == 0`. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(super) struct EmbeddingWorkload { @@ -54,14 +54,14 @@ pub(super) struct EmbeddingWorkload { /// Training steps the loss curve keeps. /// -/// A schedule is a few thousand steps, so the whole curve normally fits and the chart shows the -/// run's entire descent; a longer schedule scrolls, oldest first, rather than growing the model. +/// A schedule is a few thousand steps. The whole curve normally fits and the chart shows the run's +/// entire descent. A longer schedule scrolls, oldest first, rather than growing the model. const LOSS_CAPACITY: usize = 4_096; /// One classifier fit's cross-validation folds, how many there are and how many have completed. /// -/// The folds fit in parallel and report in completion order, so the model counts arrivals rather -/// than tracking which index is outstanding. +/// The folds fit in parallel and report in completion order. The model counts arrivals rather than +/// tracking which index is outstanding. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(super) struct ClassifierFolds { /// Cross-validation folds the fit will run. @@ -72,8 +72,8 @@ pub(super) struct ClassifierFolds { /// One placement training run, how far it has come and the loss curve it has drawn. /// -/// `losses` is the retained tail of the composite objective, oldest first, so the window is the -/// curve: `done` places it on the schedule, and the chart draws its offsets. +/// `losses` is the retained tail of the composite objective, oldest first. The window is the curve: +/// `done` places it on the schedule, and the chart draws its offsets. #[derive(Debug, Clone, PartialEq)] pub(super) struct ProjectorTraining { /// Steps the schedule will run. @@ -88,11 +88,13 @@ pub(super) struct ProjectorTraining { /// What the neighbour-table construction is doing, from its latest observation. /// -/// A construction runs one part at a time and always in the same order. Every row goes into the -/// search backend, the backend does its own linking (or NN-Descent runs its iterations, which need -/// no backend), every row's list comes back out, and the recall verdict arrives last. Each -/// observation therefore replaces the last instead of accumulating. The model carries the -/// construction's newest word, which is what the stage is doing. +/// A construction runs one part at a time and always in the same order, and the configuration picks +/// which construction runs. An index-backed one sends every row into the search backend, the +/// backend does its own linking, and every row's list comes back out. NN-Descent needs no backend +/// and runs its iterations instead. A run reports the loops or the iterations, never both. The +/// recall verdict arrives last either way. Each observation therefore replaces the last instead of +/// accumulating. The model carries the construction's newest word, which is what the stage is +/// doing. #[derive(Debug, Clone, PartialEq)] pub(super) enum KnnActivity { /// Rows entering the search backend. @@ -123,9 +125,9 @@ pub(super) struct PlacementMap { /// One thing the run reported, either a progress observation or a line it logged. /// /// This is the model's whole input vocabulary. Each variant owns what the run handed one -/// [`Progress`] method, so a reporting thread parts with it and never waits on the renderer - -/// except [`Knn`](Self::Knn), where one stage's five observations arrive as the one activity -/// vocabulary they fold into. +/// [`Progress`] method, except [`Knn`](Self::Knn): the kNN stage's five reporting methods share +/// that one variant, whose activity vocabulary they fold into. A reporting thread parts with its +/// observation and never waits on the renderer. /// /// [`Progress`]: crate::progress::Progress #[derive(Debug)] @@ -196,7 +198,7 @@ pub(super) struct RunState { placement: Option, /// The admission probe's readings, indexed as [`QualityMetric::ALL`]. /// - /// A control whose evidence is absent reports nothing, so a slot stays empty for a metric the + /// A control whose evidence is absent reports nothing. The slot stays empty for a metric the /// probe could not measure as well as for one it has not measured yet. The rail draws what /// landed and invents nothing for the rest. quality: [Option; QualityMetric::ALL.len()], @@ -261,8 +263,8 @@ impl RunState { /// Advances the open workload to a completed request's position. /// - /// This drops a batch without a split rather than guessing at it: the provider's count - /// describes its own workload, and nothing here may invent the reuse the split reported. + /// This drops a batch without a split rather than guessing at it: a batch carries its own + /// totals alone, and the reuse split arrives with the workload that opened it. pub(super) const fn advance_embedding(&mut self, batch: Batch) { let Some(workload) = self.embedding.as_mut() else { return; @@ -305,15 +307,17 @@ impl RunState { /// Records what the neighbour-table construction is doing now. /// /// Each activity replaces the last. The construction's loops, phases and verdict happen in one - /// order, so nothing behind the newest one is still in flight. + /// order. Nothing behind the newest one is still in flight. pub(super) fn report_knn(&mut self, activity: KnnActivity) { self.knn = Some(activity); } /// Records one training step of the placement. /// - /// The first step opens the curve; a later step with a different schedule length opens a fresh - /// one, so a second training run cannot inherit the first one's descent. + /// The first step opens the curve, and so does any later step carrying a different schedule + /// length or an index below the steps already counted. A repeated or out-of-order step from + /// the run in progress opens one as well. A second training run that reports from step zero + /// therefore begins its own descent instead of extending the first. pub(super) fn advance_projector(&mut self, step: usize, steps: usize, loss: &LossBreakdown) { let training = match self.projector.as_mut() { Some(training) if training.steps == steps && step >= training.done => training, @@ -330,8 +334,8 @@ impl RunState { } training.losses.push_back(loss.total()); training.last = *loss; - // Steps are zero-based and `done` counts them, so the step that - // reports index `n` is the `n + 1`th of the schedule. + // Steps are zero-based and `done` counts them. The step that reports index `n` is the `n + + // 1`th of the schedule. training.done = step + 1; } @@ -349,7 +353,7 @@ impl RunState { /// Records one measured quality metric of the admission probe. /// /// A second reading of the same metric replaces the first: the reading a control turns on is - /// one reduction over the probe's steps, so a repeat is a fresher answer to the same question + /// one reduction over the probe's steps. A repeat is a fresher answer to the same question /// rather than a second measurement. pub(super) fn probe_quality(&mut self, metric: QualityMetric, reading: f64) { let Some(index) = QualityMetric::ALL @@ -397,6 +401,10 @@ impl RunState { /// Spans are differences between completions: a stage's own span is the gap between its /// completion and its predecessor's, and the running stage's span is the gap since the last /// completion. The rail's numbers therefore always sum to the wall clock. + /// + /// # Panics + /// + /// Panics when `index` is not a position in [`Stage::ALL`]. pub(super) fn status(&self, index: usize, elapsed: Duration) -> StageStatus { let previous = index .checked_sub(1) @@ -451,12 +459,13 @@ impl RunState { self.placement.as_ref() } - /// The admission probe's readings, in [`QualityMetric::ALL`] order, skipping what it has not - /// measured. + /// Returns the admission probe's readings in [`QualityMetric::ALL`] order. + /// + /// The iterator skips what the probe has not measured. /// - /// Every reading arrives in one burst as the probe's report reduces its steps, so the sequence - /// is normally empty or whole; a short one is a battery whose evidence was absent for the - /// missing controls. + /// Every reading arrives in one burst as the probe's report reduces its steps. The sequence is + /// normally empty or whole. A short one is a battery whose evidence was absent for the missing + /// controls. pub(super) fn quality(&self) -> impl Iterator + use<> { QualityMetric::ALL .into_iter() diff --git a/libs/@local/graph/atlas/src/cli/tui/state/tests.rs b/libs/@local/graph/atlas/src/cli/tui/state/tests.rs index 494a1c7daf2..459130bb680 100644 --- a/libs/@local/graph/atlas/src/cli/tui/state/tests.rs +++ b/libs/@local/graph/atlas/src/cli/tui/state/tests.rs @@ -1,7 +1,7 @@ //! Every observation folded into the model, and everything the renderer reads back out of it. //! -//! The reduction carries no clock of its own beyond the run's start, so each question here names -//! the elapsed time it asks at ([`RunState::complete_at`]) and answers it without a terminal or a +//! The reduction carries no clock of its own beyond the run's start. Each question here names the +//! elapsed time it asks at ([`RunState::complete_at`]) and answers it without a terminal or a //! running fit. use core::time::Duration; @@ -55,6 +55,9 @@ fn secs(seconds: u64) -> Duration { Duration::from_secs(seconds) } +/// A model with nothing observed reports its first stage running for the whole elapsed time. +/// +/// Every later stage stays pending, and no stage completes. #[test] fn a_fresh_run_is_inside_its_first_stage() { let state = RunState::new(); @@ -64,6 +67,9 @@ fn a_fresh_run_is_inside_its_first_stage() { assert_eq!(state.completed_stages(), 0); } +/// Each completed stage's span is the gap between its own completion and the one before it. +/// +/// The stage after the last completion runs for the time since it, and the rest stay pending. #[test] fn spans_are_differences_between_completions() { let mut state = RunState::new(); @@ -121,6 +127,9 @@ fn a_split_opens_the_counter_and_requests_advance_it() { ); } +/// A second reported split starts its own counter from zero. +/// +/// The finished workload's progress does not carry into it. #[test] fn a_second_workload_replaces_the_first() { let mut state = RunState::new(); @@ -133,7 +142,7 @@ fn a_second_workload_replaces_the_first() { total: 49, }); - // The corpus finished; the cards are their own workload and the + // The corpus finished. The cards are their own workload and the // counter must not carry the corpus's progress into them. state.start_embedding(&CardEmbeddingStats { reused: 12, @@ -165,6 +174,9 @@ fn an_announced_fold_count_opens_the_counter_and_completions_advance_it() { ); } +/// A fold completion arriving before any fold count leaves the classifier counter closed. +/// +/// The model invents no total to count against. #[test] fn a_fold_completion_without_an_announced_count_is_dropped() { let mut state = RunState::new(); @@ -173,6 +185,10 @@ fn a_fold_completion_without_an_announced_count_is_dropped() { assert_eq!(state.classifier(), None); } +/// The knn field holds the latest reported activity only. +/// +/// Insertion, then the backend's linking, then reading back, then the recall verdict each replace +/// the predecessor. #[test] fn each_construction_activity_replaces_the_one_before_it() { let mut state = RunState::new(); @@ -203,6 +219,10 @@ fn each_construction_activity_replaces_the_one_before_it() { assert_eq!(state.knn(), Some(&KnnActivity::Measured(check(0.9021)))); } +/// The first reported step opens the descent curve with the schedule's total. +/// +/// Later steps extend the loss window in order, and the done count and the latest breakdown track +/// the newest step. #[test] fn the_first_training_step_opens_the_curve_and_the_rest_extend_it() { let mut state = RunState::new(); @@ -218,12 +238,16 @@ fn the_first_training_step_opens_the_curve_and_the_rest_extend_it() { let training = state.projector().expect("four steps opened the curve"); assert_eq!(training.steps, 300); - // Steps are zero-based, so the fourth one reports index three. + // Steps are zero-based. The fourth one reports index three. assert_eq!(training.done, 4); assert_eq!(training.losses, [8.0, 7.0, 6.0, 5.0]); assert_eq!(training.last, loss(5.0)); } +/// A run longer than the loss window keeps the window at capacity. +/// +/// The window holds the newest losses and drops as many oldest ones as the run ran over, while the +/// done count keeps the whole run's length. #[expect( clippy::cast_precision_loss, reason = "the fixture's step count is exactly representable" @@ -238,9 +262,8 @@ fn the_curve_scrolls_rather_than_growing_without_limit() { let training = state.projector().expect("the curve opened"); assert_eq!(training.losses.len(), LOSS_CAPACITY); - // The run went two steps past the window, so the two oldest - // losses are the ones that left and the window holds steps two - // onward. + // The run went two steps past the window. Its two oldest losses are the ones that left, and the + // window holds steps two onward. assert_eq!(training.losses.front(), Some(&2.0)); assert_eq!(training.losses.back(), Some(&(steps as f32 - 1.0))); assert_eq!(training.done, steps); @@ -260,6 +283,9 @@ fn a_second_training_run_does_not_inherit_the_first_curve() { assert_eq!(training.losses, [4.0]); } +/// The placement map holds the latest snapshot only. +/// +/// The map shows where the placement is, not where it has been. #[test] fn a_snapshot_replaces_the_one_before_it() { let mut state = RunState::new(); @@ -289,6 +315,9 @@ fn a_landmark_count_past_the_reported_rows_is_clamped() { assert_eq!(placement.landmarks, 1); } +/// An embedding batch arriving before any reported split leaves the workload closed. +/// +/// Opening a counter would leave no reused-against-embedded split to report. #[test] fn a_request_without_a_split_is_dropped() { let mut state = RunState::new(); @@ -302,7 +331,7 @@ fn the_batterys_burst_lands_in_metric_order_however_it_arrives() { let mut state = RunState::new(); assert_eq!(state.quality().count(), 0); - // The probe reports its readings as its report reduces them; the model owes the renderer + // The probe reports its readings as its report reduces them. The model owes the renderer // the battery's own order, not the arrival order. state.probe_quality(QualityMetric::TripletAgreement, 0.7820); state.probe_quality(QualityMetric::Recall, 0.9021); @@ -318,20 +347,26 @@ fn the_batterys_burst_lands_in_metric_order_however_it_arrives() { ); } +/// Probing one metric twice leaves a single row carrying the fresher reading. +/// +/// One reduction over the probe's steps answers one question. #[test] fn a_second_reading_of_one_metric_replaces_the_first() { let mut state = RunState::new(); state.probe_quality(QualityMetric::Continuity, 0.8000); state.probe_quality(QualityMetric::Continuity, 0.9104); - // One reduction over the probe's steps answers one question, - // so a repeat is a fresher answer and never a second row. + // One reduction over the probe's steps answers one question. A repeat is a fresher answer and + // never a second row. assert_eq!( state.quality().collect::>(), [(QualityMetric::Continuity, 0.9104)] ); } +/// Pushing one line past the log's capacity holds the tail at capacity. +/// +/// The oldest line drops and the newest stays. #[test] fn the_log_tail_evicts_its_oldest_line() { let mut state = RunState::new(); diff --git a/libs/@local/graph/atlas/src/dataset/auxiliary.rs b/libs/@local/graph/atlas/src/dataset/auxiliary.rs index 9a0509f9379..50345686d75 100644 --- a/libs/@local/graph/atlas/src/dataset/auxiliary.rs +++ b/libs/@local/graph/atlas/src/dataset/auxiliary.rs @@ -1,13 +1,12 @@ //! Display payloads a dataset supplies beside its row identities. //! -//! An identity file carries one display value per row in its payload region, stored as raw -//! bytes and read back as the typed view the id type declares through -//! [`Key::Payload`](crate::file::identity::Key::Payload). [`Legend`] is the display value of a -//! node or edge row - the row's representative ontology type beside its display label - and -//! [`Icon`] the display value of an ontology-type row. Text is UTF-8 at byte level, so casting -//! a payload span validates it and rejects a span that holds anything else. A row that -//! displays nothing carries its type's empty value: the empty icon, or a legend whose label -//! is empty. +//! An identity file carries one display value per row in its payload region, stored as raw bytes +//! and read back as the typed view the id type declares through +//! [`Key::Payload`](crate::file::identity::Key::Payload). [`Legend`] is the display value of a node +//! or edge row - the row's representative ontology type beside its display label - and [`Icon`] the +//! display value of an ontology-type row. Text is UTF-8 at byte level. Casting a payload span +//! validates it and rejects a span that holds anything else. A row that displays nothing carries +//! its type's empty value: the empty icon, or a legend whose label is empty. use alloc::sync::Arc; use core::{borrow::Borrow, clone::CloneToUninit, mem::offset_of, ops::Deref}; @@ -32,7 +31,9 @@ use crate::identity::OntologyRowId; )] #[repr(C)] pub(crate) struct Legend { + /// The ontology row of the type standing for the row. representative_ontology: OntologyRowId, + /// The display text. As the trailing field it gives the value its length. label: Label, } @@ -48,15 +49,17 @@ impl Legend { } } +// The `IntoBytes` and `CloneToUninit` arguments below rely on both facts: alignment one leaves no +// padding anywhere in a `Legend`, and the representative is at offset zero. const _: () = { assert!(align_of::() == 1); assert!(offset_of!(Legend, representative_ontology) == 0); }; -// SAFETY: `repr(C)` with `OntologyRowId` (`Unaligned` + `IntoBytes`) followed by `str` gives -// every field alignment 1, so no padding exists at any length and the value has no uninitialized -// byte. The derive cannot compute this because its padding proof sizes each field and -// special-cases only a trailing slice. `str` is layout-identical to `[u8]` but not a slice type. +// SAFETY: `repr(C)` with `OntologyRowId` (`Unaligned` + `IntoBytes`) followed by `str` gives every +// field alignment 1. No padding exists at any length and the value has no uninitialized byte. The +// derive cannot compute this because its padding proof sizes each field and special-cases only a +// trailing slice. `str` is layout-identical to `[u8]` but not a slice type. unsafe impl zerocopy::IntoBytes for Legend { #[expect( dead_code, @@ -65,18 +68,22 @@ unsafe impl zerocopy::IntoBytes for Legend { fn only_derive_is_allowed_to_implement_this_trait() {} } -// SAFETY: the implementation writes the label at its in-value offset and the representative at -// offset 0. `repr(C)` at alignment 1 puts no padding between them, so the two writes -// initialize every byte of the clone and `dest` holds a valid `Legend` on return. +// SAFETY: `CloneToUninit` requires a valid `Self` at `dest` on normal return. The label clone +// preserves its UTF-8 text and length, and the representative is `Copy`. The writes initialize +// both fields at their `repr(C)` offsets, giving a valid `Legend` with the source's metadata. +// Neither field owns resources that could leak during unwinding. unsafe impl CloneToUninit for Legend { unsafe fn clone_to_uninit(&self, dest: *mut u8) { - // SAFETY: `self.label` is a field of `self`, so both pointers lie in one allocation - // with the field's address not below the value's. + // SAFETY: Both pointers derive from `self` in the same allocation. The label begins at or + // after `self`, and the byte distance cannot exceed the allocation's size. The unsigned + // offset is valid. let offset_of_label = unsafe { (&raw const self.label).byte_offset_from_unsigned(self) }; - // SAFETY: the caller provides `dest` valid for `size_of_val(self)` bytes at alignment - // 1; the label's span and the representative's eight bytes at offset 0 both lie inside - // that span. + // SAFETY: The caller supplies writable storage for the complete value. The source's + // metadata fixes the label's length and destination layout. Its span at `offset_of_label` + // and the representative's eight bytes at offset zero are disjoint and fit within + // `size_of_val(self)`. Both fields have alignment one. These writes initialize both fields + // without creating a reference to uninitialized memory. unsafe { self.label.clone_to_uninit(dest.add(offset_of_label)); dest.add(offset_of!(Self, representative_ontology)) @@ -103,13 +110,18 @@ pub(crate) struct OwnedLegend(Box); impl OwnedLegend { /// Creates the legend pairing `representative` with `label`. + /// + /// # Panics + /// + /// Panics if the legend header plus the label exceeds [`isize::MAX`] bytes, or if the zeroed + /// allocation fails. pub(crate) fn new(representative: OntologyRowId, label: &Label) -> Self { let mut boxed = Legend::new_box_zeroed_with_elems(label.len()) .expect("a label's length fits the allocator's limits"); boxed.representative_ontology = representative; - // SAFETY: the write copies the bytes of a valid `&Label` whole, so the field holds - // valid UTF-8 when the borrow ends. + // SAFETY: the write copies the bytes of a valid `&Label` whole. The field holds valid UTF-8 + // when the borrow ends. unsafe { boxed.label.0.as_bytes_mut() }.copy_from_slice(label.as_bytes()); Self(boxed) } @@ -172,7 +184,7 @@ impl Label { let ptr = &raw const *text; let ptr = ptr as *const Self; - // SAFETY: `Label` is `repr(C)` with `str` as its only field, so it has `str`'s size, + // SAFETY: `Label` is `repr(C)` with `str` as its only field. It has `str`'s size, // alignment, and pointer metadata, and the cast keeps the address, length metadata, and // provenance of `text`. The target is therefore a live, validly initialized `Label` whose // borrow is `text`'s. @@ -180,12 +192,14 @@ impl Label { } } -// SAFETY: `Label` is `repr(C)` around `str` alone, so its clone is its text's clone and -// `str`'s implementation initializes every byte of `dest`. +// SAFETY: `Label` is `repr(C)` with one `str` field at offset zero. It has the field's size, +// alignment and length metadata. The `str` clone initializes that complete UTF-8 text, giving a +// valid `Label` with the source's metadata. unsafe impl CloneToUninit for Label { unsafe fn clone_to_uninit(&self, dest: *mut u8) { - // SAFETY: `Label` has `str`'s size and alignment, so the caller's contract for this - // value is `str`'s contract for its text. + // SAFETY: `Label` and its `str` field have identical size and alignment. The caller's + // writable range therefore covers the complete text at the required alignment. The `str` + // clone receives exactly its own destination requirements. unsafe { ::clone_to_uninit(&self.0, dest); } @@ -279,13 +293,14 @@ impl Icon { let ptr = &raw const *text; let ptr = ptr as *const Self; - // SAFETY: `Icon` is `repr(C)` with `str` as its only field, so it has `str`'s size, - // alignment, and pointer metadata, and the cast keeps the address, length metadata, and - // provenance of `text`. The target is therefore a live, validly initialized `Icon` whose - // borrow is `text`'s. + // SAFETY: `Icon` is `repr(C)` with `str` as its only field. It has `str`'s size, alignment, + // and pointer metadata, and the cast keeps the address, length metadata, and provenance of + // `text`. The target is therefore a live, validly initialized `Icon` whose borrow is + // `text`'s. unsafe { &*ptr } } + /// The empty icon, the display of a row that has none. pub(crate) const fn empty() -> &'static Self { const EMPTY: &Icon = Icon::new(""); @@ -347,15 +362,17 @@ impl Deref for OwnedIcon { mod tests { #![expect(clippy::non_ascii_literal)] - /// Every UTF-8 shape the cast has to carry: empty, ASCII, two-byte, combining mark, and a - /// four-byte scalar. + /// Selected UTF-8 inputs for the borrowed-view checks. + /// + /// The cases include empty text, ASCII, two-byte scalars, a combining mark and a four-byte + /// scalar. const SHAPES: [&str; 5] = ["", "a", "naïve", "z\u{0301}", "🦀 crab"]; /// The tests the `miri` nextest profile selects. /// /// Each test here views auxiliary text in place over its source bytes and validates the payload - /// reads behind those views. The profile selects by module path, so moving a test in or out - /// of this module is the whole edit. + /// reads behind those views. The profile selects by module path. Moving a test in or out of + /// this module is the whole edit. mod miri { use core::ptr; @@ -376,6 +393,7 @@ mod tests { assert_eq!(legend.representative_ontology(), representative); assert_eq!(legend.label(), "naïve 🦀"); + // A legend's bytes are the 8-byte representative then the label's UTF-8. let bytes = legend.as_bytes(); assert_eq!(bytes.len(), 8 + "naïve 🦀".len()); let back = Legend::try_ref_from_bytes(bytes).expect("wrote valid bytes"); @@ -399,6 +417,10 @@ mod tests { assert_eq!(reowned, owned.clone()); } + /// `Label::new` is a cast. + /// + /// For every UTF-8 shape the label has the text's bytes, size, alignment one, and the + /// text's own address. #[test] fn label_views_the_source_text_in_place() { for text in SHAPES { @@ -410,6 +432,10 @@ mod tests { } } + /// `Icon::new` is a cast. + /// + /// For every UTF-8 shape the icon has the text's bytes, size, alignment one, and the text's + /// own address. #[test] fn icon_views_the_source_text_in_place() { for text in SHAPES { @@ -421,6 +447,9 @@ mod tests { } } + /// `Borrow: WriteInto {} impl WriteAs for &T where T: WriteAs + ?Sized {} diff --git a/libs/@local/graph/atlas/src/file/morton/mod.rs b/libs/@local/graph/atlas/src/file/morton/mod.rs index 0399f3dc432..acf0dd46239 100644 --- a/libs/@local/graph/atlas/src/file/morton/mod.rs +++ b/libs/@local/graph/atlas/src/file/morton/mod.rs @@ -112,6 +112,15 @@ pub(crate) const SEGMENTS: usize = Depth::MAX.get() as usize + 1; pub(crate) struct Fenceposts([U64; POSTS], PhantomData); impl Fenceposts { + /// Checks the two structural rules every fencepost array obeys. + /// + /// The array anchors at zero and never decreases, which together make each consecutive pair + /// a well-formed range. + /// + /// # Errors + /// + /// Returns [`FencepostError::Anchor`] when the first post is not zero, and + /// [`FencepostError::Order`] naming the first post smaller than its predecessor. #[expect( clippy::cast_possible_truncation, reason = "fencepost indices are bounded by the 34 posts" @@ -170,10 +179,12 @@ impl Fenceposts { Ok(Self(posts, PhantomData)) } + /// Returns the persisted words, giving up the validated wrapper. const fn into_raw(self) -> [U64; POSTS] { self.0 } + /// Borrows the persisted words without giving up the wrapper. const fn as_raw(&self) -> &[U64; POSTS] { &self.0 } @@ -247,8 +258,10 @@ impl Fenceposts { } } -// The single variant makes the derive validate the discriminant, so parsing admits exactly the -// pinned magic value. +/// The discriminant carrier behind [`FileHeaderMagic`]. +/// +/// Parsing admits exactly the pinned magic value because the derive validates the single +/// variant's discriminant. #[derive( Debug, Copy, diff --git a/libs/@local/graph/atlas/src/file/morton/tests.rs b/libs/@local/graph/atlas/src/file/morton/tests.rs index 9610bf2e60e..a58cfaa8e84 100644 --- a/libs/@local/graph/atlas/src/file/morton/tests.rs +++ b/libs/@local/graph/atlas/src/file/morton/tests.rs @@ -1,3 +1,9 @@ +//! Certificates for the morton file's format. +//! +//! The tests pin the fencepost rules the open validates, the header's wire layout byte by byte, +//! the region geometry, the writer-to-reader round trip, and the run queries against +//! hand-computed cells - with a property test holding `run` to a linear scan over arbitrary +//! columns. #![expect( clippy::little_endian_bytes, reason = "the wire-layout assertions pin the format's canonical little-endian bytes" @@ -20,6 +26,7 @@ use crate::{ morton::{Depth, MortonCell, MortonKey}, }; +/// A subdivision depth, panicking on a fixture outside the documented domain. fn depth(value: u8) -> Depth { Depth::new(value).expect("test depths lie within the documented domain") } @@ -46,6 +53,10 @@ fn scratch(name: &str) -> PathBuf { dir.join(name) } +/// Anchored, non-decreasing fenceposts wrap, and their segment ranges and per-segment lengths +/// read back what built them, empty segments included. A first post off zero breaks the anchor +/// rule, a decreasing post reports its own index, and lengths whose running sum overflows `u64` +/// build no fenceposts at all, because they match no real column. #[test] fn fenceposts_carry_the_structural_rules() { // Anchored, non-decreasing posts wrap; segment ranges and the @@ -83,6 +94,9 @@ fn fenceposts_carry_the_structural_rules() { assert_eq!(Fenceposts::::from_lengths(&lengths), None); } +/// The header's bytes sit where the format's table says: magic, little-endian version 2, this +/// machine's information, the index stride, and all fenceposts as little-endian `u64`s - the last +/// count repeating out to the final post - with zero padding to 4096. #[test] fn header_wire_layout() { let header = PaddedFileHeader::new(FileHeader::new(512, posts_of(&[600, 400]))); @@ -130,6 +144,8 @@ fn header_parse_pins_identity() { .expect_err("an unsupported version should not parse"); } +/// The header's index-key count, code-region offset and expected file length agree with the +/// geometry computed by hand from the code count and the index stride. #[test] fn region_geometry() { // 1000 codes at stride 512 need two keys. Padding the 16-byte index to one page puts codes at @@ -163,7 +179,9 @@ fn region_geometry() { /// Depth-1 quadrant prefixes of a 64-bit key: bit 62 is the x axis's top bit, bit 63 the y axis's. const Q10: u64 = 0x4000_0000_0000_0000; +/// The quadrant (0, 1) prefix. const Q01: u64 = 0x8000_0000_0000_0000; +/// The quadrant (1, 1) prefix. const Q11: u64 = 0xC000_0000_0000_0000; /// A depth-2 sub-cell of quadrant (0, 0): top four key bits 0001. const SUB: u64 = 0x1000_0000_0000_0000; @@ -195,6 +213,7 @@ fn fixture_codes() -> Vec { .collect() } +/// Builds fenceposts for one, three, and five codes in the leading three buckets. fn fixture_posts() -> Fenceposts { posts_of(&[1, 3, 5]) } @@ -265,6 +284,8 @@ fn runs_slice_hand_computed_cells() { assert_eq!(file.run(depth(2), cell), span(7..7)); } +/// A zero-code column is valid geometry: it writes, reopens, reports no codes, and answers the +/// root cell's run with the empty range rather than failing. #[test] fn empty_column_reopens() { let path = scratch("empty.mrtn"); diff --git a/libs/@local/graph/atlas/src/file/policy/mod.rs b/libs/@local/graph/atlas/src/file/policy/mod.rs index 5acdd7d72a5..ea1e2202375 100644 --- a/libs/@local/graph/atlas/src/file/policy/mod.rs +++ b/libs/@local/graph/atlas/src/file/policy/mod.rs @@ -28,7 +28,7 @@ //! whole-file-mapping alignment guarantee of the array format applies unchanged. Map the whole //! file and slice, never mmap at a file offset. //! -//! [`read::PolicyFile`] opens a file under these rules and hands out the raw typed rows; +//! [`read::PolicyFile`] opens a file under these rules and hands out the raw typed rows. //! [`write::write_rows`] streams them into place. The format owns geometry alone - the table's //! domain invariants (strictly ascending relations, probabilities and applicability in `[0, 1]`, //! finite nonnegative strength) are `salt::policy`'s artifact contract, validated where the domain @@ -52,8 +52,10 @@ pub(crate) mod write; use super::region::machine::{Architecture, Machine}; use crate::file::region::{PAGE, header::header}; -// The single variant makes the derive validate the discriminant, so parsing admits exactly the -// pinned magic value. +/// The discriminant carrier behind [`FileHeaderMagic`]. +/// +/// Parsing admits exactly the pinned magic value because the derive validates the single +/// variant's discriminant. #[derive( Debug, Copy, diff --git a/libs/@local/graph/atlas/src/file/policy/tests.rs b/libs/@local/graph/atlas/src/file/policy/tests.rs index b72635a1780..94c549cac09 100644 --- a/libs/@local/graph/atlas/src/file/policy/tests.rs +++ b/libs/@local/graph/atlas/src/file/policy/tests.rs @@ -1,3 +1,4 @@ +//! Certificates for the policy file's format. #![expect( clippy::little_endian_bytes, reason = "the wire-layout assertions pin the format's canonical little-endian bytes" @@ -18,6 +19,10 @@ use super::{ }; use crate::file::region::{header::HeaderError, machine::Machine}; +/// The header's bytes sit where the format's table says: magic, little-endian version 2, this +/// machine's information, the policy count as a little-endian `u64`, and zero padding out to +/// 4096. The expected file length is the header page plus one 56-byte row per policy, and an +/// overflowing count reports none, because it matches no real file. #[test] fn header_wire_layout() { let header = PaddedFileHeader::new(FileHeader::new(3)); @@ -53,6 +58,7 @@ fn scratch(name: &str) -> PathBuf { dir.join(name) } +/// A resolved policy row for `relation`, varying only the coincident weights. fn fixture_row(relation: u64, coincident: f64) -> PolicyRow { PolicyRow { relation, @@ -66,6 +72,7 @@ fn fixture_row(relation: u64, coincident: f64) -> PolicyRow { } } +/// Builds a policy file with three rows in ascending relation order. fn fixture_bytes() -> Vec { let rows = [ fixture_row(2, 0.0), @@ -98,6 +105,8 @@ fn written_rows_reopen_verbatim() { assert_eq!(rows[1].strength, 1.0); } +/// A zero-row table is valid geometry: it writes, reopens, and hands out an empty row region +/// rather than failing the open. #[test] fn empty_table_reopens() { // A zero count is valid geometry: one empty region. diff --git a/libs/@local/graph/atlas/src/file/postings/mod.rs b/libs/@local/graph/atlas/src/file/postings/mod.rs index 4485fb6309e..b102f28d2fe 100644 --- a/libs/@local/graph/atlas/src/file/postings/mod.rs +++ b/libs/@local/graph/atlas/src/file/postings/mod.rs @@ -107,8 +107,10 @@ use crate::{ identity::{BasePosition, OntologyRowId}, }; -// The single variant makes the derive validate the discriminant, so parsing admits exactly the -// pinned magic value. +/// The discriminant carrier behind [`FileHeaderMagic`]. +/// +/// Parsing admits exactly the pinned magic value because the derive validates the single +/// variant's discriminant. #[derive( Debug, Copy, diff --git a/libs/@local/graph/atlas/src/file/postings/tests.rs b/libs/@local/graph/atlas/src/file/postings/tests.rs index 5cb02b952f6..5a77ccce4a1 100644 --- a/libs/@local/graph/atlas/src/file/postings/tests.rs +++ b/libs/@local/graph/atlas/src/file/postings/tests.rs @@ -1,3 +1,8 @@ +//! Certificates for the postings file's format. +//! +//! The tests pin the header's wire layout byte by byte, the region geometry over a hand-computed +//! fixture, the writer-to-reader round trip of all nine regions, the open's refusals - including +//! every way a bit set frame can contradict the header - and the writer's own preconditions. #![expect( clippy::little_endian_bytes, reason = "the wire-layout assertions pin the format's canonical little-endian bytes" @@ -55,16 +60,21 @@ fn scratch(name: &str) -> PathBuf { /// membership: runs `0:{0} 1:{1} 2:{1} 3:{0} 5:{1} 9:{0}` with the other positions empty, so the /// direct ids are `[0, 1, 1, 0, 1, 0]` (`M = 6`). const POINTS: u64 = 10; +/// The fixture's list fenceposts, one per type plus the closing post. const LIST_POSTS: [usize; 4] = [0, 3, 3, 3]; +/// The fixture's parent-edge fenceposts, one per type plus the closing post. const PARENT_POSTS: [usize; 4] = [0, 0, 0, 2]; +/// The fixture's direct-map fenceposts, one per position plus the closing post. const DIRECT_POSTS: [usize; 11] = [0, 1, 2, 3, 4, 4, 5, 5, 5, 5, 6]; +/// Builds dense-membership flags with only type 1 stored densely. fn fixture_flags() -> Box> { let mut flags = DenseBitSlice::new_empty(3); flags.insert(OntologyRowId::new(1)); flags } +/// Builds the dense set for type 1 from positions 1, 2, and 5. fn fixture_dense() -> Box> { let mut sets = DenseBitSliceArray::new_empty(10, 1); sets[0].insert(BasePosition::from_u32(1)); @@ -73,33 +83,40 @@ fn fixture_dense() -> Box> { sets } +/// Type 0's membership positions, the only list run the fixture holds. fn fixture_list_entries() -> [BasePosition; 3] { [0, 3, 9].map(BasePosition::from_u32) } +/// Type 2's parent edges, the only parent run the fixture holds. fn fixture_parent_ids() -> [OntologyRowId; 2] { [0, 1].map(OntologyRowId::new) } +/// Returns the transposed memberships in ascending position order. fn fixture_direct_ids() -> [OntologyRowId; 6] { [0, 1, 1, 0, 1, 0].map(OntologyRowId::new) } +/// Per-type membership runs over the base positions. fn fixture_lists() -> Runs { Runs::from_parts(le_posts(&LIST_POSTS), fixture_list_entries().to_vec()) .expect("the fixture posts satisfy the fencepost law") } +/// Per-type parent-edge runs, the edges the closure derives from. fn fixture_parents() -> Runs { Runs::from_parts(le_posts(&PARENT_POSTS), fixture_parent_ids().to_vec()) .expect("the fixture posts satisfy the fencepost law") } +/// Per-position direct-type runs: the same membership read the other way round. fn fixture_direct() -> Runs { Runs::from_parts(le_posts(&DIRECT_POSTS), fixture_direct_ids().to_vec()) .expect("the fixture posts satisfy the fencepost law") } +/// The whole fixture written out as a postings file's bytes. fn fixture_bytes() -> Vec { let flags = fixture_flags(); let dense_sets = fixture_dense(); @@ -121,6 +138,9 @@ fn fixture_bytes() -> Vec { bytes } +/// The header's bytes sit where the format's table says: magic, little-endian version 1, this +/// machine's information, and the six region counts as little-endian `u64`s. A wrong magic and an +/// unsupported version both fail the parse at the byte level. #[test] fn header_wire_layout() { let header = PaddedFileHeader::new(FileHeader::new(3, 10, 3, 1, 2, 6)); @@ -150,6 +170,10 @@ fn header_wire_layout() { .expect_err("an unsupported version should not parse"); } +/// The header's derived geometry - fencepost counts, dense-region length, each region's offset and +/// the expected file length - matches the figures computed by hand for three types over ten +/// points, and a count whose arithmetic overflows `u64` reports no expected length, because it +/// matches no real file. #[test] fn region_geometry() { // The geometry of three types over ten points - a 16-byte flags frame, two four-post regions, @@ -245,6 +269,8 @@ fn written_regions_reopen_verbatim() { assert_eq!(file.direct_ids(), fixture_direct_ids()); } +/// A file over no types and no points is valid geometry: it writes, reopens, and hands out empty +/// entry regions with the single closing fencepost each run column still requires. #[test] fn empty_domain_reopens() { let path = scratch("empty.post"); @@ -425,6 +451,8 @@ fn writer_rejects_mismatched_post_regions() { ); } +/// A writer handed a flags set over a different domain than the type count panics rather than +/// sealing a file the open's flags-domain check would refuse. #[test] #[should_panic(expected = "the flags set covers the type domain")] fn writer_rejects_missized_flags() { @@ -433,6 +461,8 @@ fn writer_rejects_missized_flags() { let _: std::io::Result<()> = write_regions(empty_regions(&flags), &mut sink); } +/// A writer handed a dense set whose type carries no flag bit panics rather than sealing a file +/// in which the flag population and the dense set count disagree. #[test] #[should_panic(expected = "the flags set marks one type per dense set")] fn writer_rejects_unflagged_dense_sets() { @@ -448,6 +478,8 @@ fn writer_rejects_unflagged_dense_sets() { ); } +/// A writer handed a dense set over a different point domain than the file's panics rather than +/// sealing a file whose dense frames the open would refuse. #[test] #[should_panic(expected = "every dense set covers the point domain")] fn writer_rejects_missized_dense_sets() { diff --git a/libs/@local/graph/atlas/src/file/quad/mod.rs b/libs/@local/graph/atlas/src/file/quad/mod.rs index 2694c1d8116..9952b464c47 100644 --- a/libs/@local/graph/atlas/src/file/quad/mod.rs +++ b/libs/@local/graph/atlas/src/file/quad/mod.rs @@ -88,8 +88,10 @@ mod tests; use crate::file::region::{PAGE, header::header, machine::Machine, padded_size}; -// The single variant makes the derive validate the discriminant, so parsing admits exactly the -// pinned magic value. +/// The discriminant carrier behind [`FileHeaderMagic`]. +/// +/// Parsing admits exactly the pinned magic value because the derive validates the single +/// variant's discriminant. #[derive( Debug, Copy, @@ -188,8 +190,8 @@ impl Node { /// Creates a node record. /// - /// `children` are node indexes in Morton child order, [`None`] for absent quadrants; - /// `start..start + length` is the own-bucket run in base delivery positions; `points` counts + /// `children` are node indexes in Morton child order, with [`None`] for absent quadrants. + /// `start..start + length` is the own-bucket run in base delivery positions. `points` counts /// the whole subtree. /// /// # Panics diff --git a/libs/@local/graph/atlas/src/file/quad/read.rs b/libs/@local/graph/atlas/src/file/quad/read.rs index 9740d727d88..a8962e27d48 100644 --- a/libs/@local/graph/atlas/src/file/quad/read.rs +++ b/libs/@local/graph/atlas/src/file/quad/read.rs @@ -92,9 +92,6 @@ impl Error for OpenQuadError { /// index stays inside the table and points deeper in the pre-order. Traversals and set slices /// therefore never re-check. Within-set ascending order is the writer's contract, assumed the way /// every merge assumes its sorted inputs. -/// -/// [`locate`](Self::locate) is the serving query: the node owning one tile cell, found by walking -/// the two-bit digits of the cell's key prefix from the root. #[derive(Debug)] pub(crate) struct QuadFile { map: HeaderMap, diff --git a/libs/@local/graph/atlas/src/file/quad/tests.rs b/libs/@local/graph/atlas/src/file/quad/tests.rs index 25be3d71e3e..8f9c194a6cd 100644 --- a/libs/@local/graph/atlas/src/file/quad/tests.rs +++ b/libs/@local/graph/atlas/src/file/quad/tests.rs @@ -1,3 +1,8 @@ +//! Certificates for the quad file's format. +//! +//! The tests pin the header's and node's wire layouts byte by byte, the type-set structural +//! rules, the region geometry, the writer-to-reader round trip, and every structural rule the +//! open enforces - with a property test holding the round trip over arbitrary trees. #![expect( clippy::little_endian_bytes, reason = "the wire-layout assertions pin the format's canonical little-endian bytes" @@ -55,10 +60,12 @@ fn fixture_nodes() -> Vec { ] } +/// The direct-type set of each fixture node, in node order. fn fixture_sets() -> TypeSets { TypeSets::from_sets(&[vec![1, 2, 5, 7], vec![1, 5], vec![1, 2, 7], vec![2]]) } +/// The fixture tree and its type sets, written out as a quad file's bytes. fn fixture_bytes() -> Vec { let mut bytes = Vec::new(); write_regions(&fixture_nodes(), &fixture_sets(), &mut bytes) @@ -66,6 +73,9 @@ fn fixture_bytes() -> Vec { bytes } +/// The header's bytes sit where the format's table says: magic, little-endian version 2, this +/// machine's information, the node count and type-id entry count as little-endian `u64`s, and zero +/// padding out to 4096. #[test] fn header_wire_layout() { let header = PaddedFileHeader::new(FileHeader::new(4, 10)); @@ -107,6 +117,9 @@ fn header_parse_pins_identity() { .expect_err("an unsupported version should not parse"); } +/// A node occupies 32 bytes in the order the format fixes - four child indexes, run start, run +/// length, subtree points - with an absent child stored as the `u32` sentinel. The accessors +/// read back the children, the run as a range, the point count, and leafness from those bytes. #[test] fn node_wire_layout() { let node = Node::new([Some(1), Some(2), None, Some(4)], 0x2A, 256, 1000); @@ -153,12 +166,17 @@ fn type_sets_reject_unsorted_sets() { drop(TypeSets::from_sets(&[vec![2, 1]])); } +/// A repeated id in one set panics at construction too: the order rule is strict, so equal +/// neighbours are as malformed as descending ones. #[test] #[should_panic(expected = "type set must ascend strictly")] fn type_sets_reject_duplicate_ids() { drop(TypeSets::from_sets(&[vec![3, 3]])); } +/// The header's region offsets and expected file length agree with the geometry computed by hand, +/// an empty tree still places its anchoring fencepost region, and counts whose arithmetic +/// overflows `u64` report no expected length because they match no real file. #[test] fn region_geometry() { // A 128-byte table for four nodes pads to one page, and the five posts pad to a second page. @@ -283,6 +301,9 @@ fn open_rejects_foreign_and_torn_bytes() { assert_matches!(QuadFile::open(&torn), Err(OpenQuadError::Length { .. })); } +/// The open validates the structural rules a traversal then relies on, and names the offender: +/// fenceposts that decrease or fail to close at the header's entry count report their index, and a +/// child index that points at its own node or past the table reports the node and the child slot. #[test] fn open_rejects_malformed_posts_and_children() { // The fixture's posts region starts at 8192. Post 1 raised beyond @@ -342,8 +363,6 @@ fn locate_reference(nodes: &[Node], cell: MortonCell) -> Option { } /// Every valid table and set cover roundtrips verbatim. -/// -/// The mapped locate agrees with the in-memory reference walk. #[property_test] fn written_tables_roundtrip( // Children generated strictly deeper, so construction preserves the pre-order rule. The format diff --git a/libs/@local/graph/atlas/src/file/region/header.rs b/libs/@local/graph/atlas/src/file/region/header.rs index e3518c0fe06..92aa275ab60 100644 --- a/libs/@local/graph/atlas/src/file/region/header.rs +++ b/libs/@local/graph/atlas/src/file/region/header.rs @@ -71,7 +71,6 @@ pub(crate) trait PaddedHeader: macro_rules! header { ($header:ident, $magic:ident, $version:expr) => { /// The full header page of this format. - /// #[doc = concat!("[`", stringify!($header), "`] followed by zero padding to the page")] /// boundary. This is the form that persists and that a mapped page parses as, and a /// dereference reaches the fields themselves. diff --git a/libs/@local/graph/atlas/src/file/region/tests.rs b/libs/@local/graph/atlas/src/file/region/tests.rs index 6e86e76d7f4..209b832f8cd 100644 --- a/libs/@local/graph/atlas/src/file/region/tests.rs +++ b/libs/@local/graph/atlas/src/file/region/tests.rs @@ -30,6 +30,8 @@ fn regions_pad_to_the_boundary() { assert!(bytes[5..].iter().all(|&byte| byte == 0)); } +/// A region that is already a whole page adds nothing, and closing a stream that sits on the +/// boundary adds nothing either - padding is never a full page of waste. #[test] fn aligned_regions_close_without_padding() { let mut bytes = Vec::new(); @@ -41,6 +43,8 @@ fn aligned_regions_close_without_padding() { assert_eq!(bytes.len() as u64, PAGE); } +/// Closing a streamed region pads from the byte count the caller reports out to the page +/// boundary, with zeros. #[test] fn streamed_regions_close_at_the_boundary() { let mut bytes = vec![7_u8; 10]; @@ -72,6 +76,9 @@ fn map_carves_regions() { fs::remove_file(&path).expect("the fixture file should remove"); } +/// A live mapping holds a shared lock: an exclusive lock on the same file contends while the +/// mapping lives and succeeds once it drops. This is what keeps a published file from being +/// rewritten under a reader. #[test] fn live_map_excludes_exclusive_lockers() { let path = std::env::temp_dir().join(format!( @@ -130,6 +137,8 @@ fn region_rejects_a_carve_beyond_the_mapping() { let _region = map.region(0, 4); } +/// A file shorter than one page maps, reports its true length, and has no header page - which is +/// how every format's open reports an undersized file instead of reading past the end. #[test] fn short_file_has_no_header_page() { let path = std::env::temp_dir().join(format!( diff --git a/libs/@local/graph/atlas/src/file/repository/mod.rs b/libs/@local/graph/atlas/src/file/repository/mod.rs index 687e5b6c38e..49cac14c8b4 100644 --- a/libs/@local/graph/atlas/src/file/repository/mod.rs +++ b/libs/@local/graph/atlas/src/file/repository/mod.rs @@ -48,7 +48,6 @@ impl core::error::Error for UnknownRepositoryVersion {} /// /// The entry names the file and carries the SHA-256 the publisher computed over its bytes. A /// verification hashes the file as it is on disk and compares. -// pub: rides `OpenAtlasError`'s public corruption variant. #[derive(Debug)] pub enum IntegrityVerificationError { /// The file's bytes hash to a digest other than the recorded one. @@ -194,6 +193,11 @@ impl FileName { &self.0 } + /// Answers whether `name` is usable as a file name within a generation directory. + /// + /// A usable name is nonempty and holds no `/` or NUL byte, and it does not begin with a dot. + /// The first two keep the name from addressing anything outside its directory, and the third + /// keeps it out of the hidden-file space that holds the generation's own bookkeeping. const fn valid(name: &str) -> bool { let bytes = name.as_bytes(); if bytes.is_empty() || bytes[0] == b'.' { @@ -254,10 +258,10 @@ impl AsRef for VerifiedUtf8PathBuf { } /// One published file, identified by its name within the repository and the SHA-256 of its bytes. -// pub: rides `IntegrityVerificationError`'s public checksum variant. #[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] pub struct RepositoryFile { pub name: FileName, + /// The SHA-256 the publisher computed over the file's bytes. pub hash: Sha256Digest, } diff --git a/libs/@local/graph/atlas/src/file/repository/tests.rs b/libs/@local/graph/atlas/src/file/repository/tests.rs index 3dca5dff5ec..d68711c5069 100644 --- a/libs/@local/graph/atlas/src/file/repository/tests.rs +++ b/libs/@local/graph/atlas/src/file/repository/tests.rs @@ -1,3 +1,5 @@ +//! Certificates for the repository layout version's serialized form. + use super::RepositoryVersion; #[test] diff --git a/libs/@local/graph/atlas/src/file/salt/metadata.rs b/libs/@local/graph/atlas/src/file/salt/metadata.rs index 18b12d6782f..b64106ae565 100644 --- a/libs/@local/graph/atlas/src/file/salt/metadata.rs +++ b/libs/@local/graph/atlas/src/file/salt/metadata.rs @@ -61,7 +61,7 @@ pub(crate) enum Placement { /// Where the generation's rank inputs came from. /// -/// The identity keeps the signals distinguishable wherever a reader consumes the ranking; it +/// The identity keeps the signals distinguishable wherever a reader consumes the ranking. It /// mirrors the configured [`RankingConfig`], recording what actually ran. #[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] #[serde(rename_all = "kebab-case")] @@ -445,7 +445,7 @@ pub(crate) struct LadderEvidence { /// The relation loss re-measured over the persisted aligned column. /// /// Guards the alignment application and the narrowing to `f32`. The reading is the corpus - /// total over every attraction instance, with no per-type cap; it does not store the capped + /// total over every attraction instance, with no per-type cap. It does not store the capped /// trained estimand ([`StepEvidence::capped_relation_loss`]). pub persisted_relation_loss: DNonNegative, /// The paired-movement readout beside the steps. @@ -472,7 +472,7 @@ pub(crate) struct StepEvidence { pub condition: NonNegative, /// The field's frozen relation loss at projection time. /// - /// The corpus total over every attraction instance, with no per-type cap; it does not store + /// The corpus total over every attraction instance, with no per-type cap. It does not store /// the capped trained estimand ([`Self::capped_relation_loss`]). pub relation_loss: DNonNegative, /// The capped trained estimand at this step. diff --git a/libs/@local/graph/atlas/src/file/salt/tests.rs b/libs/@local/graph/atlas/src/file/salt/tests.rs index e919c3c1e0b..b891d58e4a3 100644 --- a/libs/@local/graph/atlas/src/file/salt/tests.rs +++ b/libs/@local/graph/atlas/src/file/salt/tests.rs @@ -1,3 +1,5 @@ +//! Certificates for the SALT repository's document. + use core::num::NonZero; use hash_graph_temporal_versioning::{DecisionTime, Timestamp, TransactionTime}; @@ -52,16 +54,19 @@ use crate::{ }, }; +/// Derives a reproducible fixture digest from `seed`. fn digest(seed: &str) -> Sha256Digest { let mut hasher = Sha256::new(); hasher.update(seed.as_bytes()); hasher.finalize() } +/// A binding for artifact `A` whose digest comes from `seed` rather than from real bytes. fn binding(seed: &str) -> Binding { Binding::new(digest(seed)) } +/// Builds projector options with a short test schedule. fn placement() -> PlacementOptions { let mut options = ProjectorOptions::ratified(); options.schedule = TrainingSchedule::new( @@ -84,6 +89,7 @@ fn placement() -> PlacementOptions { PlacementOptions::Projector(options) } +/// A fit configuration exercising the non-default corners: a policy override and a fixed seed. fn config() -> FitConfig { FitConfig { seed: 0x5A17_F17D, @@ -107,6 +113,7 @@ fn config() -> FitConfig { } } +/// Every artifact slot bound, including the optional ones, each to the digest of its own name. fn files() -> SaltFiles { SaltFiles { representations: binding("representations.arr"), @@ -141,6 +148,7 @@ fn files() -> SaltFiles { } } +/// Builds the manifest fixture from its bound artifact roles and evidence. fn repository() -> SaltRepository { SaltRepository { version: RepositoryVersion::V2, @@ -172,6 +180,7 @@ fn repository() -> SaltRepository { } } +/// Builds level-of-detail readings with one middle bucket and the final bucket occupied. fn lod_measurements() -> LodMeasurements { LodMeasurements { world: Bounds2::new(Vec2::new(-4.0, -2.0), Vec2::new(8.0, 6.0)) @@ -188,6 +197,7 @@ fn lod_measurements() -> LodMeasurements { } } +/// A fitted classifier's evidence, the variant that carries every nested block. fn classifier_evidence() -> ClassifierEvidence { ClassifierEvidence::Fitted { corpus: digest("annotation-corpus.json"), @@ -285,6 +295,7 @@ fn calibration() -> ProximalCalibrationEvidence { } } +/// The evidence block a sealed generation carries, at corpus-sized readings. fn evidence() -> Evidence { Evidence { cards: CardEmbeddingStats { @@ -512,6 +523,10 @@ fn absent_verdicts_role_round_trips_as_explicit_null() { assert_eq!(decoded, repository); } +/// A baseline generation stays internally consistent on the wire: the unbound projector role +/// writes as an explicit null and the placement as `landmark-baseline`, and the document - role, +/// placement, configuration and evidence all agreeing that no projector ran - reads back +/// unchanged. #[test] fn baseline_generation_records_projector_absence_as_explicit_null() { let mut repository = repository(); @@ -535,6 +550,8 @@ fn baseline_generation_records_projector_absence_as_explicit_null() { assert_eq!(decoded, repository); } +/// Relation evidence without the multi-typed edge histogram refuses to decode, and the error names +/// the missing key rather than reporting the record as malformed. #[test] fn a_document_without_the_multi_typed_edge_histogram_refuses() { let mut document: serde_json::Value = @@ -817,8 +834,6 @@ fn tampered_configuration_echo_refuses_to_deserialize() { "/metadata/reproducibility/config/placement/projector/schedule/boundary", serde_json::json!(100), ), - // The semantic coefficient anchors the budget and must exceed - // zero. ( "/metadata/reproducibility/config/placement/projector/coefficients", serde_json::json!([0.0, 1.0, 1.0, 1.0, 0.0, 1.0]), @@ -1053,6 +1068,9 @@ fn a_decoded_document_carries_only_in_domain_readings() { } } +/// A version 2 document written before the proximal calibration and the per-type losses existed +/// decodes with the calibration absent and the loss list empty, rather than refusing or reading +/// either as a measured zero. #[test] fn an_old_document_without_the_optional_keys_decodes_as_absent() { // Decode-from-old means a published-shape repository-version-2 document written before a @@ -1093,9 +1111,6 @@ fn an_old_document_without_the_optional_keys_decodes_as_absent() { #[test] fn a_missing_required_sibling_refuses_and_names_the_field() { - // The control proves the optional keys' own absence rule decodes the old document rather - // than record-wide permissiveness. Removing an undefaulted required sibling must refuse, - // naming the field. let repository = repository(); let mut json = serde_json::to_value(&repository).expect("the repository should serialize"); @@ -1115,6 +1130,9 @@ fn a_missing_required_sibling_refuses_and_names_the_field() { ); } +/// A ladder written before the paired-movement readout existed decodes with the readout absent, +/// not as a vacuous or failed body: nothing measured it, which is a different record from a +/// measurement that found nothing. #[test] fn an_old_ladder_without_the_paired_movement_key_decodes_as_absent() { // A published-shape repository-version-2 ladder written before the readout existed @@ -1148,9 +1166,6 @@ fn an_old_ladder_without_the_paired_movement_key_decodes_as_absent() { #[test] fn a_ladder_missing_a_required_sibling_refuses_and_names_the_field() { - // The control for the ladder record: removing an undefaulted required sibling refuses, - // so the readout key's decode rests on its own absence rule rather than record-wide - // permissiveness. let repository = repository(); let mut json = serde_json::to_value(&repository).expect("the repository should serialize"); diff --git a/libs/@local/graph/atlas/src/file/sprs/mod.rs b/libs/@local/graph/atlas/src/file/sprs/mod.rs index 15a083f882e..7615a63b792 100644 --- a/libs/@local/graph/atlas/src/file/sprs/mod.rs +++ b/libs/@local/graph/atlas/src/file/sprs/mod.rs @@ -79,8 +79,10 @@ use crate::file::region::{ padded_size, }; -// The single variant makes the derive validate the discriminant, so parsing admits exactly the -// pinned magic value. +/// The discriminant carrier behind [`FileHeaderMagic`]. +/// +/// Parsing admits exactly the pinned magic value because the derive validates the single +/// variant's discriminant. #[derive( Debug, Copy, @@ -295,9 +297,11 @@ pub(crate) trait SprsIndex: SpIndex + FromBytes + IntoBytes + Immutable { const VARIANT: IndexVariant; } -// One-line impls over every fixed-width scalar: enough expansions that -// drift between hand-written copies is the likelier bug. The assert -// keeps each scalar tag's pinned width equal to the type's real width. +/// Implements [`SprsValue`] for each `type => tag` pair, asserting the widths agree. +/// +/// One-line impls over every fixed-width scalar: enough expansions that +/// drift between hand-written copies is the likelier bug. The assert +/// keeps each scalar tag's pinned width equal to the type's real width. macro_rules! sprs_value { ($($element:ty => $variant:ident,)*) => { $( @@ -312,6 +316,7 @@ macro_rules! sprs_value { }; } +/// Implements [`SprsIndex`] for each `type => variant` pair, asserting the widths agree. macro_rules! sprs_index { ($($element:ty => $variant:ident,)*) => { $( @@ -366,8 +371,10 @@ sprs_index! { i64 => I64, } -// The scalar tags mirror ArrayVariant's discriminants, so the two formats speak one scalar -// vocabulary. The asserts below keep the two sets of discriminants equal. +/// Asserts that each named tag has the same discriminant in both formats. +/// +/// The scalar tags use [`ArrayVariant`]'s discriminants. The assertions below keep the two sets +/// equal. macro_rules! tag_mirrors_variant { ($($variant:ident,)*) => { const _: () = { diff --git a/libs/@local/graph/atlas/src/file/sprs/read.rs b/libs/@local/graph/atlas/src/file/sprs/read.rs index dd1b8e4aab4..9275151adc1 100644 --- a/libs/@local/graph/atlas/src/file/sprs/read.rs +++ b/libs/@local/graph/atlas/src/file/sprs/read.rs @@ -14,7 +14,6 @@ use crate::file::region::{ }; /// Opening a sparse matrix file failed. -// pub: rides `OpenAtlasError`'s public adjacency variant. #[derive(Debug)] pub enum OpenSprsError { /// Reading the header page failed. @@ -72,14 +71,17 @@ impl Error for OpenSprsError { } /// Viewing an opened file's matrix failed. -// pub: rides `InvalidAdjacencyFile`'s public matrix variant. #[derive(Debug)] pub enum SprsMatrixError { /// The file stores different element types than the requested ones. Elements { + /// The value type the file stores. value: ValueTag, + /// The width in bytes of the value type the file stores. value_width: u64, + /// The index type the file stores for its entry coordinates. index: IndexVariant, + /// The index type the file stores for its pointer region. iptr: IndexVariant, }, /// The matrix does not fit the address space. diff --git a/libs/@local/graph/atlas/src/file/sprs/tests.rs b/libs/@local/graph/atlas/src/file/sprs/tests.rs index 6119d3886dd..52551cd453d 100644 --- a/libs/@local/graph/atlas/src/file/sprs/tests.rs +++ b/libs/@local/graph/atlas/src/file/sprs/tests.rs @@ -1,3 +1,5 @@ +//! Certificates for the sparse matrix file's format. + use core::assert_matches; use sprs::CsMatI; @@ -9,11 +11,13 @@ use super::{ write::{WriteSprsError, write_matrix}, }; +/// A two-dimensional shape, the only rank a sparse matrix file records. fn matrix_shape(rows: u64, columns: u64) -> ArrayShape { ArrayShape::new(&[Dim::new(rows), Dim::new(columns)]) .expect("two dimensions fit the maximum shape rank") } +/// A 3x3 CSR header over `f32` values with `nnz` stored entries. fn fixture_header(nnz: u64) -> FileHeader { FileHeader::new( ValueTag::F32, @@ -36,6 +40,8 @@ fn fixture() -> CsMatI { ) } +/// Each region offset and the expected file length follow the layout equations, with every region +/// padded to a page boundary and the value region ending the file unpadded. #[test] fn regions_follow_the_layout_equations() { // 4 x 4, 8 entries: 40 pointer bytes and 32 index bytes each pad @@ -56,8 +62,8 @@ fn regions_follow_the_layout_equations() { #[test] fn compressed_dimension_spans_the_pointers() { - // 2 x 1023: row-compressed needs 3 pointers, column-compressed - // 1024 - exactly two pages of u64 pointers. + // A CSR file has rows + 1 pointers, and a CSC file has columns + 1. For this 2 x 1023 + // shape, those counts are 3 and 1024. The 1024 u64 pointers occupy exactly two pages. let csr = FileHeader::new( ValueTag::F32, 4, @@ -82,6 +88,8 @@ fn compressed_dimension_spans_the_pointers() { assert_eq!(csc.indices_offset(), Some(4096 + 8192)); } +/// Narrower pointer and index types shrink their regions: the byte counts follow the recorded +/// element widths, and a region that fills less than a page still pads to one. #[test] fn narrow_elements_shrink_the_regions() { // 1023 rows of u16 pointers make 2048 pointer bytes, exactly half a page, which still pads to @@ -482,6 +490,8 @@ fn structure_only_matrix_reopens_with_conjured_units() { mod miri { use crate::file::sprs::SprsValue; + /// The unit value view materializes the requested number of elements from an empty byte + /// region, which is the case whose pointer construction Miri is here to check. #[test] fn unit_values_conjure_from_no_bytes() { let units = <() as SprsValue>::view_region(&[], 7) diff --git a/libs/@local/graph/atlas/src/identity/column.rs b/libs/@local/graph/atlas/src/identity/column.rs index 6cda9933133..31131b1d6ee 100644 --- a/libs/@local/graph/atlas/src/identity/column.rs +++ b/libs/@local/graph/atlas/src/identity/column.rs @@ -9,14 +9,16 @@ use crate::file::array::{ArrayFile, ColumnScalar}; /// One array artifact proven to hold elements of type `T`, indexed by the id domain `I`. /// -/// Construction validates the recorded element stamp once through [`ArrayFile::column`], so views -/// are infallible for the value's lifetime: the file is immutable after open and the shape cannot -/// change under it. The index domain is the column's position vocabulary, the id a caller must -/// hold to read an element. A call site therefore cannot mix a column over one domain with a -/// column over another. +/// Construction validates the recorded element stamp once through [`ArrayFile::column`]. Views +/// are then infallible for the value's lifetime: the file is immutable after open and the shape +/// cannot change under it. The index domain is the column's position vocabulary, the id a caller +/// must hold to read an element. A call site cannot mix a column over one domain with a column +/// over another. #[derive(Debug)] pub(crate) struct Column { + /// The opened array, its element stamp proven to be `T`'s. file: ArrayFile, + /// The index domain and element type, carried without owning either. domain: PhantomData T>, } diff --git a/libs/@local/graph/atlas/src/identity/key.rs b/libs/@local/graph/atlas/src/identity/key.rs index 49201eee792..fe8ce18c9ab 100644 --- a/libs/@local/graph/atlas/src/identity/key.rs +++ b/libs/@local/graph/atlas/src/identity/key.rs @@ -7,7 +7,9 @@ hashql_core::id::newtype! { /// `o` names the `o`-th point of that total order. Ordinals are dense and zero-based over the /// generation's points. /// - /// The key order and the base order are different permutations of the same points. A bucket-major cut reads the base order, while a prefix scan over spatial keys reads this one. Converting between them goes through the generation's key-order columns, never by reinterpreting the integer. + /// The key order and the [`BasePosition`] order are different permutations of the same points. + /// A bucket-major cut reads the base order, while a prefix scan over spatial keys reads this one. + /// Converting between them goes through the generation's key-order columns, never by reinterpreting the integer. /// /// [`BasePosition`]: crate::identity::BasePosition #[id(derive(Step), const)] diff --git a/libs/@local/graph/atlas/src/identity/mod.rs b/libs/@local/graph/atlas/src/identity/mod.rs index fa62052d8db..96f9458c054 100644 --- a/libs/@local/graph/atlas/src/identity/mod.rs +++ b/libs/@local/graph/atlas/src/identity/mod.rs @@ -4,25 +4,30 @@ //! zero-based domains - node rows, edge rows, ontology-type rows, annotation card rows, base //! positions, and key ordinals - and a bare integer names none of them. This module carries the id //! types that keep those domains distinct in signatures and [`Column`], the element-typed view over -//! one array artifact. The ids share the [`hashql_core::id::Id`] contract; every id is a dense -//! zero-based index, so conversions are total within the id encoding and the arithmetic never -//! wraps. Content identity - which entity or type a row is - lives with the dataset and the -//! identity tables; an id here names an index in one generation's streams, valid only against the +//! one array artifact. The ids share the [`hashql_core::id::Id`] contract. Every id is a dense +//! zero-based index. Conversions out of an id are casts, and a target integer narrower than the +//! domain keeps the low bits, which [`NodeRowId`]'s own conversion states. The stepping helpers +//! compute in `usize` and rebuild the id from the result. A value outside the domain panics, and +//! a `usize` overflow panics only where overflow checks are on. Content identity - which entity or +//! type a row is - lives with the dataset and the +//! identity tables. An id here names an index in one generation's streams, valid only against the //! generation that assigned it. //! //! Rows and orders are different domains over the same points. A row names a stream entry, while a //! base position and a key ordinal name slots in two permutations of it. The generation's own //! columns convert between them. //! -//! Row ids persist: the in-memory form is the little-endian byte form, so a column of these ids -//! writes to and reads back from artifact files without conversion. +//! Row ids persist: the in-memory form is the little-endian byte form. A column of these ids writes +//! to and reads back from artifact files without conversion. pub(crate) use self::{ card::CardRow, column::Column, edge::EdgeRowId, node::NodeRowId, ontology::OntologyRowId, position::BasePosition, rank::ImportanceRank, }; -/// Bench-only exports: the key-ordinal domain crosses typed into lod's bench module. +/// The key-ordinal domain, re-exported under the `bench` feature. +/// +/// Benchmark code outside this module then indexes by the id type rather than a bare integer. #[cfg(feature = "bench")] pub(crate) mod bench { pub(crate) use super::key::KeyOrdinal; diff --git a/libs/@local/graph/atlas/src/identity/node.rs b/libs/@local/graph/atlas/src/identity/node.rs index 14ed4fc4b3a..cdb35a1bee6 100644 --- a/libs/@local/graph/atlas/src/identity/node.rs +++ b/libs/@local/graph/atlas/src/identity/node.rs @@ -40,6 +40,11 @@ impl From for u64 { } impl From for usize { + /// Converts the row to a platform-sized index. + /// + /// # Warning + /// + /// On targets narrower than 64 bits, the conversion keeps only the low `usize::BITS` bits. #[inline] fn from(id: NodeRowId) -> Self { id.as_usize() diff --git a/libs/@local/graph/atlas/src/integrity/hash.rs b/libs/@local/graph/atlas/src/integrity/hash.rs index c2d7676af4a..b7adbb3ac7e 100644 --- a/libs/@local/graph/atlas/src/integrity/hash.rs +++ b/libs/@local/graph/atlas/src/integrity/hash.rs @@ -9,6 +9,7 @@ use super::{ writer::Update, }; +/// The digest width in bytes, read off the hash itself rather than written down as 32. const DIGEST_BYTES: usize = ::OutputSize::USIZE; /// A SHA-256 content identity. @@ -18,8 +19,7 @@ const DIGEST_BYTES: usize = ::Outp /// recompute and compare them to detect substitution or corruption. /// /// The text and JSON form is 64 characters of canonical lowercase hexadecimal. Parsing rejects -/// uppercase digits and noncanonical lengths, so a digest that round-trips through text is -/// byte-identical. +/// uppercase digits and noncanonical lengths. #[derive( Debug, Copy, @@ -43,7 +43,7 @@ const DIGEST_BYTES: usize = ::Outp #[repr(transparent)] pub struct Sha256Digest(HexBytes); -// No multi-byte fields: the digest is a byte array, so no byte order arises. +// byte arrays have identical representations on little- and big-endian targets. crate::dataset::offline::portable::self_archived!(Sha256Digest); impl Sha256Digest { @@ -53,8 +53,7 @@ impl Sha256Digest { /// Adopts `bytes` as a digest without computing anything. /// /// The caller asserts that `bytes` came out of a SHA-256 computation over the content this - /// value names. This constructor cannot verify that. Use this to restore digests from storage - /// formats that persist raw bytes rather than hexadecimal text. + /// value names. This constructor cannot verify that. #[must_use] #[inline] pub const fn from_bytes_unchecked(bytes: [u8; DIGEST_BYTES]) -> Self { @@ -139,6 +138,10 @@ impl Update for Sha256 { } } +/// Certificates for the digest's value and its text and JSON forms. +/// +/// The published SHA-256 vectors pin the hasher against the standard rather than against itself, +/// and the round-trip tests pin the canonical text form, including what it refuses. #[cfg(test)] mod tests { use core::assert_matches; @@ -146,7 +149,9 @@ mod tests { use super::{Sha256, Sha256Digest}; use crate::integrity::{Update as _, hex::ParseHexError}; + /// The published SHA-256 of `b"abc"`. const ABC_DIGEST: &str = "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"; + /// The published SHA-256 of the empty input. const EMPTY_DIGEST: &str = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; #[test] diff --git a/libs/@local/graph/atlas/src/integrity/hex.rs b/libs/@local/graph/atlas/src/integrity/hex.rs index df593d593a7..10b1b960c2e 100644 --- a/libs/@local/graph/atlas/src/integrity/hex.rs +++ b/libs/@local/graph/atlas/src/integrity/hex.rs @@ -67,9 +67,9 @@ const fn nibble(byte: u8) -> (u8, u8) { /// A fixed-width byte string with a canonical lowercase hexadecimal text form. /// /// Every fixed-width integrity value (digests, signatures, public keys) is a newtype over this: the -/// text, JSON, and [`fmt::Debug`] forms are `2 · N` lowercase hexadecimal characters, and parsing -/// is the strict inverse. A string that parses is the unique encoding of its value, so text -/// round-trips are byte-identical. +/// text, JSON, and [`fmt::Debug`] forms are `2 · N` lowercase hexadecimal characters. Parsing is +/// the strict inverse. Text round-trips are byte-identical. Every `N`-byte pattern is a valid +/// value. Equality compares the bytes without a constant-time guarantee. #[derive( Copy, Clone, @@ -195,6 +195,7 @@ impl<'de, const N: usize> serde::Deserialize<'de> for HexBytes { where D: serde::Deserializer<'de>, { + /// A canonical hexadecimal string decoder for [`HexBytes`]. struct HexVisitor; impl serde::de::Visitor<'_> for HexVisitor { @@ -229,12 +230,18 @@ impl schemars::JsonSchema for HexBytes { } } +/// Certificates for the canonical hexadecimal encoding. +/// +/// One test fixes what the encoding accepts and the forms it prints. The other fixes what it +/// refuses, with the offset and byte the error reports. #[cfg(test)] mod tests { use core::assert_matches; use super::{HexBytes, ParseHexError}; + /// Canonical text parses to the bytes it names, and prints back as the same text in both the + /// display and debug forms - the debug form quoted. #[test] fn round_trip() { let value: HexBytes<4> = "00ff10ab" diff --git a/libs/@local/graph/atlas/src/integrity/secret.rs b/libs/@local/graph/atlas/src/integrity/secret.rs index a6e2d22964a..379d444f1b7 100644 --- a/libs/@local/graph/atlas/src/integrity/secret.rs +++ b/libs/@local/graph/atlas/src/integrity/secret.rs @@ -1,3 +1,16 @@ +//! Secrets that redact themselves and zero their buffers. +//! +//! [`SecretString`] accepts arbitrary strings and compares their bytes in constant time, while +//! [`PasswordString`] trims its input and rejects an empty result. Their clones share guarded +//! allocations, which zeroize when the last guarded owner drops. +//! [`SecretString::into_unguarded`] returns an allocation without that guarantee. Both types' +//! [`fmt::Debug`] output reveals the byte length, and [`SecretString`]'s [`fmt::Display`] output is +//! a fixed redaction. +//! +//! [`SecretHexBytes`] accepts a fixed-width key in canonical lowercase hexadecimal and compares in +//! constant time. It also redacts its own display forms and zeroizes its buffer on drop. None of +//! these types implements `Serialize`. Their exposure methods reveal the secret bytes. + use alloc::{alloc::Allocator, sync::Arc}; use core::{error::Error, fmt, marker::PhantomData, mem::MaybeUninit, str::FromStr}; use std::alloc::Global; @@ -10,11 +23,11 @@ use super::{ParseHexError, hex::HexBytes}; /// A variable-length secret string. /// -/// The bytes zero on drop, and no path encodes them back out. [`fmt::Debug`] prints the length -/// alone, [`fmt::Display`] prints a fixed placeholder, and the type has no `Serialize`, so logging -/// or dumping a held secret discloses nothing. [`expose`](Self::expose) hands the buffer onward -/// under a guard that zeroizes it in turn, and [`into_unguarded`](Self::into_unguarded) is the one -/// exit that ends zeroizing custody. +/// The guarded allocation zeroizes when its last guarded owner drops, and equality compares its +/// bytes in constant time. [`fmt::Debug`] reveals the byte length, [`fmt::Display`] prints a fixed +/// placeholder, and the type has no `Serialize`. [`expose`](Self::expose) returns the shared +/// allocation under its zeroizing guard. [`into_unguarded`](Self::into_unguarded) returns an +/// allocation without that guarantee. /// /// The zeroing covers buffers this type and its guard own, never copies made from them. A consumer /// that copies the exposed value into its own storage owns that copy's end of life. The value also @@ -190,8 +203,8 @@ impl Error for EmptyPasswordError {} /// A non-empty, trimmed password. /// /// Parsing trims surrounding whitespace and refuses an input that is empty afterwards. The -/// bytes zero on drop, and [`fmt::Debug`] prints the length alone, so logging a held password -/// discloses nothing. +/// guarded allocation zeroizes when its last guarded owner drops. [`fmt::Debug`] reveals the +/// trimmed byte length, and the type has no `Serialize`. #[derive(Clone)] pub struct PasswordString(SecretString); @@ -225,13 +238,13 @@ impl FromStr for PasswordString { /// An `N`-byte secret configured in the canonical hexadecimal encoding. /// -/// The bytes zero on drop, and this type's own renderings redact: [`fmt::Debug`] prints the width -/// alone and [`fmt::Display`] prints a fixed placeholder. The type has no `Serialize`, so logging -/// or serializing the value discloses nothing. +/// Its buffer zeroizes on drop, and equality compares its bytes in constant time. This type's +/// [`fmt::Debug`] reveals the fixed width, while [`fmt::Display`] prints a fixed placeholder. The +/// type has no `Serialize`. /// -/// The redaction covers this value, not everything reachable through it: [`HexBytes`] renders every -/// byte, so rendering the dereferenced inner value writes the key in full. Code that holds a secret -/// renders the secret, never its target. +/// Redaction covers only this type's formatting implementations. [`Self::as_bytes`], [`AsRef`], +/// and [`AsMut`] expose the complete secret bytes to their callers. Every `N`-byte pattern is a +/// valid value. /// /// Parsing and deserialization accept exactly the canonical lowercase form. #[derive(Clone, zerocopy::ByteHash, zerocopy::Immutable, zerocopy::KnownLayout)] @@ -321,6 +334,10 @@ impl<'de, const N: usize> serde::Deserialize<'de> for SecretHexBytes { where D: serde::Deserializer<'de>, { + /// A canonical hexadecimal string decoder for [`SecretHexBytes`]. + /// + /// The decoded buffer zeroes on drop. Deserialization leaves the borrowed input text + /// unchanged. struct SecretHexVisitor; impl serde::de::Visitor<'_> for SecretHexVisitor { @@ -393,6 +410,7 @@ impl Clone for SecretHexBytesValueParser { } } +/// Certificates for the redaction, refusal and custody rules. #[cfg(test)] mod tests { use core::{assert_matches, str::FromStr as _}; diff --git a/libs/@local/graph/atlas/src/integrity/writer.rs b/libs/@local/graph/atlas/src/integrity/writer.rs index 1cdfd1458ef..c6ed7c2ed85 100644 --- a/libs/@local/graph/atlas/src/integrity/writer.rs +++ b/libs/@local/graph/atlas/src/integrity/writer.rs @@ -160,6 +160,7 @@ where } } +/// Certificates for the accumulator adapters. #[cfg(test)] mod tests { use core::{pin::Pin, task}; @@ -186,6 +187,7 @@ mod tests { } impl ShortWriter { + /// A writer that has taken nothing yet and accepts `limit` bytes a call. fn new(limit: usize) -> Self { Self { accepted: Vec::new(), @@ -193,6 +195,7 @@ mod tests { } } + /// Records the prefix of `buf` this writer accepts and returns its length. fn take(&mut self, buf: &[u8]) -> usize { let accepted = buf.len().min(self.limit); self.accepted.extend_from_slice(&buf[..accepted]); @@ -301,6 +304,8 @@ mod tests { assert_eq!(writer.accumulator.0, b"abcde"); } + /// The same short-write accounting holds asynchronously: the accumulated stream equals what + /// the writer accepted across the retried polls, and equals the whole input at the end. #[tokio::test] async fn async_write_short() { let mut writer = Writer { diff --git a/libs/@local/graph/atlas/src/lib.rs b/libs/@local/graph/atlas/src/lib.rs index 797c71892a1..40e9554adeb 100644 --- a/libs/@local/graph/atlas/src/lib.rs +++ b/libs/@local/graph/atlas/src/lib.rs @@ -1,8 +1,33 @@ -//! # HASH Graph Atlas +//! Fits 2D maps of HASH Graph entities from their embeddings and relationships. //! -//! Fits 2D maps over the entity embeddings stored in the HASH Graph, blending semantic similarity -//! (what entities mean) with relational structure (how they connect), and distills each map into a -//! small encoder that places new entities on an existing map without refitting. +//! Fitting blends semantic similarity (what entities mean) with relational structure (how they +//! connect). Each fit distills the map into a small encoder that places new entities on the +//! existing map without refitting. +//! +//! For the HTTP request and response contracts, start with `api`. The graph binary serves the +//! interactive API reference at `/v1/atlas/openapi`. +//! +//! This HTTP sketch requires a running deployment with a published generation and valid actor +//! credentials. Replace `{generation}` with the `generation` field returned by the first response: +//! +//! ```text +//! GET /v1/atlas/current +//! -> 200: JSON containing the generation ID +//! POST /v1/atlas/generation/{generation}/manifest (empty body) +//! -> 200: JSON manifest and an Atlas-Authority response header +//! ``` +//! +//! An empty manifest body requests an unfiltered view of the actor's permitted entities. Present +//! the returned `Atlas-Authority` token on data requests. The manifest lists the variants and +//! request limits. +//! +//! - [Modules](#modules) +//! - [Using the crate](#using-the-crate) +//! - [Crate features](#crate-features) +//! - [Performance](#performance) +//! - [Limitations](#limitations) +//! +//! # Modules //! //! The crate builds the SALT pipeline on top of a foundation of domain-independent modules: //! @@ -27,44 +52,35 @@ //! //! # Using the crate //! -//! A caller outside this crate reaches a published generation over HTTP. [`cli`] carries the -//! operator commands that fit a generation over the live store and serve the active one through the -//! `api` router the graph binary hosts. The Rust items behind that router are crate-internal by -//! design. `serve::Atlas` carries the worked example for the read path. +//! Use [`cli`] for the operator commands that fit a generation over the live store and serve the +//! active one through the graph binary. The Rust items behind the `api` router are crate-internal +//! by design. //! //! # Crate features //! //! [`device::PinnedDevice`] selects CPU, CUDA, or Metal at runtime. CPU dispatches to `NdArray`, -//! while CUDA and Metal dispatch to `CubeCL`. Cargo features expose tools around that runtime: +//! while CUDA and Metal dispatch to `CubeCL`. Cargo features expose tools around that runtime. Both +//! `bench` and `cli` are disabled by default: //! //! - `bench` exposes `bench`, the measurement hooks the five `[[bench]]` targets in `Cargo.toml` -//! consume. The lab instruments the standalone binary runs stay outside it and build with the -//! crate regardless. +//! consume. The standalone binary's lab commands build independently of this feature. //! - `cli` compiles in the standalone `hash-graph-atlas` binary's shell and its exclusive //! dependencies, `ratatui`'s dashboard and `tracing-subscriber`'s log formatting. The operator -//! commands and the read-API routes build unconditionally, so the `hash-graph` binary consumes -//! them feature-free. +//! commands and the read-API routes build unconditionally. The `hash-graph` binary consumes them +//! feature-free. //! //! # Performance //! -//! Opening a generation validates every artifact once. -//! -//! `serve::Atlas::open` maps and validates every serving artifact and their cross-artifact -//! agreement a single time, so every read after that is an mmap gather and a wire encode, never a -//! decode. Every published artifact is a plain file mapped whole by `mmap`, so serving cost after -//! open is page-cache and address-space bound rather than parse bound. An opened `serve::Atlas` -//! is `Send + Sync` and immutable, so a caller can keep one in an `Arc` across requests for the -//! process lifetime of the generation. Reads are synchronous and CPU-bound over mapped memory, so -//! an async transport schedules them on a compute pool rather than inline on its own runtime -//! threads. +//! Generation maintenance maps and validates serving artifacts before publication. Requests reuse +//! those mappings. Response assembly runs on Rayon, including synchronous store calls for detail +//! hydration. //! //! # Limitations //! //! Serving and fitting never combine implicitly. //! -//! `cli::ServeCommand` opens an already-published generation and never fits one. An empty or -//! unfitted root fails the open with a named `cli::ServeError::Missing` rather than fitting on -//! demand. +//! `cli::ServeCommand` never fits a generation. Its maintenance task opens published artifacts +//! and retries failures. The current-generation endpoint answers 503 before initial publication. //! //! ## Workspace dependencies #![cfg_attr(doc, doc = simple_mermaid::mermaid!("../docs/dependency-diagram.mmd"))] diff --git a/libs/@local/graph/atlas/src/main.rs b/libs/@local/graph/atlas/src/main.rs index 48328fa1791..72026957c75 100644 --- a/libs/@local/graph/atlas/src/main.rs +++ b/libs/@local/graph/atlas/src/main.rs @@ -1,7 +1,6 @@ -//! The standalone atlas operator binary. -//! -//! Everything lives in [`hash_graph_atlas::cli`]. This shell only redirects into it. +//! The standalone Atlas command-line application. +/// Runs an Atlas command and returns its process exit status. fn main() -> std::process::ExitCode { hash_graph_atlas::cli::main() } diff --git a/libs/@local/graph/atlas/src/math/affinity/fit.rs b/libs/@local/graph/atlas/src/math/affinity/fit.rs index f2538309993..f9560f59360 100644 --- a/libs/@local/graph/atlas/src/math/affinity/fit.rs +++ b/libs/@local/graph/atlas/src/math/affinity/fit.rs @@ -1,38 +1,46 @@ //! Least-squares fitting of the affinity curve to a membership falloff. //! -//! The fit is a two-parameter Levenberg-Marquardt loop whose normal equations are a symmetric 2x2 -//! system solved in closed form. The loop needs no matrix library and allocates nothing on any -//! path. +//! For spread σ > 0 and minimum distance δ ∈ (0, σ], the target is h(d) = 1 for 0 ≤ d < δ and h(d) +//! = exp(−(d − δ)/σ) otherwise. A grid of m ≥ 8 distances uses dᵢ = i · rσ/(m − 1) for i from 0 +//! through m − 1, where r > 0 is the range multiplier. The objective is Σᵢ(q(dᵢ) − h(dᵢ))², with +//! q(d) = 1 / (1 + a · d^(2b)). +//! +//! A two-parameter Levenberg-Marquardt iteration solves a symmetric damped 2x2 system in closed +//! form. It uses `f64` arithmetic and constant storage without a matrix library. Relative step and +//! cost-improvement thresholds terminate the search heuristically. Neither threshold certifies a +//! stationary point or a global minimum, and grid refinement can change the sampled optimum. use core::num::NonZero; use super::AffinityCurve; use crate::math::{DNonNegative, DPositive, Positive, positive, scalar::narrow_f32}; -/// The sample grid a fit runs over. +/// Sample count and distance range for the least-squares target. /// /// [`AffinityCurve::fit`] uses the default grid. [`AffinityCurve::fit_with`] accepts a custom one. -/// The grid determines which distances vote in the least-squares balance between the fitted curve -/// and the target falloff: its resolution around the membership breakpoint and how far into the -/// tail it reaches. +/// Sample spacing determines resolution near the membership breakpoint, while the range determines +/// how much of the tail contributes to the objective. +/// +/// # Example /// -/// # Examples +/// This example is ignored because the fitting API and configuration are crate-private. /// /// ```ignore +/// use crate::math::{AffinityCurve, affinity::fit::AffinityFitConfig, positive}; +/// /// let default = AffinityCurve::fit(positive!(1.0), positive!(0.1)) /// .expect("reference inputs are well-conditioned"); /// let fine = AffinityCurve::fit_with( -/// 1.0, -/// 0.1, +/// positive!(1.0), +/// positive!(0.1), /// AffinityFitConfig { /// samples: 600, /// ..AffinityFitConfig::default() /// }, /// ) -/// .expect("a finer grid stays well-conditioned"); +/// .expect("the solver accepts this grid and target"); /// -/// // Refining the grid moves the fit only slightly: the parameters are -/// // stable under discretization. +/// // Compare the grid refinement on this particular target. /// assert!((default.a() - fine.a()).abs() < 0.01); /// ``` #[derive(Debug, Copy, Clone, Default)] @@ -40,42 +48,42 @@ pub(crate) struct AffinityFitConfig { /// Number of evenly spaced sample distances for the least-squares target. /// /// More samples resolve the target falloff more finely, in particular around the - /// `minimum_distance` breakpoint, at proportionally more work per solver pass. At least - /// [`MIN_SAMPLES`](Self::MIN_SAMPLES). + /// `minimum_distance` breakpoint, at proportionally more work per solver pass. The fit requires at least [`MIN_SAMPLES`](Self::MIN_SAMPLES). The grid has 300 samples by default. pub samples: u16 = 300, /// The sampled range extends this many spreads from zero. /// - /// A wider range weights the tail of the falloff more: far samples gain votes in the - /// least-squares balance, sharpening the fitted tail exponent `b` at the cost of fidelity near - /// the breakpoint. Finite and strictly positive. + /// Finite and strictly positive, with a value of 3 by default. Increasing the range at a fixed sample count coarsens the spacing and includes more of the tail. It can change the fit without any guaranteed direction of change in b. A range ending below the membership breakpoint samples only the flat plateau. pub range_in_spreads: Positive = positive!(3.0), } impl AffinityFitConfig { /// Fewest samples [`AffinityCurve::fit_with`] accepts. /// - /// The two-parameter fit needs the grid to populate both regimes of the piecewise target (the - /// flat membership plateau inside `minimum_distance` and the exponential tail beyond it) with a - /// handful of points each. Below eight samples the grid underdetermines the fit against the - /// target the curve traces. + /// The configured fitting entry point requires eight samples, a threshold that by itself + /// neither establishes a well-conditioned system nor ensures that both target regimes are + /// sampled. pub(crate) const MIN_SAMPLES: NonZero = NonZero::new(8).unwrap(); } impl AffinityCurve { /// Fits a curve from the desired membership falloff. /// - /// The falloff keeps membership at `1` inside `minimum_distance` and decays as `exp(-(d - - /// minimum_distance) / spread)` beyond it. This method samples the falloff on the crate's - /// default grid (300 evenly spaced distances over `[0, 3 · spread]`) and fits the curve to the - /// samples by Levenberg-Marquardt least squares. The fit runs once at initialization in double - /// precision and narrows the result to the working `f32` parameters. + /// This uses the [membership-falloff model](crate::math::affinity::fit) with 300 evenly spaced + /// distances over [0, 3σ], where σ is `spread`. Levenberg-Marquardt iteration estimates the + /// parameters in `f64` before narrowing to `f32`. + /// + /// Returns [`None`] when `minimum_distance` exceeds `spread`, when the solver rejects an + /// evaluation or exhausts its retry/iteration limits, or when the final parameters do not + /// narrow to finite positive `f32` values. Successful termination follows relative step or + /// cost-improvement thresholds, without certifying an optimum. /// - /// Returns [`None`] when `minimum_distance` exceeds `spread`, or when the least-squares fit - /// fails to converge to parameters that [`new`](Self::new) accepts. + /// # Example /// - /// # Examples + /// This example is ignored because [`AffinityCurve`] is crate-private. /// /// ```ignore + /// use crate::math::{AffinityCurve, positive}; + /// /// // The reference inputs: spread 1.0, minimum distance 0.1. /// let curve = AffinityCurve::fit(positive!(1.0), positive!(0.1)) /// .expect("reference inputs are well-conditioned"); @@ -90,16 +98,14 @@ impl AffinityCurve { /// Fits a curve from the desired membership falloff over a configured sample grid. /// - /// The falloff keeps membership at `1` inside `minimum_distance` and decays as `exp(-(d - - /// minimum_distance) / spread)` beyond it. This method samples it at - /// [`samples`](AffinityFitConfig::samples) evenly spaced distances over `[0, range_in_spreads * - /// spread]` and fits the curve to the samples by Levenberg-Marquardt least squares, in double - /// precision, narrowing the result to the working `f32` parameters. [`fit`](Self::fit) - /// delegates here with the default grid. + /// This uses [`fit`](Self::fit)'s target and stopping criteria with + /// [`config`](AffinityFitConfig)'s sample count and range. The grid covers [0, rσ], where r is + /// [`range_in_spreads`](AffinityFitConfig::range_in_spreads) and σ is `spread`. Each solver + /// evaluation takes O(m) time for m samples, with constant additional storage. /// - /// Returns [`None`] when `minimum_distance` exceeds `spread`, when `config` holds fewer than - /// [`MIN_SAMPLES`](AffinityFitConfig::MIN_SAMPLES) samples, or when the least-squares fit - /// fails to converge to parameters that [`new`](Self::new) accepts. + /// Returns [`None`] when `minimum_distance` exceeds `spread`, when `config` has fewer than + /// [`MIN_SAMPLES`](AffinityFitConfig::MIN_SAMPLES) samples, or when the solver or final + /// parameter narrowing fails as described by [`fit`](Self::fit). #[must_use] pub(crate) fn fit_with( spread: Positive, @@ -117,9 +123,9 @@ impl AffinityCurve { let minimum_distance = minimum_distance.widen(); - // Total: the two range factors are f32-born positives widened exactly, so their product - // lies far inside the f64 range in both directions. The divisor, an exact integer in [7, - // u16::MAX] by the `MIN_SAMPLES` guard above, keeps the quotient positive. + // Positive f32 factors lie in [2⁻¹⁴⁹, 2¹²⁸), and their exact product lies in [2⁻²⁹⁸, 2²⁵⁶). + // The guard puts the exact integer divisor in [7, 65534]. The quotient is therefore + // positive and finite in f64, well above underflow. let step = (config.range_in_spreads.mul_wide(spread) / DPositive::from_u16(samples_zero_based)) .finish_unchecked(); @@ -140,13 +146,13 @@ impl AffinityCurve { /// Initial Levenberg-Marquardt damping factor. const INITIAL_DAMPING: f64 = 1e-3; -/// Multiplicative damping adjustment: accepted steps divide by it, rejected steps multiply. +/// Multiplicative damping adjustment for accepted and rejected steps. const DAMPING_SCALE: f64 = 3.0; -/// Upper bound on accepted Levenberg-Marquardt iterations. +/// Upper bound on Levenberg-Marquardt outer iterations. const MAX_ITERATIONS: u32 = 100; /// Upper bound on consecutively rejected steps within one iteration. const MAX_REJECTIONS: u32 = 16; -/// Relative tolerance below which a step or a cost improvement counts as converged. +/// Relative step or cost-improvement threshold for successful termination. const CONVERGENCE_TOLERANCE: f64 = 1e-10; /// Evenly spaced sample distances of the fit target, starting at zero. @@ -160,26 +166,30 @@ pub(super) struct SampleGrid { impl SampleGrid { /// Creates a grid of `samples` distances spaced `step` apart from zero. + /// + /// Every requested index-times-step product must remain finite. This requirement is numerical + /// and is not validated here. #[inline] #[must_use] pub(super) const fn new(samples: u16, step: DPositive) -> Self { Self { samples, step } } - /// Returns the sample distance at an index. + /// Returns the index-times-spacing product. + /// + /// The product must be finite, including when `index` lies outside the configured sample count. const fn distance(self, index: u16) -> DNonNegative { - // The factor is an exact integer below 2^16, so the product leaves the domain only for - // a step in the top sixteen exponent shells of `f64`, and every constructed step sits - // hundreds of shells below them: an f32-born product in `fit_with`, small literals in - // tests. Underflow rounds to zero, inside the domain. + // A nonnegative integer times a positive finite step is nonnegative. The grid's numerical + // contract supplies finiteness. Therefore the product, including underflow to zero, is in + // DNonNegative's domain. DNonNegative::new_unchecked(DNonNegative::from_u16(index).get() * self.step.get()) } } -/// Sums of one solver pass: the least-squares objective and the terms of the 2x2 normal equations. +/// Objective, normal matrix and right-hand-side terms from one solver evaluation. /// -/// With the residual vector `r` and its Jacobian `J` in `(a, b)`, the `j_*` fields are the entries -/// of the normal matrix `J^T J` and the `g_*` fields the entries of the gradient `J^T r`. +/// With residual vector r and Jacobian J in parameters (a, b), the `j_*` fields hold `JᵀJ` and the +/// `g_*` fields hold Jᵀr. The latter is the gradient of half the residual sum of squares. #[derive(Debug, Copy, Clone)] struct NormalEquations { /// Sum of squared residuals, the objective the fit minimizes. @@ -218,21 +228,30 @@ impl NormalEquations { } } -/// Fits the affinity curve `1 / (1 + a · d^(2b))` to a target sampled on a grid. +/// Estimates positive affinity parameters for a sampled target. /// -/// By Levenberg-Marquardt least squares. +/// Both parameters start at one and remain finite and strictly positive. Each iteration solves the +/// damped 2x2 normal equations of the analytic Jacobian, accepting a step only when it lowers the +/// computed residual sum of squares. Rejections multiply damping by three, and acceptance divides +/// it by three. /// -/// Both parameters start at `1` and stay strictly positive throughout. Each iteration solves the -/// damped 2x2 normal equations of the analytic Jacobian in closed form and accepts the step when it -/// lowers the residual sum of squares. Rejected steps raise the damping and retry. Returns the -/// fitted `(a, b)`, or [`None`] when the initial evaluation is non-finite, when every damping retry -/// of an iteration fails, or when the iteration cap passes without convergence. +/// Returns the current parameters when both relative damped steps are at most 10⁻¹⁰, or when an +/// accepted relative cost improvement is at most 10⁻¹⁰. A small damped step can result from large +/// damping rather than stationarity. Returns [`None`] when the initial evaluation fails its +/// finiteness check, all 16 retries of an iteration fail, or 100 outer iterations finish without a +/// stopping criterion. +/// +/// `grid` must satisfy its finite-distance contract, and `target` must return the same value for a +/// distance throughout the fit. It can be evaluated repeatedly at every grid point. +/// +/// # Panics +/// +/// Propagates a panic from `target`. pub(super) fn fit_curve( grid: SampleGrid, target: impl Fn(DNonNegative) -> f64, ) -> Option<(DPositive, DPositive)> { - // Both parameters start at 1, the neutral point of the curve's - // O(1) parametrization. The damping retries absorb a rough start. + // initialize at q(d) = 1 / (1 + d²) let (mut a, mut b) = (DPositive::ONE, DPositive::ONE); let mut equations = evaluate(grid, &target, a, b)?; @@ -248,17 +267,14 @@ pub(super) fn fit_curve( continue; }; - // A negligible step means the damped gradient no longer - // moves either parameter: a stationary point. + // small damped steps terminate the search without a separate gradient-norm test if step_a.abs() <= CONVERGENCE_TOLERANCE * a && step_b.abs() <= CONVERGENCE_TOLERANCE * b { return Some((a, b)); } - // `AffinityCurve::new` accepts strictly positive parameters - // only; a step that leaves the domain is a failed step, not - // an error. + // reject steps leaving the positive finite parameter domain let (Some(next_a), Some(next_b)) = (DPositive::new(a + step_a), DPositive::new(b + step_b)) else { @@ -301,13 +317,17 @@ pub(super) fn fit_curve( /// Accumulates one pass of the fit objective at the given parameters. /// -/// Computes the residual `1 / (1 + a · d^(2b)) - target(d)` and its analytic partial derivatives at -/// every grid distance, folding the residual sum of squares and the normal-equation sums in a -/// single pass, accumulated in double precision. `b` is strictly positive, so the zero-distance -/// sample contributes `d^(2b) = 0` to its residual; its partials are zero in both parameters, and -/// skipping them keeps `ln` off distance zero. +/// For P = d^(2b) and Z = 1 + aP, the residual is r = 1/Z − target(d). At d > 0 the analytic +/// partials are ∂r/∂a = −P/Z² and ∂r/∂b = −2aP ln(d)/Z². At d = 0, positive b gives P = 0 and both +/// partials vanish. Skipping those partials avoids evaluating ln(0). /// -/// Returns [`None`] when any accumulated sum turns non-finite. +/// The pass accumulates Σr², `JᵀJ` and Jᵀr in `f64`. Every grid distance must satisfy +/// [`DNonNegative`]'s finite-result contract. Returns [`None`] when 2b overflows or the final +/// accumulated sums include a non-finite value. +/// +/// # Panics +/// +/// Propagates a panic from `target`. fn evaluate( grid: SampleGrid, target: &impl Fn(DNonNegative) -> f64, @@ -317,17 +337,13 @@ fn evaluate( let mut sums = NormalEquations::ZERO; for index in 0..grid.samples { - // Solver interior: the parameters roam during exploration, and an overflow here is an - // expected rejection for the pass-end finiteness check rather than a defective input. - // The arithmetic is raw until that check. let distance = grid.distance(index); let power = distance.powf(2.0 * b).get(); let denominator = a.get().mul_add(power, 1.0); let residual = 1.0 / denominator - target(distance); sums.residual_sum_of_squares = residual.mul_add(residual, sums.residual_sum_of_squares); - // The zero-distance sample contributes value alone; its partials vanish, and the - // narrowing keeps `ln` off distance zero. + // the zero-distance sample contributes residual error with zero parameter partials let Some(distance) = distance.positive() else { continue; }; @@ -345,12 +361,16 @@ fn evaluate( sums.is_finite().then_some(sums) } -/// Solves the damped normal equations for one Levenberg-Marquardt step in closed form. +/// Solves the multiplicatively damped 2x2 normal system. +/// +/// With H = `JᵀJ` and g = Jᵀr, the system is MΔ = −g, where M = H + λ diag(H) and λ is `damping`. +/// Cramer's rule gives each step component from the determinant. For exact positive-semidefinite H +/// with both diagonal entries positive, λ > 0 makes M positive definite. A zero diagonal remains +/// zero under this damping. /// -/// Dampens each diagonal entry of `J^T J` by `1 + damping` and solves the symmetric 2x2 system `M * -/// step = -g` by Cramer's rule. Returns [`None`] when the damped determinant falls to the -/// cancellation floor (the system is numerically singular at this damping; a larger damping factor -/// restores diagonal dominance) or when the step is non-finite. +/// Returns [`None`] when the computed determinant is non-finite or at most ε · M₀₀ · M₁₁, where ε +/// is [`f64::EPSILON`], or when a computed step is non-finite. This numerical floor rejects +/// near-cancellation, without certifying exact conditioning. fn solve_damped(equations: NormalEquations, damping: f64) -> Option<(f64, f64)> { let damped_aa = equations.j_aa * (1.0 + damping); let damped_bb = equations.j_bb * (1.0 + damping); diff --git a/libs/@local/graph/atlas/src/math/affinity/mod.rs b/libs/@local/graph/atlas/src/math/affinity/mod.rs index 9a58290b14b..a927f90994c 100644 --- a/libs/@local/graph/atlas/src/math/affinity/mod.rs +++ b/libs/@local/graph/atlas/src/math/affinity/mod.rs @@ -1,30 +1,28 @@ -//! The affinity curve of force-directed layouts and its gradient steps. +//! UMAP-style affinity and clipped attraction/repulsion updates. //! -//! A UMAP-style layout keeps a low-dimensional affinity curve +//! For a point difference Δ = from − to ∈ ℝ² and squared distance ρ = ‖Δ‖² ≥ 0, the affinity model +//! is q(ρ) = 1 / (1 + aρᵇ), with a, b > 0. Equivalently, at Euclidean distance d it is 1 / (1 + a · +//! d^(2b)). [`AffinityCurve`] holds these parameters and evaluates scalar or four-pair SIMD +//! updates. //! -//! ```text -//! q(d) = 1 / (1 + a · d^(2b)) -//! ``` +//! An edge contributes loss −ln q. Differentiating with respect to `from` and negating gives the +//! attraction update −2abρ^(b−1)Δ / (1 + aρᵇ). A non-edge with weight γ ≥ 0 contributes −γ ln(1 − +//! q), whose negative gradient is 2γbΔ / (ρ(1 + aρᵇ)) for ρ > 0. Repulsion replaces the leading +//! denominator ρ with ρ + ε, where ε = 0.001, to regularize close pairs. //! -//! over the 2D distance `d` between points, and descends its cross-entropy against the -//! high-dimensional neighbour graph by stochastic gradient steps. Sampled edges pull their -//! endpoints together (attraction), and sampled non-edges push them apart (repulsion). -//! [`AffinityCurve`] holds the fitted `a` and `b` parameters and evaluates both gradient families -//! for four point pairs at a time over [`Vec2x4T`] batches. +//! Each update is clamped componentwise to ±[`GRADIENT_CLIP`](AffinityCurve::GRADIENT_CLIP). A +//! finite clipped update has Euclidean norm at most 4√2 before multiplication by a learning rate. +//! Componentwise clipping preserves each component's sign but can change the direction from a +//! scalar multiple of Δ. These functions return updates without moving either endpoint. //! -//! [`AffinityCurve`] clamps every per-axis gradient component to -//! [`GRADIENT_CLIP`](AffinityCurve::GRADIENT_CLIP) before the caller applies the learning rate, -//! which bounds the step a single sample can take and stops one sample from flinging an early, -//! badly-placed point across the layout. +//! Finite coincident points receive zero updates because their difference supplies no direction. +//! Distinct points whose computed squared distance underflows to zero also receive zero. Gradient +//! arithmetic uses `f32`, with explicit fused multiply-adds. SIMD powers use [`pow_f32x4`], while +//! scalar powers use [`f32::powf`]. Their approximations and distance grouping can differ. Finite +//! input coordinates and positive parameters alone do not prevent intermediate overflow or NaNs, +//! and clipping does not establish a universally finite result. //! -//! Exactly coincident points receive no gradient in either direction. Their difference vector gives -//! no direction for a descent step, so layouts rely on distinct initial placement to separate -//! identical points. -//! -//! All gradient arithmetic is `f32` with FMA contraction where the target provides it, and the -//! kernels are fully vectorized, including the `d^(2b)` power. The one exception is -//! [`AffinityCurve::fit`], the one-shot least-squares parameter fit at initialization, which runs -//! in double precision and narrows its result to `f32`. +//! [`AffinityCurve::fit`] estimates the parameters in `f64` and narrows the result to `f32`. #![expect( clippy::min_ident_chars, reason = "`a` and `b` are the canonical names of the UMAP curve parameters throughout the \ @@ -47,20 +45,24 @@ pub(crate) use self::fit::AffinityFitConfig; #[cfg(test)] mod tests; -/// The affinity curve `1 / (1 + a · d^(2b))` mapping layout distance to edge probability. +/// A positive-parameter affinity curve for layout distances. +/// +/// The model is q(ρ) = 1 / (1 + aρᵇ) for squared distance ρ ≥ 0. Parameter a sets the distance +/// scale and b shapes the decay. Both are finite and strictly positive by construction. +/// [`fit`](Self::fit) estimates them from a desired membership falloff. /// -/// The parameters come from fitting the curve against the desired membership falloff (spread and -/// minimum distance) with [`fit`](Self::fit), as UMAP's `a` and `b`; `a` scales the curve and `b` -/// shapes its tail. Both are strictly positive and finite by construction. +/// # Example /// -/// # Examples +/// This example is ignored because [`AffinityCurve`] is crate-private. /// /// ```ignore -/// let curve = AffinityCurve::new(1.577, 0.895).expect("parameters are positive and finite"); +/// use crate::math::{AffinityCurve, NonNegative, Vec2, non_negative, positive}; +/// +/// let curve = AffinityCurve::new(positive!(1.577), positive!(0.895)); /// /// // Affinity is 1 at zero distance and falls off monotonically. -/// assert_eq!(curve.affinity(0.0), 1.0); -/// assert!(curve.affinity(1.0) > curve.affinity(4.0)); +/// assert_eq!(curve.affinity(NonNegative::ZERO), 1.0); +/// assert!(curve.affinity(non_negative!(1.0)) > curve.affinity(non_negative!(4.0))); /// /// // Attraction pulls the endpoint toward the anchor. /// let gradient = curve.attraction(Vec2::new(2.0, 0.0), Vec2::ZERO); @@ -74,47 +76,44 @@ pub(crate) struct AffinityCurve { } impl AffinityCurve { - /// The symmetric per-axis bound on every gradient component. + /// The symmetric per-axis clip for finite gradient components. /// - /// Coefficients diverge as distances approach zero; the clamp bounds the displacement a single - /// sampled pair can cause, before the learning rate scales it. A displacement bound takes its - /// scale from the frame it moves in: the clip and the caller's layout extent fix one ratio, so - /// a caller sizing its initial frame sizes it against this constant. + /// Clipping limits each component to [−4, 4] before a learning rate is applied. Compare this + /// scale with the coordinate extent when choosing update magnitudes. pub(crate) const GRADIENT_CLIP: f32 = 4.0; /// Additive guard in the repulsion denominator. /// - /// Keeps the coefficient finite as the squared distance approaches zero, bounding the repulsion - /// between near-coincident points. + /// Replaces ρ with ρ + ε in the repulsion denominator, with ε = 0.001. In real arithmetic this + /// bounds its nonnegative coefficient by 2γb/ε near zero. It does not prevent `f32` overflow + /// for arbitrary parameter magnitudes. const REPULSION_GUARD: NonNegative = non_negative!(0.001); /// Creates a curve from its fitted parameters. - /// - /// Returns [`None`] unless both parameters are finite and strictly positive; the gradient - /// expressions divide by `a`-scaled powers and multiply by `b`, so zero, negative, or - /// non-finite parameters produce meaningless layouts. #[must_use] pub(crate) fn new(a: f32, b: f32) -> Option { (a.is_finite() && a > 0.0 && b.is_finite() && b > 0.0).then_some(Self { a, b }) } - /// Returns the `a` parameter. + /// Returns the coefficient controlling the affinity's distance scale. #[inline] #[must_use] pub(crate) const fn a(self) -> f32 { self.a } - /// Returns the `b` parameter. + /// Returns the exponent shaping the affinity's decay. #[inline] #[must_use] pub(crate) const fn b(self) -> f32 { self.b } - /// Evaluates the affinity `1 / (1 + a · d^(2b))` at a squared distance. + /// Evaluates q(ρ) = 1 / (1 + aρᵇ) at a squared distance. /// - /// The affinity is `1` at distance zero and falls monotonically toward zero; it is the - /// low-dimensional edge probability the layout optimizes toward. + /// A zero `distance_squared` returns one. For positive inputs, this evaluates the model + /// with [`f32::powf`] and a fused denominator. The real curve decreases monotonically, but no + /// strict monotonicity or ULP guarantee is made for the approximation. Overflow in the positive + /// denominator can produce a zero affinity. #[must_use] pub(crate) fn affinity(self, distance_squared: f32) -> f32 { if distance_squared <= 0.0 { @@ -126,11 +125,13 @@ impl AffinityCurve { /// Computes the clipped attraction gradients of four point pairs. /// - /// Entry `i` is the gradient acting on `from[i]` for the edge toward `to[i]`. It is a negative - /// multiple of the difference vector, clamped per axis, so it points from `from` toward `to`. - /// The symmetric update applies `+lr · gradient` to `from` and `-lr · gradient` to `to`. + /// Lane i acts on `from[i]` toward `to[i]`, following the module's attraction formula and + /// componentwise clip. For a symmetric update with learning rate η, add ηg to `from` and + /// subtract ηg from `to`. /// - /// Coincident pairs receive a zero gradient. + /// Both endpoint batches must be finite. To obtain finite updates, the computed coefficient and + /// scaled differences must avoid NaNs. Finite coincident pairs, including computed-zero squared + /// distances, receive zero. #[must_use] pub(crate) fn attraction_x4(self, from: Vec2x4T, to: Vec2x4T) -> Vec2x4T { let distance_squared = from.distance_squared(to); @@ -154,11 +155,13 @@ impl AffinityCurve { /// Computes the clipped repulsion gradients of four point pairs. /// - /// Entry `i` is the gradient acting on `from[i]` away from the negative sample `to[i]`: a - /// positive multiple of the difference vector, clamped per axis. Only `from` moves; negative - /// samples stay in place. `repulsion_strength` is the `gamma` weight of the repulsive term. + /// Lane i acts on `from[i]` away from `to[i]`, following the module's regularized repulsion + /// formula and componentwise clip. `repulsion_strength` is γ. The function moves neither + /// endpoint. /// - /// Coincident pairs receive a zero gradient. + /// Both endpoint batches must be finite. To obtain finite updates, the computed coefficient and + /// scaled differences must avoid NaNs. Finite coincident pairs, including computed-zero squared + /// distances, receive zero. #[must_use] pub(crate) fn repulsion_x4( self, @@ -183,8 +186,10 @@ impl AffinityCurve { /// Computes the clipped attraction gradient of a single point pair. /// - /// Scalar twin of [`attraction_x4`](Self::attraction_x4) for loop remainders; the semantics are - /// identical. + /// This uses [`attraction_x4`](Self::attraction_x4)'s model with scalar powers and distance + /// arithmetic. Both points and the computed squared distance must be finite. Computed-zero + /// squared distance returns zero. The coefficient and scaled differences must avoid NaNs + /// for a finite clipped result. Scalar and SIMD values can differ. #[must_use] pub(crate) fn attraction(self, from: Vec2, to: Vec2) -> Vec2 { let distance_squared = from.distance_squared(to); @@ -201,8 +206,10 @@ impl AffinityCurve { /// Computes the clipped repulsion gradient of a single point pair. /// - /// Scalar twin of [`repulsion_x4`](Self::repulsion_x4) for loop remainders; the semantics are - /// identical. + /// This uses [`repulsion_x4`](Self::repulsion_x4)'s model with scalar powers and distance + /// arithmetic. Both points and the computed squared distance must be finite. Computed-zero + /// squared distance returns zero. The coefficient and scaled differences must avoid NaNs for a + /// finite clipped result. Scalar and SIMD values can differ. #[must_use] pub(crate) fn repulsion(self, from: Vec2, to: Vec2, repulsion_strength: f32) -> Vec2 { let distance_squared = from.distance_squared(to); diff --git a/libs/@local/graph/atlas/src/math/affinity/tests.rs b/libs/@local/graph/atlas/src/math/affinity/tests.rs index 483d0526a66..101f737e5bd 100644 --- a/libs/@local/graph/atlas/src/math/affinity/tests.rs +++ b/libs/@local/graph/atlas/src/math/affinity/tests.rs @@ -23,13 +23,10 @@ use crate::math::{AffinityCurve, Positive, Vec2, Vec2x4T, d_positive, positive, const CURVE_A: f32 = 1.577; const CURVE_B: f32 = 0.895; -/// Independent f64 reference for the fit objective. +/// Computes the target residual sum of squares with a separate `f64` loop. /// -/// The residual sum of squares of a candidate curve against the target falloff sampled on the -/// crate's default grid, 300 samples over `[0, 3 · spread]`. -/// -/// This reference spells the grid constants out by hand, so it fails when the documented default -/// contract changes. +/// The grid uses 300 samples over [0, 3σ], where σ is `spread`. The fixed constants match the +/// documented default grid without using the solver's grid construction. fn reference_rss(spread: f64, minimum_distance: f64, curve_a: f64, curve_b: f64) -> f64 { let mut rss = 0.0; for index in 0..300_u16 { @@ -46,13 +43,16 @@ fn reference_rss(spread: f64, minimum_distance: f64, curve_a: f64, curve_b: f64) rss } +/// Creates the curve with parameters [`CURVE_A`] and [`CURVE_B`]. fn curve() -> AffinityCurve { AffinityCurve::new(CURVE_A, CURVE_B).expect("reference parameters are positive and finite") } -/// Independent f64 reference for the attraction coefficient. +/// Computes the attraction update using separate `f64` powers. /// -/// Transcribed from the pre-SIMD scalar kernel rather than from the implementation under test. +/// The squared distance is first computed in `f32`, then widened. The coefficient evaluates ρ^(b−1) +/// and ρᵇ separately before clipping. Inputs must meet [`Vec2::distance_squared`]'s finite-result +/// contract. fn reference_attraction(from: Vec2, to: Vec2) -> Vec2 { let distance_squared = f64::from(from.distance_squared(to)); if distance_squared <= 0.0 { @@ -66,7 +66,10 @@ fn reference_attraction(from: Vec2, to: Vec2) -> Vec2 { reference_clipped(from, to, coefficient) } -/// Independent f64 reference for the repulsion coefficient. +/// Computes the regularized repulsion update with a `f64` coefficient. +/// +/// The squared distance is first computed in `f32`, then widened. Inputs must meet +/// [`Vec2::distance_squared`]'s finite-result contract. fn reference_repulsion(from: Vec2, to: Vec2, repulsion_strength: f64) -> Vec2 { let distance_squared = f64::from(from.distance_squared(to)); if distance_squared <= 0.0 { @@ -80,6 +83,7 @@ fn reference_repulsion(from: Vec2, to: Vec2, repulsion_strength: f64) -> Vec2 { reference_clipped(from, to, coefficient) } +/// Scales widened `f32` differences, clips to ±4 and narrows to `f32`. fn reference_clipped(from: Vec2, to: Vec2, coefficient: f64) -> Vec2 { #[expect( clippy::cast_possible_truncation, @@ -91,6 +95,13 @@ fn reference_clipped(from: Vec2, to: Vec2, coefficient: f64) -> Vec2 { Vec2::new(component(from.x() - to.x()), component(from.y() - to.y())) } +/// Asserts componentwise agreement within a relative/absolute tolerance. +/// +/// Each absolute difference must be below 10⁻⁵ · max(|expected|, 1). +/// +/// # Panics +/// +/// Panics when either component fails the comparison, naming `context`. #[track_caller] fn assert_close(actual: Vec2, expected: Vec2, context: &str) { let tolerance = |reference: f32| 1e-5 * reference.abs().max(1.0); @@ -102,6 +113,7 @@ fn assert_close(actual: Vec2, expected: Vec2, context: &str) { ); } +/// Anchor points for paired updates against [`POINTS`]. const ANCHORS: [Vec2; 4] = [ Vec2::new(0.0, 0.0), Vec2::new(1.5, 6.5), @@ -126,11 +138,12 @@ fn fit_reproduces_the_reference_parameters() { ); } +/// Compares the fitted objective with a local parameter grid. +/// +/// The comparison varies a by a relative ±10⁻³ and b by an additive ±10⁻³, checking only the +/// resulting grid points. #[test] fn fit_result_is_a_local_minimum_of_the_sampled_objective() { - // Local-optimality certificate: the fitted parameters score at least - // as well as every small perturbation on a grid around them, against - // the same sampled objective recomputed independently in f64. let fitted = AffinityCurve::fit(positive!(1.0), positive!(0.1)) .expect("the reference inputs are well-conditioned"); let (curve_a, curve_b) = (f64::from(fitted.a()), f64::from(fitted.b())); @@ -155,11 +168,8 @@ fn fit_result_is_a_local_minimum_of_the_sampled_objective() { #[test] fn fit_recovers_the_parameters_of_an_exact_affinity_target() { - // Exact-recovery certificate: when the sampled target IS an affinity - // curve, the zero-residual minimum sits at its parameters and the - // solver must return them. This drives the internal solver directly; - // the public API only exposes the exponential-falloff target, which - // no affinity curve reproduces exactly. + // A target from the same curve family has zero real-arithmetic residual at its generating + // parameters. This gives known coefficients to compare with the numerical solve. let (known_a, known_b) = (1.5_f64, 0.9_f64); let grid = SampleGrid::new(300, d_positive!(3.0 / 299.0)); @@ -178,12 +188,14 @@ fn fit_recovers_the_parameters_of_an_exact_affinity_target() { ); } +/// Compares fitted parameters under a rescaling of distances. + #[test] fn fit_scales_equivariantly_with_distance() { - // Scaling-law certificate: scaling all distances by `s` maps a - // solution `(a, b)` to `(a · s^(-2b), b)` exactly, and - // `fit(s · spread, s · minimum_distance)` samples the same target at - // distances scaled by `s`. + // In real arithmetic, replacing each distance d by sd gives a · s^(−2b) · (sd)^(2b) = a · + // d^(2b). Scaling spread and minimum distance by the same s preserves the target values at + // corresponding grid points. Therefore (a · s^(−2b), b) is the corresponding fitted model. The + // assertions allow numerical fitting and narrowing error. let base = AffinityCurve::fit(positive!(1.0), positive!(0.1)) .expect("the reference inputs are well-conditioned"); @@ -239,16 +251,11 @@ fn fitted_curve_tracks_its_target_falloff() { #[test] fn fit_rejects_a_minimum_distance_beyond_the_spread() { - // The signature's `Positive` domain refuses non-finite and non-positive inputs before the - // fit sees them. The ordering between the two is the one refusal left to the fit itself. assert!(AffinityCurve::fit(positive!(1.0), positive!(2.0)).is_none()); } #[test] fn fit_with_rejects_degenerate_configs() { - // Too few samples for the two-parameter fit. The range's `Positive` domain refuses a - // degenerate range at construction, so the sample floor is the config refusal left to the - // fit itself. assert!( AffinityCurve::fit_with( positive!(1.0), @@ -278,8 +285,8 @@ fn fit_with_rejects_degenerate_configs() { #[test] fn fit_is_stable_under_sample_refinement() { - // Refining the discretization must not move the minimizer: the - // sampled objective converges to its continuous limit. + // the 10⁻² tolerance accommodates shifts in the minimizer as grid refinement changes the + // sampled objective. let base = AffinityCurve::fit(positive!(1.0), positive!(0.1)) .expect("the reference inputs are well-conditioned"); @@ -308,19 +315,13 @@ fn fit_is_stable_under_sample_refinement() { #[test] fn fit_accepts_a_minimum_distance_equal_to_the_spread() { - // The documented domain excludes only `minimum_distance > spread`; - // equality is the boundary case that stays inside. assert!(AffinityCurve::fit(positive!(1.0), positive!(1.0)).is_some()); } #[test] fn fit_with_divides_the_range_into_samples_minus_one_steps() { - // The spacing divides the sampled range by `samples - 1`, not by - // `samples`, and the final sample therefore sits exactly at the range - // end. The explicit grid spells that spacing out; fitting over it runs - // the identical solver on identical inputs, so the results agree to - // narrowing precision. A coarse eight-sample grid makes any spacing - // drift orders of magnitude wider than the tolerance. + // spacing 3/7 places the last of eight samples at distance 3 in real arithmetic. The explicit + // grid uses the same rounded spacing and widened minimum distance as fit_with. let fitted = AffinityCurve::fit_with( positive!(1.0), positive!(0.1), @@ -353,12 +354,8 @@ fn fit_with_divides_the_range_into_samples_minus_one_steps() { #[test] fn solver_refuses_a_target_that_poisons_only_the_objective() { - // The zero-distance sample contributes its residual and nothing else - // (its partials are skipped), so a NaN target at zero drives exactly - // one accumulated sum non-finite: the residual sum of squares. The - // finiteness gate is the only check standing between that poisoned - // objective and a solver that converges happily on the remaining - // samples. + // At distance zero both parameter partials vanish and are skipped. A NaN target there affects + // only the residual sum of squares, while the normal matrix and right-hand side stay finite. let result = fit_curve(SampleGrid::new(300, d_positive!(3.0 / 299.0)), |distance| { if distance == 0.0 { f64::NAN @@ -375,11 +372,9 @@ fn solver_refuses_a_target_that_poisons_only_the_objective() { #[test] fn solver_refuses_a_grid_that_cannot_move_the_exponent() { - // With distance one as the only positive sample, `ln 1 = 0` zeroes - // every `b`-partial: the normal matrix's `b` diagonal is exactly zero, - // and multiplicative damping keeps it zero. The damped solve must - // refuse the singular system at every retry rather than invent a step - // from an additively repaired diagonal. + // The only positive sample is d = 1, where ln(1) = 0 makes every b-partial zero. The + // corresponding normal-matrix diagonal is zero and multiplicative damping leaves it zero. + // Therefore the determinant test rejects every retry. let result = fit_curve(SampleGrid::new(2, d_positive!(1.0)), |distance| { if distance == 0.0 { 1.0 } else { 0.3 } }); @@ -392,11 +387,9 @@ fn solver_refuses_a_grid_that_cannot_move_the_exponent() { #[test] fn solver_solves_a_well_conditioned_system_of_tiny_magnitudes() { - // Distances of 1e-5 shrink every normal-equation entry by tens of - // orders of magnitude while the system stays perfectly solvable. The - // cancellation floor scales with the product of BOTH damped diagonals; - // a floor divided by either diagonal inflates astronomically here and - // refuses every step. + // Tiny distances make the Jacobian entries small. Scaling the determinant floor by the product + // of both damped diagonals tests relative cancellation without imposing an absolute + // matrix-magnitude floor. let (fitted_a, fitted_b) = fit_curve(SampleGrid::new(4, d_positive!(1e-5)), |distance| { 1.0 / (1.0 + 2.0 * distance.powf(2.0)) }) @@ -414,12 +407,7 @@ fn solver_solves_a_well_conditioned_system_of_tiny_magnitudes() { #[test] fn fit_recovers_parameters_orders_of_magnitude_from_the_start() { - // From the fixed start (1, 1), an exact target at a = 100 or a = 1e8 - // is reached only through the damping controller's full cycle: - // rejected overshoots raise the damping multiplicatively, accepted - // steps relax it, and the tiny-step exit measures each step against - // its own parameter. Flipping or rescaling any of those strands the - // walk short of recovery. + // the target coefficient a spans several orders of magnitude from the fixed start at one for known_a in [100.0, 1e8] { let (fitted_a, fitted_b) = fit_curve(SampleGrid::new(300, d_positive!(3.0 / 299.0)), |distance| { @@ -438,19 +426,17 @@ fn fit_recovers_parameters_orders_of_magnitude_from_the_start() { } } +/// Terminates a fit to a target extending above the affinity range. + #[test] fn solver_terminates_a_creep_along_the_domain_boundary() { - // A target thirty times the curve's whole range pulls `a` toward the - // zero boundary it may never cross: every Newton step overshoots into - // the forbidden quadrant, and only a damped fraction survives. The - // walk ends through the improvement difference falling under its - // relative tolerance. Additive damping growth on domain rejections - // and both broken improvement readings leave the creep unfinished. + // The target approaches 30 near zero, while the curve cannot exceed one. For positive + // distances, decreasing a moves the curve toward one. This puts the fit near its + // positive-parameter boundary, where damping and stopping thresholds matter. let result = fit_curve(SampleGrid::new(300, d_positive!(3.0 / 299.0)), |distance| { 30.0 / (1.0 + 1.5 * distance.powf(1.8)) }); - // The returned parameters are positive by type. Convergence is the claim left to assert. assert!( result.is_some(), "the boundary creep must converge: {result:?}", @@ -459,12 +445,9 @@ fn solver_terminates_a_creep_along_the_domain_boundary() { #[test] fn tiny_step_convergence_requires_both_parameters() { - // The `a` parameter is bisected so the first damped step moves `a` by - // under 1e-17 (seven orders inside the relative tolerance) while - // moving `b` by 0.84. An exit that accepts either tiny component - // alone, or measures `b` against an absolute threshold, stops at the - // start (1, 1); the conjunction of relative thresholds walks on and - // recovers the target exactly. + // At the initial (a, b) = (1, 1), this target gives a first damped a-step near 8 × 10⁻¹⁸ and a + // b-step near 0.843. One component is well below the relative step threshold while the other + // remains large. Both components must satisfy the threshold to stop. const TUNED_A: f64 = 0.931_322_701_934_099; let (fitted_a, fitted_b) = fit_curve(SampleGrid::new(12, d_positive!(0.25)), |distance| { @@ -484,11 +467,7 @@ fn tiny_step_convergence_requires_both_parameters() { #[test] fn solver_rescues_a_walk_whose_steps_worsen_the_objective() { - // A steep falloff thirty spreads out makes whole stretches of proposed - // steps worsen the objective before the walk finds the descent again. - // The rescue lives in the rejection damping growing multiplicatively: - // sixteen additive bumps cap the damping near fifty, and the walk - // never re-enters the acceptable region. + // the target a = b = 30 has a steep falloff over a grid extending to distance 30 let (fitted_a, fitted_b) = fit_curve(SampleGrid::new(50, d_positive!(30.0 / 49.0)), |distance| { 1.0 / (1.0 + 30.0 * distance.powf(60.0)) @@ -609,9 +588,8 @@ fn coincident_pairs_receive_no_gradient() { #[test] fn near_coincident_repulsion_saturates_the_clip() { let curve = curve(); - // Close along x, but far enough that the coefficient (capped near - // 2 · γ · b / 0.001 by the repulsion guard) times the difference - // still exceeds the clip: 0.01 · ~1600 is ~16, clamped to 4. + // With ρ ≈ 10⁻⁴ and γ = 1, the denominator is about 0.0011 and the numerator about 1.79. The + // coefficient is about 1600, giving an x update near 16 before clipping to 4. let from = Vec2::new(0.01, 0.0); let to = Vec2::ZERO; @@ -636,35 +614,32 @@ fn gradients_stay_finite_at_extreme_distances() { assert!(batch.get(0).is_finite()); } -/// A point with coordinates bounded to the well-conditioned `-1e3..1e3` range. -/// -/// The example-based tests above pin extreme-distance behaviour. +/// Generates finite points with coordinates in `-1e3..1e3`. fn point_strategy() -> impl Strategy { (-1e3_f32..1e3, -1e3_f32..1e3).prop_map(|(x, y)| Vec2::new(x, y)) } -/// Arbitrary in-range points, one per batch lane. +/// Generates a four-point batch from [`point_strategy`]. fn point_array_strategy() -> impl Strategy { proptest::array::uniform4(point_strategy()) } -/// Curve parameters bounded to `a` in `1e-3..1e3` and `b` in `0.1..5`. -/// -/// Where `a · d^(2b)` stays finite over the strategy's distances. +/// Generates curves with a in `1e-3..1e3` and b in `0.1..5`. fn curve_strategy() -> impl Strategy { (1e-3_f32..1e3, 0.1_f32..5.0).prop_map(|(curve_a, curve_b)| { AffinityCurve::new(curve_a, curve_b).expect("the strategy's ranges are positive and finite") }) } -/// Asserts a batch lane agrees with its scalar twin within a relative tolerance of `1e-3`. +/// Asserts componentwise agreement under a relative tolerance and absolute floor. /// -/// A matching absolute floor covers near-zero gradients. +/// The allowance is 10⁻³ · max(|expected|, 10⁻³), giving an absolute floor of 10⁻⁶. Scalar and SIMD +/// paths use different distance grouping and power approximations. This is the fixture's comparison +/// tolerance, not a certified ULP bound for either power implementation. /// -/// The bound covers the batch kernels' vectorized `d^(2b)` power, which composes sleef's 3.5-ulp -/// `exp2`/`log2` stages. The exponent's absolute error grows with `|log2(d^2)|`, so the power's -/// relative error reaches a few times `1e-5` over the strategy's distance range, well inside -/// `1e-3`, against the scalar path's 0.5-ulp libm `powf`. +/// # Panics +/// +/// Panics when either component fails the comparison, naming `context`. #[track_caller] fn assert_lane_close(actual: Vec2, expected: Vec2, context: &str) { let tolerance = |reference: f32| 1e-3 * reference.abs().max(1e-3); @@ -676,10 +651,10 @@ fn assert_lane_close(actual: Vec2, expected: Vec2, context: &str) { ); } -/// The affinity lies in `(0, 1]` and is monotone non-increasing in the squared distance. +/// Samples affinity range and approximate ordering on bounded squared distances. /// -/// Monotonicity holds up to a few ulps of libm `powf` rounding. This test bounds squared distances -/// to `0..1e6`, where `a · d^(2b)` stays finite for every curve in the strategy. +/// With ρ < 10⁶, a < 10³ and b < 5, the real-arithmetic product aρᵇ is below 10³³ for ρ ≥ 1, +/// keeping it within `f32` range. Ordering uses a relative tolerance. #[property_test] fn affinity_is_a_monotone_probability( #[strategy = curve_strategy()] curve: AffinityCurve, @@ -698,9 +673,7 @@ fn affinity_is_a_monotone_probability( prop_assert!(affinity <= 1.0); } - // `powf` is accurate to a fraction of an ulp but not proven - // monotone; the slack admits a few ulps of the result without - // accepting a real ordering violation. + // allow an ordering discrepancy up to 8 · EPSILON times the near affinity let slack = 8.0 * f32::EPSILON * curve.affinity(near); prop_assert!( curve.affinity(near) >= curve.affinity(far) - slack, @@ -712,10 +685,10 @@ fn affinity_is_a_monotone_probability( ); } -/// Attraction pulls `from` toward `to`, and repulsion pushes it away. +/// Checks attraction and repulsion signs relative to the point difference. /// -/// For distinct points, the attraction gradient is anti-parallel to the difference vector and the -/// repulsion gradient is parallel. The separation floor keeps the coefficients away from underflow. +/// The squared-distance floor 10⁻⁶ excludes computed-coincident pairs. The assertions test +/// dot-product signs, without requiring clipped updates to remain parallel to the difference. #[property_test] fn gradients_align_with_the_difference_vector( #[strategy = point_strategy()] from: Vec2, @@ -729,10 +702,6 @@ fn gradients_align_with_the_difference_vector( prop_assert!(curve.repulsion(from, to, 1.0).dot(difference) > 0.0); } -/// The batch attraction kernel agrees with the scalar kernel in every lane. -/// -/// This crosses the sleef `exp2`/`log2` pow path against the scalar libm `powf` path over the whole -/// in-range input space. The tolerance follows the kernel's documented 3.5-ulp-stage bound. #[property_test] fn attraction_x4_matches_scalar_attraction_per_lane( #[strategy = point_array_strategy()] from: [Vec2; 4], @@ -750,9 +719,6 @@ fn attraction_x4_matches_scalar_attraction_per_lane( } } -/// The batch repulsion kernel agrees with the scalar kernel in every lane. -/// -/// The same pow-path bound as attraction applies. #[property_test] fn repulsion_x4_matches_scalar_repulsion_per_lane( #[strategy = point_array_strategy()] from: [Vec2; 4], diff --git a/libs/@local/graph/atlas/src/math/bench.rs b/libs/@local/graph/atlas/src/math/bench.rs index 93ea54d2a3a..3f180f07307 100644 --- a/libs/@local/graph/atlas/src/math/bench.rs +++ b/libs/@local/graph/atlas/src/math/bench.rs @@ -1,12 +1,11 @@ //! Benchmark entry points for the vector and geometry kernels. //! -//! The `math_kernels` benchmark target times each kernel exactly as production calls it. Inputs -//! are built once ahead of the timed region, because production holds its vectors and transforms -//! across the fit loop's hot calls. Each timed function is a transparently inlined forwarder whose -//! operands stay pinned behind [`black_box`] at the same per-operand points the target used when -//! it named these types itself. Results that are plain numbers return to the caller. Results in -//! crate types are pinned here and dropped, so no internal type escapes. Nothing here is API for -//! consumers of the crate. +//! These entry points expose primitive inputs and opaque fixtures to the `math_kernels` target +//! while keeping the math types crate-private. Fixture construction can be timed separately from +//! repeated kernel calls. [`black_box`] marks the operands and results whose computation the +//! benchmark intends to retain, without establishing a universal optimizer or production-cost +//! guarantee. Reference entry points measure the stated scalar formulations, which can differ in +//! arithmetic and output precision. use core::hint::black_box; @@ -21,7 +20,7 @@ use super::{ field::POINT_CHUNK, transform::Transform, vec2::Vec2x4, }; -/// A dot-product operand pair, built once ahead of the timed region. +/// Fixed-size operands for vector-kernel benchmarks. pub struct VecNPair { left: VecN, right: VecN, @@ -36,7 +35,7 @@ pub fn vecn_pair(left: [f32; N], right: [f32; N]) -> VecNPair } } -/// Dot product, as production calls it. +/// Evaluates the vector dot product and returns its `f32` result. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -48,7 +47,9 @@ pub fn vecn_dot(pair: &VecNPair) -> f32 { black_box(&pair.left).dot(black_box(&pair.right)) } -/// The dot product's scalar reference accumulates lanewise `f64` products over the raw components. +/// Accumulates scalar `f64` products over the raw vector components. +/// +/// This reference returns the `f64` sum without the kernel's final `f32` narrowing. #[expect( clippy::inline_always, reason = "the benchmark must measure the reference as the target formulated it: transparently \ @@ -65,7 +66,9 @@ pub fn vecn_dot_scalar_reference(pair: &VecNPair) -> f64 { .sum::() } -/// Cosine distance, as production calls it. +/// Evaluates the cosine distance and returns its raw reading. +/// +/// Both operands must have finite components, as required by [`VecN::cosine_distance`]. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -74,8 +77,6 @@ pub fn vecn_dot_scalar_reference(pair: &VecNPair) -> f64 { #[inline(always)] #[must_use] pub fn vecn_cosine_distance(pair: &VecNPair) -> f32 { - // The raw reading crosses the hook because the scalar family is crate-internal and the - // bench target is another crate. black_box(&pair.left) .cosine_distance(black_box(&pair.right)) .get() @@ -91,7 +92,7 @@ fn vec2_batch(points: [[f32; 2]; 4]) -> Vec2x4T { ]) } -/// An affinity curve with one four-lane endpoint batch, built once ahead of the timed region. +/// Curve parameters and endpoint batches for gradient benchmarks. pub struct AffinityState { curve: AffinityCurve, from: Vec2x4T, @@ -118,7 +119,9 @@ pub fn affinity_state( } } -/// Four-lane attraction, as production calls it. +/// Evaluates and consumes a four-pair SIMD attraction update. +/// +/// The endpoints must meet [`AffinityCurve::attraction_x4`]'s numerical conditions. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -129,7 +132,9 @@ pub fn affinity_attraction_x4(state: &AffinityState) { black_box(black_box(state.curve).attraction_x4(black_box(state.from), black_box(state.to))); } -/// The four-lane attraction's scalar reference runs one lane at a time through the scalar kernel. +/// Evaluates and consumes scalar attraction updates for all four pairs. +/// +/// The endpoints must meet [`AffinityCurve::attraction`]'s numerical conditions. #[expect( clippy::inline_always, reason = "the benchmark must measure the reference as the target formulated it: transparently \ @@ -145,7 +150,13 @@ pub fn affinity_attraction_scalar_reference(state: &AffinityState) { })); } -/// Four-lane repulsion, as production calls it. +/// Evaluates and consumes a four-pair SIMD repulsion update. +/// +/// Endpoints must meet [`AffinityCurve::repulsion_x4`]'s numerical conditions. +/// +/// # Panics +/// +/// Panics if `repulsion_strength` is negative or non-finite. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -160,12 +171,13 @@ pub fn affinity_repulsion_x4(state: &AffinityState, repulsion_strength: f32) { )); } -/// The curve fit at one reference point, as production calls it. +/// Evaluates and consumes a fit for the supplied spread and minimum distance. +/// +/// The fit's optional result is consumed without requiring success. /// /// # Panics /// -/// This panics when either input is not finite and strictly positive, because the benchmark -/// synthesizes its own inputs and a degenerate one is a harness defect. +/// Panics when either input is not finite and strictly positive. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -181,7 +193,7 @@ pub fn affinity_fit(spread: f32, minimum_distance: f32) { )); } -/// A composed transform with one four-lane point batch, built once ahead of the timed region. +/// Affine coefficients and point batch for application benchmarks. pub struct TransformBatch { transform: Transform, batch: Vec2x4T, @@ -202,7 +214,7 @@ pub fn transform_batch( } } -/// Four-lane transform application, as production calls it. +/// Evaluates and consumes affine application to a four-point SIMD batch. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -213,7 +225,7 @@ pub fn transform_apply_x4(state: &TransformBatch) { black_box(black_box(state.transform).apply_x4(black_box(state.batch))); } -/// The four-lane application's scalar reference runs one lane at a time through the scalar kernel. +/// Evaluates and consumes scalar affine application to each point in the batch. #[expect( clippy::inline_always, reason = "the benchmark must measure the reference as the target formulated it: transparently \ @@ -226,34 +238,35 @@ pub fn transform_apply_scalar_reference(state: &TransformBatch) { })); } -/// A point corpus for the bounds and similarity kernels, built once ahead of the timed region. +/// Owned point fixture for bounds and similarity benchmarks. pub struct Points(Vec); impl Points { - /// The point count, for throughput declarations. + /// Returns the number of fixture points. #[must_use] pub const fn len(&self) -> usize { self.0.len() } - /// Whether the corpus is empty. + /// Returns whether the point fixture is empty. #[must_use] pub const fn is_empty(&self) -> bool { self.0.is_empty() } } -/// Deterministic 2D points spread over a non-degenerate box. +/// Builds a repeating sequence of finite points on the line y = 1000 − 2x. /// -/// # Panics -/// -/// The modulus bounds every index below `u16::MAX`, so the conversion inside never panics. +/// The sequence repeats every 40,000 points. A nonempty prefix lies on this line even when its +/// axis-aligned bounding box has area. The index modulus is below 40,000, making the conversion to +/// `u16` representable. #[expect( clippy::integer_division_remainder_used, reason = "the modulus is the fixture's deterministic spread rule, as the benchmark target \ wrote it" )] #[must_use] +#[expect(clippy::missing_panics_doc)] pub fn scattered_points(count: usize) -> Points { Points( (0..count) @@ -266,7 +279,7 @@ pub fn scattered_points(count: usize) -> Points { ) } -/// SIMD bounds over a slice, as production calls it. +/// Evaluates and consumes SIMD bounds over the fixture slice. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -277,7 +290,7 @@ pub fn bounds_from_slice(points: &Points) { black_box(Bounds2::from_slice(black_box(&points.0))); } -/// The bounds kernel's scalar reference is the point-iterator fold. +/// Evaluates and consumes bounds from the scalar point-iterator fold. #[expect( clippy::inline_always, reason = "the benchmark must measure the reference as the target formulated it: transparently \ @@ -288,7 +301,7 @@ pub fn bounds_from_points_scalar_reference(points: &Points) { black_box(Bounds2::from_points(black_box(&points.0).iter().copied())); } -/// Parallel SIMD bounds over a slice, as production calls it. +/// Evaluates and consumes parallel SIMD bounds over the fixture slice. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -299,20 +312,22 @@ pub fn bounds_from_slice_par(points: &Points) { black_box(Bounds2::from_slice_par(black_box(&points.0))); } -/// A weighted source/target correspondence for the similarity fit, built once ahead of the timed -/// region. +/// Owned point correspondences and weights for similarity-fitting benchmarks. pub struct SimilarityFixture { source: Vec, target: Vec, weights: Vec, } -/// Builds the correspondence: `count` scattered source points mapped through the reference -/// similarity given as its five-element array form, with unit weights. +/// Maps a collinear source fixture through a reference similarity with unit weights. +/// +/// `reference` uses [`Similarity::from_array`]'s [scale, cos, sin, x, y] order. Its rotation and +/// translation must meet that constructor's numerical contract. The target coordinates can still +/// overflow during application. /// /// # Panics /// -/// This panics when the reference array's scale is not normal and positive. +/// Panics when [`Similarity::from_array`] rejects the reference coefficients. #[must_use] pub fn similarity_fixture(count: usize, reference: [f32; 5]) -> SimilarityFixture { let Points(source) = scattered_points(count); @@ -328,20 +343,20 @@ pub fn similarity_fixture(count: usize, reference: [f32; 5]) -> SimilarityFixtur } impl SimilarityFixture { - /// The correspondence's point count, for throughput declarations. + /// Returns the number of point correspondences. #[must_use] pub const fn len(&self) -> usize { self.source.len() } - /// Whether the correspondence is empty. + /// Returns whether the correspondence fixture is empty. #[must_use] pub const fn is_empty(&self) -> bool { self.source.is_empty() } } -/// The weighted similarity fit, as production calls it. +/// Evaluates and consumes the serial weighted similarity fit. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -356,7 +371,7 @@ pub fn similarity_fit(fixture: &SimilarityFixture) { )); } -/// The parallel weighted similarity fit, as production calls it. +/// Evaluates and consumes the parallel weighted similarity fit. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -371,7 +386,7 @@ pub fn similarity_fit_par(fixture: &SimilarityFixture) { )); } -/// A logit vector for the softmax kernel, built once ahead of the timed region. +/// Fixed-size logit fixture for softmax benchmarks. pub struct Logits(DVecN); /// Builds the logit vector from plain components. @@ -380,7 +395,7 @@ pub fn logits(components: [f64; N]) -> Logits { Logits(DVecN::new(components)) } -/// Softmax, as production calls it. +/// Evaluates and consumes the vector's max-shifted softmax. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -392,17 +407,18 @@ pub fn dvecn_softmax(logits: &Logits) { } hashql_core::id::newtype! { - /// The bench fields' row domain. + /// Row identifiers for finite-field benchmark fixtures. + /// #[id(const)] pub struct BenchRowId(u32) } -/// A finite point field at one row count, built once ahead of the timed region. +/// Owned finite coordinates for field-validation benchmarks. pub struct FiniteField { points: Vec, } -/// Builds `rows` deterministic, sign-varying finite points. +/// Builds `rows` finite points from a repeating 256-value sequence. #[must_use] pub fn finite_field(rows: usize) -> FiniteField { let mut counter = 0_u8; @@ -419,7 +435,7 @@ pub fn finite_field(rows: usize) -> FiniteField { FiniteField { points } } -/// The finiteness scan, as the field's constructor runs it: serial, four points per batch. +/// Tests the fixture with the field constructor's serial finiteness scan. #[expect( clippy::inline_always, reason = "the benchmark must measure the kernel as production calls it: transparently \ @@ -431,9 +447,9 @@ pub fn finite_scan_serial(field: &FiniteField) -> bool { FinitePointField::new(IdSlice::::from_raw(black_box(&field.points))).is_ok() } -/// Rayon's per-point search for the first non-finite point. +/// Tests every point for finiteness with a parallel per-point search. /// -/// True exactly when the search comes back empty, so every point is finite. +/// Returns true exactly when every coordinate is finite. #[expect( clippy::inline_always, reason = "the benchmark must measure the reference transparently inlined, with only the \ @@ -448,9 +464,10 @@ pub fn finite_scan_per_point(field: &FiniteField) -> bool { .is_none() } -/// The serial scan's batch predicate distributed over rayon chunks of [`POINT_CHUNK`] points. +/// Tests finiteness with SIMD predicates over parallel point chunks. /// -/// True exactly when every chunk passes the batch predicate, so every point is finite. +/// Chunks contain at most [`POINT_CHUNK`] points. Each uses scalar alignment prefix/suffix checks +/// and four-point batch checks. Returns true exactly when every coordinate is finite. #[expect( clippy::inline_always, reason = "the benchmark must measure the reference formulation whole: transparently inlined, \ diff --git a/libs/@local/graph/atlas/src/math/bounds/mod.rs b/libs/@local/graph/atlas/src/math/bounds/mod.rs index 21207567f56..83d571d1c31 100644 --- a/libs/@local/graph/atlas/src/math/bounds/mod.rs +++ b/libs/@local/graph/atlas/src/math/bounds/mod.rs @@ -23,18 +23,20 @@ mod tests; /// An axis-aligned bounding box with finite, ordered corners. /// -/// The minimum and maximum corners define a [`Bounds2`]. Every value upholds two invariants: both -/// corners are finite, and `min ≤ max` holds per component. Constructors enforce this by returning -/// [`None`] for invalid input, so downstream code can rely on the box being usable without -/// re-validating. +/// Every value has finite minimum and maximum corners with `min ≤ max` per component. Constructors +/// return [`None`] for invalid input. /// -/// The primary workflow is: gather the extent of a point set with -/// [`from_points`](Self::from_points), then map the points onto a target region with -/// [`normalize_into`](Self::normalize_into), the per-axis affine box-to-box map. +/// Gather the extent of a point set with [`from_points`](Self::from_points), then map the points +/// onto a target region with [`normalize_into`](Self::normalize_into), the per-axis affine +/// box-to-box map. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{Bounds2, Vec2}; +/// /// let bounds = Bounds2::from_points([ /// Vec2::new(2.0, -1.0), /// Vec2::new(6.0, 3.0), @@ -47,9 +49,6 @@ mod tests; /// assert_eq!(bounds.size(), Vec2::new(4.0, 4.0)); /// assert_eq!(bounds.centre(), Vec2::new(4.0, 1.0)); /// ``` -// No `FromBytes`: it would construct boxes with NaN or inverted corners in -// safe code, bypassing the validating constructors. `FromZeros` is fine -// (the zeroed box is the valid degenerate box at the origin). #[derive( Debug, Copy, @@ -151,10 +150,10 @@ impl Bounds2 { /// or any non-finite coordinate. Rayon workers fold chunks of the slice with /// [`from_slice`](Self::from_slice), and [`union`](Self::union) combines the results. /// - /// The fold is memory-bound. A single core already streams near the machine's bandwidth, so the - /// parallel gain is real but modest (measured around a third at a million points) and does not - /// grow with core count. Below about a hundred thousand points, - /// [`from_slice`](Self::from_slice) is faster outright. + /// The recorded bounds benchmark measured a gain of around a third at a million points, + /// with the serial [`from_slice`](Self::from_slice) faster below about a hundred thousand. + /// These are machine-dependent crossover points. Measure with wall time when choosing + /// between the serial and parallel forms. /// /// Work splits into chunks of [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK) points. Use /// [`from_slice_par_with`](Self::from_slice_par_with) to choose a different chunk size. @@ -195,7 +194,8 @@ impl Bounds2 { /// Returns the per-axis extent, `max - min`. /// - /// Both components are non-negative by the type's invariant. + /// Both components are non-negative by the type's invariant. A difference can overflow to + /// positive infinity even though both corners are finite. #[inline] #[must_use] pub(crate) const fn size(self) -> Vec2 { @@ -203,6 +203,10 @@ impl Bounds2 { } /// Returns the centre of the box. + /// + /// Each component is the exact midpoint of its axis rounded once to the nearest `f32`. The + /// result is finite and lies inside the box: the midpoint lies between two `f32` corners and + /// rounding is monotone. #[inline] #[must_use] pub(crate) const fn centre(self) -> Vec2 { @@ -233,10 +237,14 @@ impl Bounds2 { /// Widens any axis narrower than `minimum` to exactly `minimum`. /// - /// Symmetrically around its centre. + /// Each narrow axis grows symmetrically in exact arithmetic. Rounding follows + /// [`with_aspect_ratio`](Self::with_aspect_ratio). An axis already at least `minimum` wide + /// keeps its corners bit for bit. /// /// This repairs degenerate boxes (all points on a line, or a single point) before operations /// that divide by the extent, such as box-to-box fitting or density rasterization. + /// + /// Returns [`None`] when a widened corner would lie beyond the finite `f32` range. #[inline] #[must_use] pub(crate) fn with_minimum_extent(self, minimum: f32) -> Self { @@ -251,20 +259,39 @@ impl Bounds2 { } } - /// Grows the box about its centre until its extent has the given width-to-height ratio. + /// Grows the shorter axis toward the given width-to-height ratio. + /// + /// The result contains this box. The axis already long enough for the ratio keeps its corners + /// bit for bit, and the other grows symmetrically in exact arithmetic. + /// + /// A viewport on a grid of square cells needs equal data extent per cell on both axes. Growing + /// to `across / down` supplies that ratio in exact arithmetic. The represented ratio also + /// depends on the corner spacing described below. + /// + /// An axis with no extent grows from the other axis's extent. A box degenerate on both axes + /// remains degenerate. [`with_minimum_extent`](Self::with_minimum_extent) supplies an extent. /// - /// The result is the smallest box of ratio `ratio` containing this one: the axis already long - /// enough keeps its extent, the other grows about the shared centre, and `size().x() / - /// size().y() = ratio` holds to within one rounding of the ratio itself. + /// Returns [`None`] when a grown corner would lie beyond the finite `f32` range. /// - /// This is the viewport operation for a grid of square cells. A point set drawn on `across` by - /// `down` cells keeps its own shape when this method grows its extent to `across / down` first, - /// because equal extent per cell on both axes is what one square cell means. Fitting the grown - /// box onto its target then scales both axes by the same factor. + /// # Rounding /// - /// An axis with no extent grows out of the other one, so a point set collapsed onto a line - /// still yields a viewport. A box degenerate on both axes has no extent to take a ratio of and - /// stays degenerate. [`with_minimum_extent`](Self::with_minimum_extent) gives it one. + /// This method, [`with_minimum_extent`](Self::with_minimum_extent) and + /// [`scaled_about_centre`](Self::scaled_about_centre) compute per-corner shifts in `f64`, then + /// round the low corner down and the high corner up to `f32`. A positive outward shift too + /// small to change the `f64` corner instead takes one outward `f32` step. Growth contains the + /// original axis and shrinking lies within it. A target equal to the current extent preserves + /// both corners bit for bit. + /// + /// Rounding can move the midpoint and change the achieved extent. Outward narrowing encloses + /// the computed `f64` corners, not necessarily the exact real-valued result: earlier `f64` + /// rounding can leave an extent slightly short of its target. + /// + /// # Warning + /// + /// Corner spacing can be comparable to the box's extent far from the origin. For example, + /// `[2²⁴, 0]..[2²⁴ + 2, 2]` grown to ratio 2 becomes `[2²⁴ - 1, 0]..[2²⁴ + 4, 2]`, with + /// ratio 2.5. The requested upper corner `2²⁴ + 3` lies between adjacent `f32` values. + /// Containment takes precedence over an exact ratio. /// /// # Examples /// @@ -274,7 +301,9 @@ impl Bounds2 { /// let ratio = Positive::new(4.0).expect("4 is positive"); /// /// // The box is 16 by 2, wider than 4:1, so the height grows to 4 and the width stays. - /// let viewport = bounds.with_aspect_ratio(ratio); + /// let viewport = bounds + /// .with_aspect_ratio(ratio) + /// .expect("the grown corners are far inside the `f32` range"); /// assert_eq!(viewport.size(), Vec2::new(16.0, 4.0)); /// assert_eq!(viewport.centre(), bounds.centre()); /// ``` @@ -299,18 +328,27 @@ impl Bounds2 { /// Scales the box about its centre by `factor`. /// - /// Both axes scale by the same factor and the centre does not move, so the result contains this - /// box for a factor above one and sits inside it below one. A factor of one returns the same - /// extent, to within the rounding of the halved arithmetic. + /// The result contains this box for a factor above one and lies within it for a factor below + /// one. A factor of one returns the box bit for bit. Rounding follows + /// [`with_aspect_ratio`](Self::with_aspect_ratio) and can leave an axis unchanged when its + /// corners are adjacent `f32` values. /// - /// # Examples + /// Returns [`None`] when a scaled corner would lie beyond the finite `f32` range. + /// + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{Bounds2, Positive, Vec2}; + /// /// let bounds = /// Bounds2::new(Vec2::splat(-1.0), Vec2::splat(1.0)).expect("corners are finite and ordered"); /// let margin = Positive::new(1.5).expect("1.5 is positive"); /// - /// let widened = bounds.scaled_about_centre(margin); + /// let widened = bounds + /// .scaled_about_centre(margin) + /// .expect("the scaled corners are far inside the `f32` range"); /// assert_eq!(widened.min(), Vec2::splat(-1.5)); /// assert_eq!(widened.max(), Vec2::splat(1.5)); /// ``` @@ -326,21 +364,26 @@ impl Bounds2 { } } - /// Returns the transform mapping this box onto `target`. + /// Fits an axis-aligned transform from this box to `target`. /// - /// The transform scales and translates each axis independently, so `self.min` lands on - /// `target.min` and `self.max` on `target.max`. This is the normalize-into-viewport operation: - /// fit a layout's extent, then map every point into `[0, size]` coordinates with one batched - /// transform. + /// In exact arithmetic, independent scale and translation map each source endpoint to the + /// corresponding target endpoint. Fit a layout's extent, then apply the transform in batches to + /// map points into viewport coordinates. The `f32` extents, scales and composed coefficients + /// round, and endpoint equality is not guaranteed. /// - /// Returns [`None`] when this box has an axis with zero, subnormal, or otherwise non-normal - /// extent, where the scale factor degenerates. Widen with - /// [`with_minimum_extent`](Self::with_minimum_extent) first when the point set may be - /// collinear. + /// Returns [`None`] when a source extent computed by [`Self::size`] is zero, subnormal or + /// infinite. Widen with [`Self::with_minimum_extent`] first when the point set may be + /// collinear. Target extents and computed coefficients are not validated: target-extent or + /// scale overflow can produce a transform containing infinities or NaNs. Use + /// [`Self::normalize_into`] for per-point mapping with widened arithmetic. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{Bounds2, Vec2}; + /// /// let layout = Bounds2::new(Vec2::new(-2.0, 0.0), Vec2::new(6.0, 4.0)) /// .expect("corners are finite and ordered"); /// let viewport = @@ -368,22 +411,30 @@ impl Bounds2 { ) } - /// Maps points from this box onto `target`, exactly per axis. + /// Maps points between boxes with per-axis double-precision arithmetic. /// - /// Each axis maps affinely in `f64` - subtract this box's minimum, divide by its extent, scale - /// onto the target axis - and rounds once to `f32`, so every output component is within one - /// `f32` ULP of the exact mapping for every input magnitude, including boxes sitting far from - /// the origin relative to their extent. This box's corners map onto the target's corners. - /// Points outside this box extrapolate along the same map. A zero-extent axis (every point - /// identical on it) maps to the centre of the target's axis. + /// For source axis [a, b] and target axis [c, d], the map is + /// c + (p − a) · (d − c)/(b − a) in exact arithmetic. A zero-extent source axis maps to the + /// target midpoint. Points outside this box extrapolate along the same map. /// - /// This method maps points in parallel, four at a time per axis: each batch converts to - /// [`Vec2x4T`] at the loop boundary and widens its lane groups to `f64`, so the batched and - /// scalar paths round identically and the output is independent of the split. + /// Coordinates widen before the differences are formed. Computing the unit coordinate + /// before target scaling avoids an `f32` scale-translation composition that loses small + /// offsets far from the origin. The `f64` operations still round before the final narrowing. + /// Cancellation near zero can magnify their error in output ULPs, and a rounded target + /// extent can prevent even a source endpoint from reaching its target endpoint exactly. + /// Extrapolated results can overflow to infinity. /// - /// # Examples + /// The parallel batch and scalar remainder use the same sequence of rounded operations. + /// Finite input points give the same results regardless of the split. NaN payloads are not + /// part of this agreement. + /// + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{Bounds2, Vec2}; + /// /// let layout = Bounds2::new(Vec2::new(-2.0, 0.0), Vec2::new(6.0, 4.0)) /// .expect("corners are finite and ordered"); /// let frame = Bounds2::new(Vec2::splat(-1.0), Vec2::splat(1.0)).expect("the frame is valid"); @@ -422,14 +473,24 @@ impl Bounds2 { impl Bounds2 { /// Quantizes a point onto the bounds' 32-bit-per-axis grid. /// - /// Each axis maps affinely onto `[0, 2^32)` in `f64` (so every `f32` coordinate quantizes - /// exactly) and floors. Coordinates outside the bounds clamp onto the boundary cells: points at - /// or beyond the maximum edge take the last cell, points below the minimum take cell zero. A - /// zero-extent axis maps to cell zero. + /// Divides the coordinate's offset from the minimum by the extent returned by [`Self::size`], + /// then scales by 2³². The offset and division use `f64`, but the extent is already rounded + /// to `f32`. The float-to-integer conversion truncates toward zero and saturates to the + /// `u32` range. NaN maps to cell zero. /// - /// # Examples + /// # Warning + /// + /// Rounding the extent can move a grid boundary, including making the maximum corner map + /// below the last cell. A zero or infinite extent maps every coordinate to cell zero. This + /// operation does not guarantee exact quantization of the real-valued box. + /// + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{Bounds2, Vec2}; + /// /// let bounds = Bounds2::new(Vec2::ZERO, Vec2::new(1.0, 1.0)).expect("the bounds are ordered"); /// assert_eq!(bounds.quantize(Vec2::ZERO), [0, 0]); /// assert_eq!(bounds.quantize(Vec2::new(1.0, 0.5)), [u32::MAX, 1 << 31]); @@ -463,9 +524,8 @@ fn quantize_axis(value: f32, min: f32, extent: f32) -> u32 { } let unit = (f64::from(value) - f64::from(min)) / f64::from(extent); - // Rust float-to-int casts saturate: negative inputs clamp to cell - // zero, inputs at or beyond the maximum edge to the last cell, and - // a NaN coordinate to cell zero. + // the cast saturates the computed scaled value: negatives and NaN become zero, and values at or + // above u32::MAX become the last cell. (unit * AXIS_CELLS) as u32 } @@ -480,6 +540,9 @@ struct AxisMap { } impl AxisMap { + /// Builds the map sending `[minimum, maximum]` onto `[target_minimum, target_maximum]`. + /// + /// Every input widens to `f64` before the extents and the target midpoint are formed. fn new(minimum: f32, maximum: f32, target_minimum: f32, target_maximum: f32) -> Self { let minimum = f64::from(minimum); let target_minimum = f64::from(target_minimum); @@ -505,8 +568,8 @@ impl AxisMap { } let unit = (f64::from(value) - self.minimum) / self.extent; - // f64 fused multiply-add is correctly rounded by IEEE 754, so it - // is both more accurate and byte-reproducible across targets. + // fusion rounds the target product-plus-sum once. The unit coordinate and target + // extent already carry any rounding from their construction. unit.mul_add(self.target_extent, self.target_minimum) as f32 } diff --git a/libs/@local/graph/atlas/src/math/bounds/tests.rs b/libs/@local/graph/atlas/src/math/bounds/tests.rs index c6c7d0979a0..570c76017f6 100644 --- a/libs/@local/graph/atlas/src/math/bounds/tests.rs +++ b/libs/@local/graph/atlas/src/math/bounds/tests.rs @@ -30,6 +30,7 @@ fn new_validates_corners() { assert!(Bounds2::new(Vec2::ZERO, Vec2::new(f32::INFINITY, 1.0)).is_none()); } +/// `from_points` over the shared fixture yields the tight minimum, maximum, size and centre. #[test] fn from_points_finds_tight_extent() { let bounds = Bounds2::from_points(POINTS).expect("points are finite and non-empty"); @@ -40,6 +41,7 @@ fn from_points_finds_tight_extent() { assert_eq!(bounds.centre(), Vec2::new(2.5, 6.5)); } +/// `from_points` returns `None` for no points and for any non-finite point wherever it sits. #[test] fn from_points_rejects_empty_and_non_finite() { assert!(Bounds2::from_points([]).is_none()); @@ -63,6 +65,7 @@ fn contains_is_boundary_inclusive() { assert!(!bounds.contains(Vec2::new(f32::NAN, 1.0))); } +/// `union` takes the componentwise minimum of the minima and maximum of the maxima. #[test] fn union_covers_both_operands() { let left = Bounds2::new(Vec2::new(-1.0, 0.0), Vec2::new(1.0, 2.0)) @@ -75,6 +78,8 @@ fn union_covers_both_operands() { assert_eq!(union.max(), Vec2::new(4.0, 2.0)); } +/// `with_minimum_extent` widens a zero-extent axis symmetrically about the centre and leaves an +/// axis already wider alone. #[test] fn minimum_extent_widens_degenerate_axes_only() { // All points on a vertical line: x extent is zero, y extent is 4. @@ -89,6 +94,8 @@ fn minimum_extent_widens_degenerate_axes_only() { assert_eq!(widened.max().x(), 4.0); } +/// `with_aspect_ratio` grows only the axis that is short for the ratio, keeping the centre: a +/// wide box gains height and a tall box gains width. #[test] fn aspect_ratio_grows_the_axis_that_is_short_for_it() { let wide = Bounds2::new(Vec2::new(-8.0, -1.0), Vec2::new(8.0, 1.0)) @@ -108,6 +115,8 @@ fn aspect_ratio_grows_the_axis_that_is_short_for_it() { assert_eq!(grown.centre(), tall.centre()); } +/// A zero-extent axis grows out of the other under `with_aspect_ratio`, and a single point stays +/// a point. #[test] fn aspect_ratio_takes_a_degenerate_axis_out_of_the_other() { let ratio = Positive::new(2.0).expect("2 is positive"); @@ -119,11 +128,13 @@ fn aspect_ratio_takes_a_degenerate_axis_out_of_the_other() { assert_eq!(grown.size(), Vec2::new(8.0, 4.0)); assert_eq!(grown.centre(), line.centre()); - // A single point has no extent to take a ratio of. + // A single point has no extent to take a ratio of, and comes back bit for bit. let point = Bounds2::new(Vec2::splat(3.0), Vec2::splat(3.0)).expect("a point is a valid box"); assert_eq!(point.with_aspect_ratio(ratio).size(), Vec2::ZERO); } +/// `scaled_about_centre` multiplies the size by the factor while keeping the centre, for factors +/// above and below one. #[test] fn scaling_about_the_centre_moves_both_corners() { let bounds = Bounds2::new(Vec2::new(0.0, 2.0), Vec2::new(4.0, 6.0)) @@ -144,8 +155,8 @@ fn scaling_about_the_centre_moves_both_corners() { fn quantize_maps_onto_the_axis_grid_and_clamps_outside_points() { let bounds = Bounds2::new(Vec2::ZERO, Vec2::new(1.0, 1.0)).expect("the corners are ordered"); - // The minimum corner takes cell zero; the maximum edge takes the last cell; the midpoint - // lands exactly on the middle cell because every `f32` coordinate quantizes exactly. + // the unit interval's midpoint scales to exactly 2³¹. The maximum endpoint clamps to + // `u32::MAX`, while the minimum maps to zero. assert_eq!(bounds.quantize(Vec2::ZERO), [0, 0]); assert_eq!(bounds.quantize(Vec2::new(1.0, 0.5)), [u32::MAX, 1 << 31]); @@ -180,6 +191,7 @@ fn from_slice_par_matches_serial() { assert_eq!(Bounds2::from_slice_par(&[]), None); } +/// A NaN deep in a later parallel chunk makes `from_slice_par` return `None`. #[test] fn from_slice_par_poisons_on_non_finite_in_any_chunk() { let mut points = scattered_points(10_000); @@ -189,6 +201,9 @@ fn from_slice_par_poisons_on_non_finite_in_any_chunk() { assert_eq!(Bounds2::from_slice_par(&points), None); } +/// `fit` maps the layout's corners and centre onto the viewport's within rounding, the batched +/// application agrees with the scalar one, and every mapped point lies inside the viewport up to +/// an ulp-scale margin. #[test] fn fit_maps_corners_onto_target() { let layout = Bounds2::from_points(POINTS).expect("points are finite and non-empty"); @@ -223,6 +238,8 @@ fn fit_maps_corners_onto_target() { } } +/// `fit` returns `None` for a zero-extent axis and `Some` once `with_minimum_extent` has widened +/// it. #[test] fn fit_rejects_degenerate_extents_until_widened() { let target = @@ -234,6 +251,8 @@ fn fit_rejects_degenerate_extents_until_widened() { assert!(collinear.with_minimum_extent(1.0).fit(target).is_some()); } +/// `normalize_into` lands the corners and centre exactly on the target's, since it computes the +/// unit coordinate before scaling. #[test] fn normalize_into_maps_corners_and_midpoints_exactly() { let layout = Bounds2::from_points(POINTS).expect("points are finite and non-empty"); @@ -247,6 +266,8 @@ fn normalize_into_maps_corners_and_midpoints_exactly() { assert_eq!(mapped, [Vec2::ZERO, Vec2::splat(10.0), Vec2::splat(5.0)]); } +/// `normalize_into` maps a zero-extent axis to the target's centre while the other axis maps +/// affinely. #[test] fn normalize_into_collapses_a_zero_extent_axis_to_the_target_centre() { let collinear = Bounds2::from_points([Vec2::new(3.0, 0.0), Vec2::new(3.0, 4.0)]) @@ -260,6 +281,8 @@ fn normalize_into_collapses_a_zero_extent_axis_to_the_target_centre() { assert_eq!(mapped, [Vec2::new(0.0, -0.5)]); } +/// A unit box at `2¹⁴` maps its quarter point exactly to `-0.5` through the per-axis `f64` map, +/// where an `f32` scale-translate composition would cancel. #[test] fn normalize_into_stays_exact_far_from_the_origin() { // A box sitting at 2^14 with unit extent: the world minimum dwarfs @@ -438,11 +461,6 @@ fn aspect_ratio_contains_the_box_and_holds_its_ratio( ); } -/// Scaling about the centre scales both extents by the factor and fixes the centre. -/// -/// A factor above one grows the box and one below shrinks it, so the same law states containment in -/// whichever direction the factor points. Tolerances scale with the scaled box's magnitude, since -/// that is what its corners were rebuilt from. #[property_test] fn scaling_about_the_centre_scales_both_extents( #[strategy = bounds_strategy()] bounds: Bounds2, @@ -491,6 +509,8 @@ mod miri { use super::scattered_points; use crate::math::{Bounds2, Vec2}; + /// `from_slice` equals `from_points` for lengths 0, 1, 3, 4, 5, 8 and 11, covering the empty, + /// remainder-only, exact-batch and mixed paths. #[test] fn from_slice_matches_from_points_for_every_remainder_length() { // Cover empty, remainder-only, exact-batch, and mixed lengths. @@ -505,6 +525,7 @@ mod miri { } } + /// `from_slice` equals `from_points` for every start offset across a batch stride. #[test] fn from_slice_matches_from_points_at_every_alignment_offset() { // Slide the slice start across a full batch stride so the split lands @@ -522,6 +543,7 @@ mod miri { } } + /// A NaN or infinity in the batched body or the remainder makes `from_slice` return `None`. #[test] fn from_slice_rejects_non_finite_in_batch_and_remainder() { // Position 2 falls in the batched body, position 9 in the remainder of an 11-point slice. diff --git a/libs/@local/graph/atlas/src/math/dsquare/mod.rs b/libs/@local/graph/atlas/src/math/dsquare/mod.rs index da5c80259d0..ab59c8ba588 100644 --- a/libs/@local/graph/atlas/src/math/dsquare/mod.rs +++ b/libs/@local/graph/atlas/src/math/dsquare/mod.rs @@ -3,26 +3,29 @@ //! [`DSquareMatrix`] holds an order × order matrix chosen at runtime in one SIMD-aligned //! allocation. Entries fill in place through [`row_mut`](DSquareMatrix::row_mut). //! [`DSquareMatrix::cholesky`] consumes the matrix and factors its lower triangle in place into the -//! lower-triangular [`DCholeskyFactor`] `L` with `A = L·Lᵀ`, and -//! [`solve_in_place`](DCholeskyFactor::solve_in_place) then answers `A·x = b` by forward and back -//! substitution. A matrix whose lower triangle is not positive-definite fails the factorization at -//! its first bad pivot with a [`DCholeskyError`]. +//! lower-triangular [`DCholeskyFactor`] L approximating A = L·Lᵀ. +//! [`solve_in_place`](DCholeskyFactor::solve_in_place) then solves A·x = b by forward and back +//! substitution, with floating-point rounding. Factorization returns [`DCholeskyError`] at its +//! first non-finite or nonpositive computed pivot. Rounding can cause even a positive-definite +//! input to be rejected. //! //! # Determinism //! -//! Every reduction folds in a fixed order that depends only on the operand lengths. Prefix dots -//! fold eight fused lanes at a time into two interleaved accumulators and finish with a scalar -//! tail. Factoring the same bytes therefore yields bit-identical factors. Solving with the same -//! factor and right-hand side yields bit-identical solutions. The kernels are single-threaded by -//! construction. +//! Prefix dots fold eight fused lanes at a time into two interleaved accumulators, reduce their +//! lane-wise sum, then finish with a scalar tail. The kernels are single-threaded. Block height +//! changes which entries are computed together, not the arithmetic within each entry. The final +//! lane-reduction order follows portable SIMD, and byte identity across targets or builds is not +//! guaranteed. //! //! # Layout //! -//! The constructor pads rows to a stride of whole [`f64x8`] lanes and aligns the allocation for -//! [`f64x8`], so every row starts at an aligned address. Row views carry that alignment as a type -//! invariant, so the kernels load whole aligned lanes. Padding components are `0.0` from -//! construction on and are never read as data: the triangular prefixes the factorization reduces -//! end mid-lane, so their tails fold scalarly instead of reading into the padding. +//! A stride of whole [`f64x8`] lanes preserves alignment from an aligned allocation base. The +//! constructor pads rows to that stride and aligns the allocation for `f64x8`. Every row starts at +//! an aligned address, which row views require as a type invariant for aligned lane loads. +//! +//! Padding components are `0.0` from construction on and are never read as data. The triangular +//! prefixes reduced during factorization can end mid-lane. Their tails fold scalarly over the +//! remaining data components, without reading the padding. use alloc::alloc::Global; use core::{ @@ -40,7 +43,13 @@ use super::kernel::mul_add_f64x8; #[cfg(test)] mod tests; -/// The row stride in components, the order rounded up to whole [`f64x8`] lanes. +/// Rounds the row order up to a multiple of eight components. +/// +/// `order` must not exceed `usize::MAX - 7`. +/// +/// # Panics +/// +/// Panics when `order > usize::MAX - 7` and integer overflow checking is enabled. const fn stride_for(order: usize) -> usize { order.next_multiple_of(8) } @@ -48,19 +57,18 @@ const fn stride_for(order: usize) -> usize { /// A lane-aligned view of a row, or row prefix, of the factorization's storage. /// /// Every row of a [`DSquareMatrix`] or [`DCholeskyFactor`] starts a whole number of [`f64x8`] lanes -/// into an allocation aligned for [`f64x8`], and a prefix shares its row's start; +/// into an allocation aligned for [`f64x8`], and a prefix shares its row's start. /// [`from_slice`](Self::from_slice) admits exactly such slices. [`lanes`](Self::lanes) therefore /// splits into aligned lane loads plus a scalar tail, with nothing in front. -// No byte-level constructors (zerocopy `FromBytes`): `transmute_ref!` could then mint views of +// No byte-level constructors (zerocopy `FromBytes`): `transmute_ref!` could then create views of // unaligned slices, bypassing the alignment invariant `from_slice` checks. #[repr(transparent)] struct DSquareRowBlock([f64]); impl DSquareRowBlock { - /// Wraps a slice starting at an address aligned for [`f64x8`]. + /// Borrows a slice whose start is aligned for [`f64x8`]. /// - /// Views come from rows of the aligned allocation and their prefixes; debug builds check the - /// address. + /// The caller must establish the alignment. // This is a safe fn because the alignment invariant guards which lane split `lanes` sees, a // correctness property rather than memory safety. #[inline] @@ -70,8 +78,9 @@ impl DSquareRowBlock { "a row view must start at an address aligned for f64x8" ); - // SAFETY: `Self` is a transparent wrapper around `[f64]`; the cast preserves the slice - // metadata. + // SAFETY: repr(transparent) preserves the slice's layout and validity. The cast retains its + // pointer, length and shared-borrow lifetime, and adds no mutation. Therefore the same + // initialized range may be borrowed as Self. unsafe { &*(ptr::from_ref(value) as *const Self) } } @@ -83,7 +92,7 @@ impl DSquareRowBlock { /// Returns the components as aligned 8-lane groups plus a scalar remainder. /// - /// Group `i` holds components `8 · i` through `8 · i + 7`; the remainder holds the trailing + /// Group `i` holds components `8 · i` through `8 · i + 7`. The remainder holds the trailing /// `len % 8` components. The alignment invariant means no components precede the groups. #[inline] fn lanes(&self) -> (&[f64x8], &[f64]) { @@ -96,11 +105,10 @@ impl DSquareRowBlock { (chunks, remainder) } - /// Returns the dot product of two equal-length views in a fixed fold order. + /// Returns the dot product of two equal-length views. /// - /// Fused products accumulate eight lanes at a time into two interleaved accumulators, and - /// the trailing `len % 8` components fold scalarly, so the summation order depends only on - /// the length. + /// Fused products accumulate into two interleaved eight-lane accumulators, followed by a lane + /// reduction and a fused scalar tail. The view lengths must match. #[inline] fn dot(&self, other: &Self) -> f64 { debug_assert_eq!(self.len(), other.len()); @@ -125,8 +133,8 @@ impl DSquareRowBlock { /// Returns the dot product with a plain slice, in the fold order of [`dot`](Self::dot). /// - /// `vector` may have any alignment: its lanes load component-wise while the view's load - /// aligned, and equal inputs reduce to identical bits through either dot. + /// `vector` must match the view's length and may have any alignment. It uses the same grouping + /// and fused operations as [`Self::dot`]. #[inline] fn dot_vector(&self, vector: &[f64]) -> f64 { debug_assert_eq!(self.len(), vector.len()); @@ -151,8 +159,8 @@ impl DSquareRowBlock { /// Subtracts `factor` times this view from `destination`, component-wise. /// - /// One fused multiply-add per component, eight lanes at a time with a scalar tail; the update - /// is elementwise, so no summation order exists. `destination` may have any alignment. + /// Uses one fused multiply-add per component. `destination` must match the view's length and + /// may have any alignment. #[inline] fn subtract_scaled(&self, destination: &mut [f64], factor: f64) { debug_assert_eq!(self.len(), destination.len()); @@ -178,13 +186,14 @@ impl DSquareRowBlock { /// attempts no perturbation or recovery. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum DCholeskyError { - /// The pivot is NaN or infinite: the fate of any non-finite component in the lower triangle's - /// rows up to and including `index`. + /// The computed pivot is NaN or infinite, from non-finite input or intermediate arithmetic. NonFinitePivot { /// The diagonal position of the first non-finite pivot. index: usize, }, - /// The pivot is finite but zero or negative: the lower triangle is not positive-definite. + /// The computed pivot is finite but zero or negative. + /// + /// This can reflect a non-positive-definite input or rounding in the factorization. NonPositivePivot { /// The diagonal position of the first non-positive pivot. index: usize, @@ -193,18 +202,19 @@ pub(crate) enum DCholeskyError { }, } -/// The active-block working set a panel pass keeps cache-resident, in bytes. -// Mid-plateau: at orders 1024-4096 every budget from 128 KiB to 1 MiB factors within -// measurement noise of the best, while 64 KiB collapses the larger orders to two-to-three-row -// blocks and loses the streamed-traffic reduction. A quarter MiB also sits inside any modern -// per-core private cache. +/// The target active-block working-set size for choosing a panel height, in bytes. +// A recorded sweep over orders 1024-4096 found budgets from 128 KiB to 1 MiB within measurement +// noise of the best. A 64 KiB budget reduced the larger orders to two- or three-row blocks and lost +// the streamed-traffic reduction. The selected 256 KiB is inside that measured interval, not a +// guarantee of cache residency on every processor. const BLOCK_BUDGET_BYTES: usize = 256 * 1024; /// The block height for `stride`: the tallest block whose rows fit the working-set budget. /// -/// Each settled row streams once per block, so the streamed traffic of the settled triangle falls -/// by the block height while the block's own rows stay cache-resident. A stride past the whole -/// budget degrades to single-row blocks: the unblocked row-wise algorithm. +/// The panel pass reuses each settled row across the active block. The budget targets locality of +/// the active rows. A row larger than the budget selects a single-row block. +/// +/// `stride · size_of::()` must fit `usize`. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -228,9 +238,13 @@ const fn block_rows_for(stride: usize) -> NonZero { /// [`cholesky`](Self::cholesky) consumes the matrix and factors it. Only the lower triangle is /// authoritative for the factorization, which ignores entries above the diagonal. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{DSquareMatrix}; +/// /// // A = [[4, 2], [2, 5]], written as its lower triangle. /// let mut matrix = DSquareMatrix::zeroed(2); /// matrix.row_mut(0)[0] = 4.0; @@ -250,12 +264,13 @@ pub(crate) struct DSquareMatrix { } impl DSquareMatrix { - /// Creates the zero matrix of the given order in a new aligned allocation in the global - /// allocator. + /// Allocates a zero matrix in the global allocator. + /// + /// See [`Self::zeroed_in`] for the order and allocation conditions. /// /// # Panics /// - /// This panics when the padded component count overflows the address space. + /// Panics under the same conditions as [`Self::zeroed_in`]. #[inline] #[must_use] pub(crate) fn zeroed(order: usize) -> Self { @@ -264,14 +279,16 @@ impl DSquareMatrix { } impl DSquareMatrix { - /// The allocation layout shared by the matrix and its factor. + /// Computes the allocation layout shared by the matrix and its factor. /// - /// `order · stride` components, padded to the alignment of [`f64x8`]. Allocation and - /// deallocation must agree on this. + /// Row padding must fit `usize`. Allocation and deallocation use the same layout, which covers + /// `order · stride` components with [`f64x8`] alignment without adding bytes to the component + /// count when raising the alignment. /// /// # Panics /// - /// This panics when the component count overflows the address space. + /// Panics if `order > usize::MAX - 7` with integer overflow checking enabled. Also panics if + /// the checked component-count multiplication or aligned layout construction fails. fn layout_for(order: usize) -> Layout { order .checked_mul(stride_for(order)) @@ -282,15 +299,16 @@ impl DSquareMatrix { ) } - /// Creates the zero matrix of the given order in a new aligned allocation in `alloc`. + /// Allocates a zero matrix with SIMD-aligned rows in `alloc`. /// - /// Every component is `0.0` and the buffer fills in place through [`row_mut`](Self::row_mut). - /// [`handle_alloc_error`](alloc::alloc::handle_alloc_error) aborts the process when the - /// allocator cannot provide the buffer. + /// Rounding `order` up to a multiple of eight must fit `usize`. Every component is `0.0`, and + /// [`Self::row_mut`] provides mutable row access. Allocation failure invokes + /// [`alloc::alloc::handle_alloc_error`]. /// /// # Panics /// - /// This panics when the padded component count overflows the address space. + /// Panics if `order > usize::MAX - 7` with integer overflow checking enabled. Also panics if + /// the checked padded component-count multiplication or aligned layout construction fails. #[inline] #[must_use] pub(crate) fn zeroed_in(order: usize, alloc: A) -> Self { @@ -319,17 +337,23 @@ impl DSquareMatrix { stride_for(self.order) } - /// The components as one row-major slice of `order · stride` components. + /// Borrows the row-major buffer, including padding. const fn components(&self) -> &[f64] { - // SAFETY: `ptr` owns an initialized buffer of `order · stride` components for as long as - // `self` lives. + // SAFETY: A raw slice requires an aligned non-null pointer to one initialized allocation + // with a representable byte length. layout_for checks this component count, and + // allocate_zeroed initializes every f64, including padding. The allocator supplies + // alignment even for zero length. The buffer remains owned and immutable through this + // shared borrow. Therefore the slice is valid for the borrow's lifetime. unsafe { slice::from_raw_parts(self.ptr.as_ptr(), self.order * self.stride()) } } - /// The components as one mutable row-major slice of `order · stride` components. + /// Mutably borrows the row-major buffer, including padding. const fn components_mut(&mut self) -> &mut [f64] { - // SAFETY: `ptr` owns an initialized buffer of `order · stride` components for as long as - // `self` lives. The exclusive borrow of `self` guards the exclusive reference. + // SAFETY: A mutable raw slice additionally requires exclusive access. The constructor + // supplies an aligned non-null pointer and initializes the complete layout-checked + // component range, including the zero-length case. This exclusive Self borrow excludes + // other buffer access and bounds the slice's lifetime. Therefore the mutable slice is + // valid. unsafe { slice::from_raw_parts_mut(self.ptr.as_ptr(), self.order * self.stride()) } } @@ -373,14 +397,18 @@ impl DSquareMatrix { /// /// The factorization reads only the lower triangle: entry `(i, j)` with `j ≤ i` is `A[i][j]`, /// and it ignores the strict upper triangle. The returned factor owns the same allocation and - /// holds `L` with `A = L·Lᵀ`, zeros above the diagonal, and the padding untouched. + /// holds a rounded factor L approximating A = L·Lᵀ, with zeros above the diagonal and the + /// padding untouched. The zero-order matrix returns an empty factor. /// /// # Errors /// - /// [`DCholeskyError::NonFinitePivot`] when a pivot is NaN or infinite, the fate of any - /// non-finite value in the lower triangle; [`DCholeskyError::NonPositivePivot`] when a finite - /// pivot is zero or negative, meaning the lower triangle is not positive-definite. The - /// factorization stops at the first bad pivot. + /// Returns [`DCholeskyError`] at the first non-finite or nonpositive computed pivot. A + /// positive-definite input can fail when rounding removes a small positive pivot. + /// + /// # Complexity + /// + /// Takes O(n³) arithmetic operations for order n and constant additional storage beyond the + /// owned matrix. #[inline] pub(crate) fn cholesky(self) -> Result, DCholeskyError> { let block_height = block_rows_for(stride_for(self.order)); @@ -389,11 +417,9 @@ impl DSquareMatrix { /// Factors like [`cholesky`](Self::cholesky) with an explicit block height. /// - /// The factor's bytes are identical at every block height, because every entry is the same - /// prefix-dot expression regardless of the blocking; the height chooses only how much of the - /// active triangle stays cache-resident per pass. [`cholesky`](Self::cholesky) derives the - /// height that fits the working-set budget, and this form takes the height directly, so a - /// caller can cross block boundaries at any order. + /// Every entry uses the same prefix-dot expression regardless of blocking. The height controls + /// row reuse and the active working-set size. It does not change the within-entry grouping of + /// floating-point operations. /// /// # Errors /// @@ -404,10 +430,13 @@ impl DSquareMatrix { ) -> Result, DCholeskyError> { self.factorize(block_height)?; - // The factor takes over the allocation; skipping the matrix's drop keeps ownership - // unique. + // transfer allocation ownership to the factor without running the matrix's destructor. let matrix = ManuallyDrop::new(self); - // SAFETY: `ManuallyDrop` skips the matrix's drop, so the allocator moves out exactly once. + // SAFETY: ptr::read requires an aligned initialized value, and ownership of a non-Copy + // result must not be duplicated. matrix.alloc is initialized and addressable, while + // ManuallyDrop suppresses its original destruction. No panicking operation follows before + // the factor takes the pointer, order and allocator. Therefore the read transfers the + // allocator's ownership exactly once. let alloc = unsafe { ptr::read(&raw const matrix.alloc) }; Ok(DCholeskyFactor { ptr: matrix.ptr, @@ -416,17 +445,22 @@ impl DSquareMatrix { }) } - /// Factors the lower triangle in place into `L` with `A = L·Lᵀ`. + /// Computes the rounded Cholesky factor in the lower triangle. /// /// Row-wise Cholesky: `L[i][j] = (A[i][j] − Σ_{p) -> Result<(), DCholeskyError> { let order = self.order; let stride = self.stride(); @@ -502,8 +536,10 @@ impl fmt::Debug for DSquareMatrix { impl Drop for DSquareMatrix { #[inline] fn drop(&mut self) { - // SAFETY: `alloc` allocated `ptr` in `zeroed_in` with the same order-derived layout and - // nothing has deallocated it since. + // SAFETY: Deallocation requires the allocator and layout of a live allocation. zeroed_in + // stores both the allocating instance and the pointer, and order never changes. This owner + // has not transferred its buffer to a factor or deallocated it. Therefore this destructor + // may release the allocation with the original layout. unsafe { self.alloc .deallocate(self.ptr.cast::(), Self::layout_for(self.order)); @@ -511,19 +547,23 @@ impl Drop for DSquareMatrix { } } -// SAFETY: the matrix owns its buffer exclusively and its `f64` components are `Send`. The -// allocator's own thread-safety carries the bound. +// SAFETY: Send permits transferring ownership between threads. The matrix exclusively owns its f64 +// buffer, whose contents have no thread affinity, and A: Send permits moving the allocating +// instance with it. Borrowed views prevent moving the owner while in use. Therefore transferring +// the matrix preserves exclusive ownership and its deallocation capability. unsafe impl Send for DSquareMatrix {} -// SAFETY: shared access hands out only `&[f64]`-shaped views of the owned buffer with no interior -// mutability. The allocator's own thread-safety carries the bound. +// SAFETY: Sync requires shared access to avoid data races. Shared matrix methods expose immutable +// f64 views without interior mutation, and A: Sync covers sharing the allocator. Writes and +// deallocation require exclusive ownership. Therefore shared matrix references are safe across +// threads. unsafe impl Sync for DSquareMatrix {} /// The lower-triangular Cholesky factor `L` of a factored [`DSquareMatrix`]. /// /// The factor owns the allocation of the matrix that produced it: row `i` holds `L[i][0..=i]` -/// followed by zeros, and `L·Lᵀ` recovers the factored matrix's lower triangle. -/// [`solve_in_place`](Self::solve_in_place) answers `A·x = b` for the factored `A`. +/// followed by zeros. The product L·Lᵀ approximates the matrix represented by the input's lower +/// triangle. [`Self::solve_in_place`] uses this rounded factor to solve a linear system. pub(crate) struct DCholeskyFactor { ptr: NonNull, order: usize, @@ -543,14 +583,25 @@ impl DCholeskyFactor { stride_for(self.order) } - /// The components as one row-major slice of `order · stride` components. + /// Borrows the factor's row-major buffer, including padding. const fn components(&self) -> &[f64] { - // SAFETY: `ptr` owns an initialized buffer of `order · stride` components for as long as - // `self` lives. + // SAFETY: A raw slice requires an aligned non-null pointer and an initialized range within + // one allocation. The factor inherits the matrix's layout-checked buffer and unchanged + // order. Factorization writes valid f64 values without changing its extent, and the + // allocator supplied alignment even for zero length. This shared borrow retains ownership + // and forbids mutation. Therefore the slice is valid for its lifetime. unsafe { slice::from_raw_parts(self.ptr.as_ptr(), self.order * self.stride()) } } - /// Row `index` of the factor as its `order` components. + /// Returns row `index` of the factor as its `order` components. + /// + /// `index` must be less than the factor's order. + /// + /// # Panics + /// + /// Panics if the computed row offset exceeds the buffer length or leaves fewer than `order` + /// components. With integer overflow checking enabled, also panics if the row-offset + /// multiplication overflows. const fn row(&self, index: usize) -> &[f64] { &self.components()[index * self.stride()..][..self.order] } @@ -559,11 +610,15 @@ impl DCholeskyFactor { /// /// `vector` enters as the right-hand side `b` and leaves as the solution `x`. Forward /// substitution solves `L·y = b` top-down, each component a prefix dot of the factor row with - /// the settled solution prefix; back substitution solves `Lᵀ·x = y` bottom-up, each settled - /// component removing its column's contribution from the equations above it - a column of `Lᵀ` - /// is a row of `L`, so both passes read the factor along its rows. The factor's rows load as - /// aligned lanes. `vector` may have any alignment. The solution's bytes depend only on the - /// factor's and right-hand side's bytes. + /// the settled solution prefix. Back substitution solves `Lᵀ·x = y` bottom-up, each settled + /// component removing its column's contribution from the equations above it. Reading columns of + /// `Lᵀ` as rows of `L` gives both passes row-wise access to the factor. `vector` may have any + /// alignment. Arithmetic rounds in `f64`, and a non-finite right-hand side or intermediate can + /// produce a non-finite solution. + /// + /// # Complexity + /// + /// Takes O(n²) arithmetic operations for order n and constant additional storage. /// /// # Panics /// @@ -609,9 +664,11 @@ impl fmt::Debug for DCholeskyFactor { impl Drop for DCholeskyFactor { #[inline] fn drop(&mut self) { - // SAFETY: `alloc` allocated `ptr` in `DSquareMatrix::zeroed_in` with the same order-derived - // layout. `cholesky` moved ownership of both here and skipped the matrix's drop, so - // no other deallocation happens. + // SAFETY: Deallocation requires a live allocation with the original allocator and layout. + // cholesky transfers the pointer, unchanged order and allocator from the matrix while + // suppressing its destructor. This factor has neither transferred nor released that + // ownership. Therefore it may deallocate the buffer exactly once with the matrix's original + // layout. unsafe { self.alloc.deallocate( self.ptr.cast::(), @@ -621,10 +678,13 @@ impl Drop for DCholeskyFactor { } } -// SAFETY: the factor owns its buffer exclusively and its `f64` components are `Send`. The -// allocator's own thread-safety carries the bound. +// SAFETY: Send permits transferring ownership between threads. The factor exclusively owns its f64 +// buffer, and A: Send permits moving the allocating instance with it. Borrowed views prevent moving +// the owner while in use. Therefore the factor and its deallocation capability may be transferred +// together. unsafe impl Send for DCholeskyFactor {} -// SAFETY: shared access hands out only `&[f64]`-shaped views of the owned buffer with no interior -// mutability. The allocator's own thread-safety carries the bound. +// SAFETY: Sync requires shared access to avoid data races. The factor is immutable after +// construction, and solving writes only to the separately borrowed right-hand side. A: Sync covers +// sharing the allocator. Therefore shared factor references are safe across threads. unsafe impl Sync for DCholeskyFactor {} diff --git a/libs/@local/graph/atlas/src/math/dsquare/tests.rs b/libs/@local/graph/atlas/src/math/dsquare/tests.rs index c015a79169c..82b9adb7a8a 100644 --- a/libs/@local/graph/atlas/src/math/dsquare/tests.rs +++ b/libs/@local/graph/atlas/src/math/dsquare/tests.rs @@ -9,15 +9,25 @@ use super::{DCholeskyError, DSquareMatrix}; -/// A deterministic integer pattern in `[−5, 5]`. +/// Returns a deterministic integer pattern in [−5, 5]. +/// +/// # Panics +/// +/// Panics if `row * 31 + column * 17 + 5` overflows when overflow checks are enabled. fn pattern(row: usize, column: usize) -> f64 { ((row * 31 + column * 17 + 5) % 11) as f64 - 5.0 } -/// The exact entry `A[i][j] = Σ_k G[k][i]·G[k][j] + [i = j]·order` of `A = GᵀG + order·I`. +/// Computes an entry of the fixture A = `GᵀG` + order · I. +/// +/// With G[k][i] given by [`pattern`], A[i][j] = Σₖ G[k][i] · G[k][j] + [i = j] · order. Products +/// have magnitude at most 25. For the small orders used here, 26 · order < 2⁵³ bounds every +/// intermediate integer, making the entry exact. For positive order, `GᵀG` is positive-semidefinite +/// and gives λₘᵢₙ ≥ order and λₘₐₓ ≤ trace(A). +/// +/// # Panics /// -/// Every term is a small integer, so the sums stay far below 2⁵³ and the fixture is exactly -/// reproducible. `GᵀG` is positive-semidefinite, so `λ_min ≥ order` and `λ_max ≤ trace(A)`. +/// Panics on index arithmetic overflow in [`pattern`] when overflow checks are enabled. fn fixture_entry(order: usize, row: usize, column: usize) -> f64 { let products: f64 = (0..order) .map(|index| pattern(index, row) * pattern(index, column)) @@ -30,7 +40,12 @@ fn fixture_entry(order: usize, row: usize, column: usize) -> f64 { } } -/// The fixture `A = GᵀG + order·I`, written into the lower triangle only. +/// Writes the fixture A = `GᵀG` + order · I into a matrix's lower triangle. +/// +/// # Panics +/// +/// Panics if [`DSquareMatrix::zeroed`] cannot represent the layout or [`fixture_entry`] overflows +/// its index arithmetic. fn spd_fixture(order: usize) -> DSquareMatrix { let mut matrix = DSquareMatrix::zeroed(order); for row in 0..order { @@ -42,9 +57,11 @@ fn spd_fixture(order: usize) -> DSquareMatrix { matrix } -/// The orders the certificates cover: 1, 2, a padded stride (7), whole lanes (8), and four lanes -/// with a tail (33). Block-boundary crossings are [`block_height_invariance`]'s subject, which -/// drives the block height directly. +/// Matrix orders covering minimal, padded-stride and whole-lane geometries. +/// +/// Orders 7 and 33 require row padding. Order 8 occupies one full eight-lane group, while 33 has +/// four groups and a one-component tail. [`miri::block_height_invariance`] varies the block height +/// separately. const ORDERS: [usize; 5] = [1, 2, 7, 8, 33]; #[test] @@ -54,9 +71,10 @@ fn factor_times_its_transpose_recovers_the_lower_triangle() { .cholesky() .expect("the fixture is positive-definite"); - // In IEEE arithmetic the factor satisfies A − L·Lᵀ = ΔA with - // |ΔA[i][j]| ≤ c·(order + 1)·ε·‖Lᵢ‖·‖Lⱼ‖, and ‖Lᵢ‖² = A[i][i], so the largest diagonal - // entry bounds every ‖Lᵢ‖·‖Lⱼ‖. Margin 8 absorbs the constant c. + // Cholesky reconstruction error scales with reduction length and products of row norms. For + // an exact factor, ‖Lᵢ‖² = A[i][i], making the largest diagonal a natural input scale. The + // test allows 8 · (order + 1) · ε times that scale for factorization and reconstruction + // rounding. let max_diagonal = (0..order) .map(|index| fixture_entry(order, index, index)) .fold(0.0_f64, f64::max); @@ -197,11 +215,11 @@ fn the_strict_upper_triangle_never_reaches_the_factor() { } } -/// A 2 × 2 factorization whose every intermediate is exact. +/// Checks the printed rows of an exactly factored 2 × 2 matrix. /// -/// `A = [[4, 2], [2, 5]]` factors as `L = [[2, 0], [1, 2]]`: `√4`, `2/2`, and `√(5 − 1)` all -/// land on representable integers. The `Debug` forms print the matrix's rows and the factor's -/// lower triangle. +/// A = [[4, 2], [2, 5]] factors as L = [[2, 0], [1, 2]]: √4, 2/2 and √(5 − 1) are all representable +/// integers. The [`Debug`](core::fmt::Debug) forms print the matrix's rows and the factor's lower +/// triangle. #[test] fn exact_factor_reports_its_order_and_debug_forms() { let mut matrix = DSquareMatrix::zeroed(2); @@ -214,18 +232,11 @@ fn exact_factor_reports_its_order_and_debug_forms() { assert_eq!(format!("{factor:?}"), "[[2.0], [1.0, 2.0]]"); } -/// The derived block height splits a production-scale order into more than one block. #[test] fn block_rows_for_production_order() { assert!(super::block_rows_for(super::stride_for(200)).get() < 200); } -/// The tests the `miri` nextest profile selects. -/// -/// Each test here drives the dense square matrix through its row storage. That covers the -/// alignment and offsets of the rows, the dot reductions, factor and solve, and the return of the -/// buffer to its allocator on drop. The profile selects by module path, so moving a test in or out -/// of this module is the whole edit. mod miri { use core::simd::f64x8; @@ -247,10 +258,9 @@ mod miri { let mut solution: Vec = (0..order).map(|index| pattern(index, 3)).collect(); factor.solve_in_place(&mut solution); - // Factor-and-substitute is backward stable: (A + ΔA)·x̂ = b with - // |ΔA[i][j]| ≤ c·order·ε·(max diagonal), so each residual component obeys - // |A·x̂ − b|ᵢ ≤ c·order·ε·(max diagonal)·Σⱼ|x̂ⱼ|. The residual recomputation below rounds - // at the same order. Margin 8 absorbs both constants. + // If (A + ΔA)·x̂ = b, then |A·x̂ − b|ᵢ ≤ Σⱼ |ΔA[i][j]| · |x̂ⱼ|. The test uses 8 · order · + // ε · max diagonal as its entrywise error allowance and multiplies by Σⱼ |x̂ⱼ|. This + // scales the residual tolerance with both input magnitude and the computed solution. let max_diagonal = (0..order) .map(|index| fixture_entry(order, index, index)) .fold(0.0_f64, f64::max); @@ -270,12 +280,10 @@ mod miri { } } - /// The factor's bytes are identical at every block height. + /// Compares selected block heights against a single-block factorization. /// - /// Heights below the order make the panel pass cross block boundaries, and the derived height - /// at this order is the single-block path, so the sweep compares every crossing shape - /// against the unblocked reference. The order is small enough for Miri to interpret the - /// crossings. + /// The derived height at order 13 uses one block. Heights 1, 2, 3, 5 and 8 require panel + /// updates across block boundaries while retaining the arithmetic order within each entry. #[test] fn block_height_invariance() { const ORDER: usize = 13; @@ -306,9 +314,8 @@ mod miri { #[test] fn both_dots_reduce_equal_inputs_to_identical_bits() { - // The operands come from two aligned matrix rows; the shifted copy hands the plain-slice - // dot a start the lane loads cannot assume. Lengths sweep zero, mid-lane tails, and - // whole lanes. + // the shifted copy supplies a slice without the matrix row's alignment guarantee. Prefix + // lengths cover the empty, partial-lane and whole-lane paths. const ORDER: usize = 40; let mut storage = DSquareMatrix::zeroed(ORDER); for column in 0..ORDER { @@ -339,10 +346,8 @@ mod miri { #[test] fn the_row_dot_and_dvecn_dot_reduce_equal_bytes_to_identical_bits() { - // The module doc ties the prefix dots to the fold shape of `DVecN::dot`; the test above and - // dvecn's aligned-reduction test guard each family internally, and this pins the families - // to each other. Length 29 makes three eight-lane folds - the interleave visits - // both accumulators, unevenly - and a five-component scalar tail. + // length 29 makes three eight-lane groups and a five-component tail. The interleaved fold + // updates accumulator zero twice and accumulator one once. const LENGTH: usize = 29; let mut storage = DSquareMatrix::zeroed(LENGTH); for column in 0..LENGTH { @@ -435,7 +440,6 @@ mod miri { factor.solve_in_place(&mut []); } - /// Dropping a matrix returns its buffer to the allocator that provided it. #[test] fn matrix_drop_returns_the_buffer_to_its_allocator() { let alloc = CountingAllocator::new(); @@ -447,7 +451,6 @@ mod miri { assert_eq!(alloc.deallocations(), 1); } - /// The factorization moves the buffer, and dropping the factor returns it exactly once. #[test] fn factor_drop_returns_the_moved_buffer_to_its_allocator() { let alloc = CountingAllocator::new(); diff --git a/libs/@local/graph/atlas/src/math/dvec2/mod.rs b/libs/@local/graph/atlas/src/math/dvec2/mod.rs index 5e956230552..ae39b798d28 100644 --- a/libs/@local/graph/atlas/src/math/dvec2/mod.rs +++ b/libs/@local/graph/atlas/src/math/dvec2/mod.rs @@ -18,18 +18,20 @@ mod tests; /// A 2D vector of `f64` components, for accumulating over [`Vec2`] data. /// -/// A [`DVec2`] is the double-precision accumulator twin of [`Vec2`]: sums of weighted points, -/// centroids, and moment corrections live here while a reduction runs, then narrow back to the -/// working precision once at the end via [`narrow`](Self::narrow). Widening a [`Vec2`] through -/// [`From`] is exact for every value, so per-component products of widened inputs carry no `f32` -/// rounding. +/// Accumulate weighted points and moment corrections in double precision, then convert to [`Vec2`] +/// through [`Self::narrow`]. Widening finite `f32` components is exact. The accumulation still +/// rounds in `f64`, but it avoids rounding each update to `f32`. /// -/// The surface is the accumulator's own: arithmetic, the two products, and the exact widening and -/// checked narrowing conversions. Geometry (interpolation, clamping, bounds) belongs to [`Vec2`]. +/// Components may be non-finite. The product methods return a [`Derivation`] whose final value can +/// be validated before use. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{DVec2, Vec2}; +/// /// // Accumulate a weighted centroid in double precision. /// let points = [Vec2::new(1.0, 2.0), Vec2::new(3.0, -2.0)]; /// let mut sum = DVec2::ZERO; @@ -81,7 +83,7 @@ impl DVec2 { self.0[1] } - /// The shared raw dot fold under [`dot`](Self::dot) and [`norm_squared`](Self::norm_squared). + /// Computes the dot product with one rounded product and one fused multiply-add. #[inline] fn dot_impl(self, other: Self) -> f64 { self.x().mul_add(other.x(), self.y() * other.y()) @@ -89,8 +91,8 @@ impl DVec2 { /// Returns the dot product of the two vectors. /// - /// The exponents compose, so the fold rides as an unclaimed derivation to its consumer's - /// own finish. + /// Returns an unvalidated [`Derivation`]. Products and sums of arbitrary `f64` components can + /// be non-finite. #[inline] pub(crate) fn dot(self, other: Self) -> Derivation { Derivation::raw(self.dot_impl(other)) @@ -98,7 +100,9 @@ impl DVec2 { /// Returns the perpendicular dot product, the `z` component of the 3D cross product. /// - /// The sign semantics match [`Vec2::perp_dot`]. + /// Approximates x₁y₂ − y₁x₂ with a rounded y₁x₂ product followed by a fused multiply-add. + /// Cancellation can change the sign near zero or leave a nonzero value for parallel inputs. Use + /// a predicate with a guaranteed orientation sign when orientation decides topology. #[inline] pub(crate) fn perp_dot(self, other: Self) -> Derivation { Derivation::raw(self.x().mul_add(other.y(), -(self.y() * other.x()))) @@ -132,7 +136,7 @@ impl DVec2 { /// Narrows both components to the working precision. /// - /// Returns [`None`] when either component leaves the finite `f32` range, following + /// Returns [`None`] when either component is NaN or rounds to an infinity, following /// [`narrow_f32`]. #[inline] #[must_use] @@ -147,8 +151,7 @@ impl DVec2 { Some(Vec2::new(x, y)) } - /// Narrows both components to the working precision, with round-to-nearest and no - /// finiteness check. + /// Narrows both components with round-to-nearest, allowing non-finite results. /// /// A component beyond the finite `f32` range overflows to `±∞` rather than refusing. A /// caller that must reject an out-of-range component calls [`narrow`](Self::narrow) instead, @@ -164,9 +167,6 @@ impl DVec2 { } } -/// Widens both components. -/// -/// The conversion is exact for every [`Vec2`]. const impl From for DVec2 { #[inline] fn from(vec: Vec2) -> Self { @@ -246,17 +246,18 @@ const impl Neg for DVec2 { /// Four double-precision 2D vectors packed in transposed (structure-of-arrays) order. /// -/// The `f64` twin of [`Vec2x4T`]: all four `x` values followed by all four `y` values, aligned for -/// [`Simd`](Simd). The surface is fold-shaped - widen a [`Vec2x4T`] batch through [`From`] -/// (exact for every component), form lane-wise products, accumulate with -/// [`mul_add`](Self::mul_add), and terminally [`reduce_sum`](Self::reduce_sum) to a [`DVec2`] - -/// because the -/// type exists for double-precision moment accumulation over batches of working-precision points. +/// All four x values precede all four y values, in storage aligned for [`Simd`](Simd). +/// Widen a [`Vec2x4T`] batch, accumulate weighted moments through [`Self::mul_add`], then combine +/// the lanes through [`Self::reduce_sum`]. Widening finite `f32` components is exact. +/// +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private and uses nightly portable SIMD. /// /// ```ignore /// # #![feature(portable_simd)] +/// use crate::math::{DVec2, DVec2x4T, Vec2, Vec2x4T}; +/// /// # use core::simd::Simd; /// /// let batch = DVec2x4T::from(Vec2x4T::from([ @@ -268,7 +269,7 @@ const impl Neg for DVec2 { /// /// // Accumulate the weighted sum per lane, then reduce once. /// let weighted = batch.mul_add(Simd::splat(0.5), DVec2x4T::ZERO); -/// assert_eq!(weighted.reduce(), DVec2::new(5.0, 13.0)); +/// assert_eq!(weighted.reduce_sum(), DVec2::new(5.0, 13.0)); /// ``` #[derive( Debug, @@ -290,8 +291,7 @@ impl DVec2x4T { /// Creates a batch holding four copies of `vec`. /// - /// Every lane of the `x` group holds `vec.x()` and every lane of the `y` group holds - /// `vec.y()`, so one point compares against a whole batch lane-wise. + /// Repeating the point supports lane-wise comparison against four distinct points. #[inline] #[must_use] pub const fn splat(vec: DVec2) -> Self { @@ -315,10 +315,11 @@ impl DVec2x4T { let this = &raw const *self; let this = this.cast::(); - // SAFETY: `Self` is `repr(C)` over `[f64; 8]` whose first four elements are the `x` - // lane group, `Simd` is layout-compatible with `[f64; 4]`, and `Self`'s - // 64-byte alignment satisfies `Simd`'s (const-asserted below); the borrow - // covers bytes owned by `self` and inherits its lifetime. + // SAFETY: The cast relies on Simd's contiguous array-element layout. Self's repr(C) storage + // has four initialized x components at offset zero. Its 64-byte alignment and the + // assertions below cover SIMD alignment, and the pointer retains the shared borrow's + // provenance and lifetime. Under that layout contract, this group may be borrowed as + // Simd. unsafe { &*this.cast::>() } } @@ -336,10 +337,11 @@ impl DVec2x4T { let this = &raw const *self; let this = this.cast::(); - // SAFETY: elements `4..8` of `Self`'s `repr(C)` `[f64; 8]` storage are the `y` lane - // group; the 32-byte offset from the 64-byte-aligned base satisfies `Simd`'s - // alignment (const-asserted below), and the borrow covers bytes owned by `self` and - // inherits its lifetime. + // SAFETY: The cast relies on Simd's contiguous array-element layout. Self's repr(C) storage + // has four initialized y components at byte offset 32. The asserted SIMD alignment divides + // that offset and the 64-byte base alignment. Pointer addition remains within Self, + // preserving the shared borrow's provenance and lifetime. Under that layout contract, this + // group may be borrowed as Simd. unsafe { &*this.add(4).cast::>() } } @@ -355,9 +357,10 @@ impl DVec2x4T { reason = "the suggested `From` conversion is not const-callable" )] pub const fn into_lanes(self) -> (Simd, Simd) { - // SAFETY: `Self` is `repr(C)` over `[f64; 8]`, the `x` lane group followed by the `y` - // lane group, exactly `[Simd; 2]`'s memory order; sizes match and every bit - // pattern is a valid `f64`. + // SAFETY: This transmute relies on each SIMD vector having its array's element layout + // without padding. Self contains initialized x then y groups in repr(C) storage, and + // transmute checks equality of the complete sizes. Every component bit pattern is valid as + // f64. Under that SIMD layout contract, both destination groups are initialized and valid. let [xs, ys] = unsafe { core::mem::transmute::; 2]>(self) }; (xs, ys) @@ -370,30 +373,33 @@ impl DVec2x4T { #[must_use] pub const fn from_lanes(xs: Simd, ys: Simd) -> Self { let this = [xs, ys]; - // SAFETY: `[Simd; 2]` lays out the `x` lane group followed by the `y` lane - // group, exactly `Self`'s `repr(C)` `[f64; 8]` memory order; sizes match and every - // bit pattern is a valid `f64`. + // SAFETY: This transmute relies on each SIMD vector having its array's element layout + // without padding. The source array places the initialized x group before y, matching + // Self's repr(C) component order, and transmute checks equal sizes. Under that layout + // contract, all destination components are initialized and valid. unsafe { core::mem::transmute::<[Simd; 2], Self>(this) } } /// Returns all eight components as a single SIMD vector. /// - /// The lane order is the memory order: `x0 x1 x2 x3 y0 y1 y2 y3`. This compiles to a single - /// full-width vector load. + /// The lane order is `x0 x1 x2 x3 y0 y1 y2 y3`. #[inline] #[must_use] pub const fn to_simd(self) -> Simd { - // SAFETY: `Self` is `repr(C)` over `[f64; 8]`, which is layout-compatible with - // `Simd` (sizes const-asserted below); every bit pattern is a valid `f64`, so - // the reinterpretation is total. + // SAFETY: This transmute relies on Simd's contiguous array-element layout. Self has eight + // initialized f64 components in lane order, and the assertion below checks equal size. + // Under that layout contract, those components form a valid SIMD value. unsafe { core::mem::transmute::>(self) } } /// Returns the four pairwise dot products as SIMD lanes. /// - /// Lane `i` holds `self[i] . other[i]`. On targets with native FMA the multiply-add fuses. For - /// components widened from `f32` both lane products are exact, so the fused and separate forms - /// agree bit for bit and the result carries a single rounding either way. + /// Lane i computes aₓbₓ + aᵧbᵧ for the corresponding pair of vectors. The y product rounds + /// first, followed by a fused multiply-add for the x product and sum. + /// + /// Products of finite `f32` values need at most 48 significand bits and remain within the `f64` + /// exponent range. When both vectors are widened from finite `f32` components, both products + /// are therefore exact and only the final addition rounds. #[inline] #[must_use] pub fn dot(self, other: Self) -> Simd { @@ -402,8 +408,8 @@ impl DVec2x4T { /// Returns the four pairwise perpendicular dot products as SIMD lanes. /// - /// Lane `i` holds `self[i].perp_dot(other[i])`, with the sign semantics of [`Vec2::perp_dot`]. - /// The rounding behaviour is [`dot`](Self::dot)'s. + /// Each lane uses [`DVec2::perp_dot`]'s expression and numerical limits. For arbitrary `f64` + /// components, cancellation can leave a nonzero result even for parallel inputs. #[inline] #[must_use] pub fn perp_dot(self, other: Self) -> Simd { @@ -419,12 +425,10 @@ impl DVec2x4T { /// Returns the four pairwise squared Euclidean distances. /// - /// Component `i` holds `self[i].distance_squared(other[i])`. The subtraction and squaring run - /// at full batch width, and only the final add combines the axis halves. Every operation - /// rounds separately, exactly as the scalar form does, so each component agrees with - /// [`DVec2::distance_squared`] bit for bit on every input. Fusing the multiply-add would - /// change roundings and break that equality, so this kernel deliberately stays unfused, - /// unlike [`dot`](Self::dot). + /// Component `i` holds `self[i].distance_squared(other[i])`. Separate subtraction, squaring and + /// addition preserve the rounding sequence of [`DVec2::distance_squared`]. Fusing the final + /// multiply-add can change the result even for coordinates widened from `f32`. NaN payload + /// equality is not guaranteed. #[inline] #[must_use] pub fn distance_squared(self, other: Self) -> DVecN<4> { @@ -457,7 +461,9 @@ impl DVec2x4T { ) } - /// Sums the four vectors into one [`DVec2`]: the terminal reduction of an accumulation. + /// Sums the four vectors into one [`DVec2`]. + /// + /// The final bits can depend on the SIMD reduction's summation order. #[inline] #[must_use] pub fn reduce_sum(self) -> DVec2 { @@ -465,9 +471,6 @@ impl DVec2x4T { } } -/// Widens every component. -/// -/// The conversion is exact for every [`Vec2x4T`]. impl From for DVec2x4T { #[inline] fn from(batch: Vec2x4T) -> Self { @@ -475,7 +478,6 @@ impl From for DVec2x4T { } } -/// Widens four whole vectors: one interleave shuffle, then the exact per-component widening. impl From<[Vec2; 4]> for DVec2x4T { #[inline] fn from(vecs: [Vec2; 4]) -> Self { @@ -483,7 +485,6 @@ impl From<[Vec2; 4]> for DVec2x4T { } } -/// Adds the batches vector by vector: the unweighted accumulation step. impl Add for DVec2x4T { type Output = Self; @@ -500,7 +501,6 @@ impl AddAssign for DVec2x4T { } } -/// Subtracts the batches vector by vector: the deviation-from-centre step. impl Sub for DVec2x4T { type Output = Self; @@ -511,12 +511,12 @@ impl Sub for DVec2x4T { } const impl From> for DVec2x4T { - /// Reinterprets eight lanes in `x0 x1 x2 x3 y0 y1 y2 y3` order. #[inline] fn from(lanes: Simd) -> Self { - // SAFETY: `Simd` is layout-compatible with `[f64; 8]`, `Self`'s `repr(C)` - // storage (sizes const-asserted below); every bit pattern is a valid `f64`, so the - // reinterpretation is total. + // SAFETY: This transmute relies on Simd's contiguous array-element layout without padding. + // Self stores eight f64 components in the same order, with no additional validity + // conditions, and equal sizes are asserted below. Under that layout contract, the + // initialized lanes form a valid batch. unsafe { core::mem::transmute::, DVec2x4T>(lanes) } } } @@ -528,9 +528,9 @@ const impl From for Simd { } } -// The batch must back `Simd` (identical size, at least its alignment), and the lane -// views borrow `Simd` groups at offsets 0 and 32, so the half-width alignment must -// not exceed the offset. +// The batch must match `Simd`'s size and meet its alignment. Borrowed `Simd` groups +// begin at byte offsets 0 and 32. Their alignment must not exceed 32 bytes to keep both group +// addresses aligned. const _: () = assert!(size_of::() == size_of::>()); const _: () = assert!(align_of::() >= align_of::>()); const _: () = assert!(align_of::>() <= 32); diff --git a/libs/@local/graph/atlas/src/math/dvec2/tests.rs b/libs/@local/graph/atlas/src/math/dvec2/tests.rs index 6580da2eb25..03151d13407 100644 --- a/libs/@local/graph/atlas/src/math/dvec2/tests.rs +++ b/libs/@local/graph/atlas/src/math/dvec2/tests.rs @@ -73,12 +73,11 @@ fn batch_of(vectors: [DVec2; 4]) -> DVec2x4T { ) } -/// Four full-mantissa vector pairs on which the fused and separate distance forms disagree. +/// Returns vector pairs whose squared distances distinguish fused from separate evaluation. /// -/// Points widened from `f32` cannot discriminate: their lane differences and squares are exact -/// in `f64`, so a fused mutant agrees with the unfused kernel on that whole domain. These pairs -/// come from a search for the property that `dx.mul_add(dx, dy * dy)` differs from -/// `dx * dx + dy * dy` in every lane. +/// Every pair gives different values for `dx.mul_add(dx, dy * dy)` and `dx * dx + dy * dy`. +/// Widening `f32` coordinates does not generally make their squared differences exact either: dx = +/// 1 − 2⁻²⁷ and dy = 2⁻²⁷ already distinguish the two evaluations. fn distance_pairs() -> ([DVec2; 4], [DVec2; 4]) { ( [ @@ -135,7 +134,6 @@ fn products_refine_the_f32_counterparts( prop_assert!((narrow_dot - wide_dot).abs() <= tolerance); } -/// The lane metric is the scalar metric, bit for bit, on every input. #[property_test] fn distance_squared_lanes_match_the_scalar_metric_bitwise( #[strategy = -1e150_f64..1e150] ax: f64, @@ -151,10 +149,6 @@ fn distance_squared_lanes_match_the_scalar_metric_bitwise( prop_assert_eq!(distances, DVecN::new([source.distance_squared(target); 4])); } -/// The tests the `miri` nextest profile selects. -/// -/// Each test here runs the transposed double-precision batch beside its scalar twin, lane by lane. -/// The profile selects by module path, so moving a test in or out of this module is the whole edit. mod miri { use super::{batch_of, batch_points, distance_pairs}; use crate::math::{DVec2, DVec2x4T, Vec2, Vec2x4T}; @@ -186,8 +180,7 @@ mod miri { let perp_dot = source.perp_dot(target); let length_squared = source.length_squared(); - // The lane kernels share `DVec2`'s fused shape, so the paths agree - // bit for bit on every input. + // scalar and lane products use the same fused arithmetic on these finite inputs for index in 0..4 { let source = DVec2::from(sources[index]); let target = DVec2::from(targets[index]); @@ -234,8 +227,7 @@ mod miri { fn dvec2x4t_distance_squared_matches_the_scalar_twin_per_lane() { let (sources, targets) = distance_pairs(); - // The fixture can fail under fusion: every lane's fused form disagrees with the separate - // form, so a `mul_add` mutant in the kernel dies in all four lanes. + // each pair distinguishes separate rounding from a fused multiply-add for (source, target) in sources.iter().zip(&targets) { let dx = source.x() - target.x(); let dy = source.y() - target.y(); diff --git a/libs/@local/graph/atlas/src/math/dvecn/mod.rs b/libs/@local/graph/atlas/src/math/dvecn/mod.rs index c1241787383..f8aaf2e6e15 100644 --- a/libs/@local/graph/atlas/src/math/dvecn/mod.rs +++ b/libs/@local/graph/atlas/src/math/dvecn/mod.rs @@ -1,15 +1,12 @@ //! Double-precision `N`-dimensional vectors and their reductions. //! -//! [`DVecN`] is the `f64` twin of [`VecN`], for the few consumers whose algorithms demand -//! double precision throughout, such as classifier logits feeding the bounded trust-region -//! exact-Newton solver. -//! Its reductions ([`softmax`](DVecN::softmax), [`log_sum_exp`](DVecN::log_sum_exp)) shift, -//! exponentiate, and fold four lanes at a time. The exponential goes through -//! [`kernel::exp_f64x4`](super::kernel), which currently lowers to one libm call per lane. +//! [`DVecN`] supports arithmetic and stable reductions in double precision. [`BoxedDVecN`] owns +//! heap storage aligned for [`f64x8`], exposed through [`AlignedDVecN`] views. Use it for large +//! vectors that need in-place initialization or aligned lane access. //! -//! [`BoxedDVecN`] owns a heap allocation aligned for [`f64x8`] and hands out [`AlignedDVecN`] -//! references to it, mirroring [`BoxedVecN`](super::BoxedVecN): the storage for optimizer state - -//! parameter and gradient vectors - whose dimension is far too large for the stack. +//! The reductions use floating-point sums with rounding at each accumulation step. Aligned and +//! ordinary views use matching lane groups, but portable-SIMD horizontal reductions do not +//! establish a cross-target or cross-build bit-identity guarantee. use alloc::alloc::Global; use core::{ @@ -31,12 +28,17 @@ mod tests; /// An `N`-dimensional vector of `f64` components. /// -/// A [`DVecN`] is guaranteed to have the same layout as `[f64; N]`, so borrowed arrays convert in -/// place through [`from_ref`](Self::from_ref) and [`from_mut`](Self::from_mut) without copying. +/// A [`DVecN`] is guaranteed to have the same layout as an array of `N` `f64` components. Borrow +/// arrays in place through [`from_ref`](Self::from_ref) and [`from_mut`](Self::from_mut), without +/// copying. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{DVecN}; +/// /// let logits = DVecN::new([2.0, 1.0, -1.0]); /// /// let probabilities = logits.softmax(); @@ -77,8 +79,9 @@ impl DVecN { #[must_use] pub const fn from_mut(value: &mut [f64; N]) -> &mut Self { let ptr = (&raw mut *value).cast::(); - // SAFETY: `Self` is a transparent wrapper around `[f64; N]`, so the cast preserves layout - // and validity. The mutable borrow passes through to the wrapper unchanged. + // SAFETY: repr(transparent) preserves the array's layout and validity. The input reference + // supplies initialized components, alignment and exclusive access, and the cast retains its + // provenance and lifetime. Therefore the same array may be borrowed mutably as Self. unsafe { &mut *ptr } } @@ -91,8 +94,8 @@ impl DVecN { /// Returns the largest component. /// - /// NaN components lose, following [`f64::max`]; the maximum of the empty vector is - /// [`f64::NEG_INFINITY`], the identity of the fold. + /// Ignores NaN components, following [`f64::max`]. Returns [`f64::NEG_INFINITY`] for an empty + /// vector or one containing only NaNs. #[inline] #[must_use] pub fn max(self) -> f64 { @@ -130,15 +133,22 @@ impl DVecN { /// Computes the softmax of the components with max-shifting for stability. /// - /// Subtracting the maximum component before exponentiation keeps the result finite for any - /// finite input, including components with magnitudes far beyond the range where a naive `exp` - /// overflows. Every output lies in `[0, 1]`, the outputs sum to 1 up to rounding whenever `N ≥ - /// 1`, and shifting all components by a common constant leaves the result unchanged up to - /// rounding. For `N = 0` the result is the empty vector. + /// For finite components xᵢ, let m = maxᵢ xᵢ and eᵢ = exp(xᵢ − m). The result approximates eᵢ / + /// Σⱼ eⱼ. Max-shifting keeps the exponential arguments nonpositive, avoiding overflow from + /// exponentiating a large positive component directly. Outputs lie in `[0, 1]` and sum to one + /// up to rounding. For `N = 0` the result is empty. + /// + /// Adding a common constant preserves the real-valued formula. In floating-point arithmetic, a + /// large shift can round distinct components to the same value and change the distribution. + /// Non-finite inputs can produce NaN outputs. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{DVecN}; + /// /// // A naive `exp(1000.0)` overflows. The shifted form stays finite. /// let probabilities = DVecN::new([1_000.0, 999.0, -1_000.0]).softmax(); /// @@ -162,9 +172,13 @@ impl DVecN { /// components give `value + ln(N)`. For `N = 0` the result is [`f64::NEG_INFINITY`], the /// logarithm of the empty sum. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{DVecN}; + /// /// // A naive `exp(1000.0)` overflows. The shifted form stays finite. /// let result = DVecN::new([1_000.0, 1_000.0]).log_sum_exp(); /// assert!((result - (1_000.0 + 2.0_f64.ln())).abs() < 1e-9); @@ -177,15 +191,15 @@ impl DVecN { let maximum = self.max(); let (_, sum) = self.shifted_exponentials(maximum); - // The empty case needs no branch. The fold leaves the maximum at negative infinity while - // the empty sum is zero, and `ln(0)` is negative infinity, so the two addends agree on the - // empty-sum identity. + // For an empty vector, the maximum is −∞ and the exponential sum is zero. The final + // expression is −∞ + ln(0) = −∞, the logarithm of the empty sum. maximum + sum.ln() } /// Computes `exp(component - shift)` for every component and their sum in a single pass. /// - /// Processes four lanes at a time. + /// Uses [`exp_f64x4`] for complete four-lane groups and [`f64::exp`] for the scalar remainder. + /// These approximations can round differently. #[inline] #[must_use] fn shifted_exponentials(mut self, shift: f64) -> (Self, f64) { @@ -209,14 +223,11 @@ impl DVecN { (self, sum) } - /// The shared raw fold under [`dot`](Self::dot) and [`norm_squared`](Self::norm_squared). + /// Accumulates the dot product in two interleaved eight-lane groups. /// - /// The kernel fuses and sums the products eight lanes at a time. See [`VecN::dot_wide`] for - /// the mixed-precision variant over `f32` data. - // Lane-width choice: as in `VecN::dot_accumulated` - `f64x8` is a - // fourfold unroll on 128-bit NEON, and two independent accumulators - // keep enough FMA chains in flight to cover the latency-throughput - // product. + /// Fused products accumulate per lane, followed by a horizontal sum and a fused scalar tail. + /// See [`VecN::dot_wide`] for the mixed-precision variant over `f32` data. + // the independent accumulators reduce serial dependence between successive lane-group updates #[inline] fn dot_impl(&self, other: &Self) -> f64 { let (chunks_left, remainder_left) = self.0.as_chunks::<8>(); @@ -246,8 +257,8 @@ impl DVecN { /// Returns the dot product of the two vectors. /// - /// The exponents compose, so the fold rides as an unclaimed derivation to its consumer's - /// own finish. + /// Returns an unvalidated [`Derivation`]. Products and sums of arbitrary `f64` components can + /// be non-finite. #[inline] pub(crate) fn dot(&self, other: &Self) -> Derivation { Derivation::raw(self.dot_impl(other)) @@ -282,9 +293,8 @@ impl DVecN { /// Returns the largest component magnitude, or `0.0` for the empty vector. /// - /// Folds `simd_max` over the absolute lanes and finishes with the scalar remainder. The - /// maximum follows IEEE-754 `maxNum`: the fold ignores NaN components in favor of any finite - /// magnitude, so callers that must reject NaN check [`is_finite`](Self::is_finite) first. + /// Follows IEEE-754 `maxNum` semantics, ignoring NaN components in favor of finite magnitudes. + /// If you must reject NaN, check [`is_finite`](Self::is_finite) first. #[inline] #[must_use] pub fn max_abs(&self) -> f64 { @@ -305,23 +315,21 @@ impl DVecN { /// Returns the Euclidean norm through a scaled two-pass sum of squares. /// - /// The first pass takes the largest magnitude as the scale ([`max_abs`](Self::max_abs)); the - /// second divides every component by it (one division each - no reciprocal, so every ratio - /// lies in `[0, 1]` exactly) and accumulates the squared ratios eight fused lanes at a time - /// into two interleaved accumulators. The result is `scale · √Σratio²`: - /// subnormal components keep their norm and magnitudes near [`f64::MAX`] stay finite where - /// naive squared accumulation would overflow. This is the norm kernel of solvers whose control - /// decisions must survive extreme scales. + /// For finite components xᵢ and scale s = maxᵢ |xᵢ| > 0, computes s · √Σᵢ (xᵢ / s)². Each ratio + /// lies in [−1, 1] after rounding, preventing overflow of its square. Scaling also avoids + /// losing an entire subnormal-only vector when direct squaring would underflow. Small relative + /// contributions can still round away, and the final multiplication can overflow when the norm + /// is too large. /// /// The all-zero and empty vectors have norm `0.0`. A vector containing NaN or an infinity - /// yields a non-finite result: infinities force a NaN or infinite product through the second - /// pass, and the zero-scale finiteness check catches a NaN alongside only zeros. + /// yields a non-finite result. Division by the scale avoids an overflowing reciprocal when the + /// scale is subnormal. #[inline] #[must_use] pub fn stable_l2(&self) -> f64 { let scale = self.max_abs(); if scale == 0.0 { - // maxNum ignores NaN, so a zero scale still needs the finiteness check. + // maxNum can give a zero scale for a mixture of zeros and NaNs return if self.is_finite() { 0.0 } else { f64::NAN }; } @@ -347,9 +355,7 @@ impl DVecN { /// Returns whether every component is finite. /// - /// The lane groups accumulate one finiteness mask with a single horizontal test at the end: - /// the all-finite case - the expected case - runs branch-free at load - /// bandwidth, measured ~12% faster than a short-circuiting scan on the same vector. + /// Returns `true` for an empty vector. #[inline] #[must_use] pub fn is_finite(&self) -> bool { @@ -359,6 +365,8 @@ impl DVecN { return false; } + // one mask avoids short-circuiting within the lane groups. A recorded all-finite-vector + // comparison measured this scan about 12% faster than a short-circuiting scan. let mut finite = Mask::splat(true); for chunk in chunks { finite &= f64x8::from_array(*chunk).is_finite(); @@ -369,9 +377,8 @@ impl DVecN { /// Adds a working-precision vector, component-wise. /// - /// Each `f32` component of `rhs` widens to `f64` exactly, so the update carries only the - /// rounding of the addition itself. This is the moment-accumulation kernel of statistics kept - /// in double precision over single-precision data. + /// Widening `rhs` to `f64` adds no numeric rounding. The update carries only the addition's + /// rounding, supporting double-precision moment accumulation over single-precision data. #[inline] pub fn add_widened(&mut self, rhs: &VecN) { let (chunks, remainder) = self.0.as_chunks_mut::<8>(); @@ -388,10 +395,9 @@ impl DVecN { /// Adds `factor` times a working-precision vector, component-wise. /// - /// Each `f32` component of `direction` widens to `f64` exactly, so the update `self += - /// direction * factor` carries only the rounding of the fused multiply-add itself. This is the - /// gradient-accumulation kernel of optimizers that keep their state in double precision over - /// single-precision data. + /// Widening `direction` to `f64` adds no numeric rounding. The update `self += direction * + /// factor` carries only one fused multiply-add rounding per component, supporting + /// double-precision gradient accumulation over single-precision data. #[inline] pub fn add_scaled(&mut self, direction: &VecN, factor: f64) { let scale = f64x8::splat(factor); @@ -473,10 +479,9 @@ impl DVecN { /// Adds the squared deviation of a working-precision vector from `mean`, component-wise. /// - /// Each `f32` component of `value` widens to `f64` exactly, so the update `self += (value - - /// mean)^2` carries only the rounding of the subtraction and the fused multiply-add. This is - /// the second-moment kernel of diagonal-variance fits kept in double precision over - /// single-precision data. + /// Widening `value` to `f64` adds no numeric rounding. The update `self += (value - mean)^2` + /// carries only the subtraction and fused multiply-add roundings, supporting double-precision + /// second-moment accumulation over single-precision data. #[inline] pub fn add_squared_deviation(&mut self, value: &VecN, mean: &Self) { let (chunks, remainder) = self.0.as_chunks_mut::<8>(); @@ -537,8 +542,8 @@ const impl From> for [f64; N] { /// `align_of::()`. The transparent layout means any array that happens to be aligned can be /// wrapped in place. /// -/// The payoff is [`lanes`](Self::lanes): every 8-lane load comes from an aligned address, so -/// iteration over the vector never splits a cache line. +/// [`Self::lanes`] splits the components into aligned eight-lane groups and a scalar remainder, +/// with no prefix before the lane groups. // No `FromBytes`/`FromZeros`: a byte-level constructor would let // `zerocopy::transmute_ref!` produce references to unaligned arrays, // bypassing the alignment invariant. @@ -558,8 +563,10 @@ impl AlignedDVecN { #[inline] #[must_use] pub const unsafe fn from_ref_unchecked(value: &[f64; N]) -> &Self { - // SAFETY: `Self` is a transparent wrapper around `[f64; N]`, and the alignment invariant is - // the caller's contract. + // SAFETY: repr(transparent) preserves the array's layout and validity. The input reference + // supplies initialized storage and a shared-borrow lifetime, while the caller supplies the + // stronger f64x8 alignment. The cast retains the pointer and borrow. Therefore the result + // is a valid aligned view for that lifetime. unsafe { &*ptr::from_ref(value).cast::() } } @@ -572,8 +579,10 @@ impl AlignedDVecN { #[inline] #[must_use] pub const unsafe fn from_mut_unchecked(value: &mut [f64; N]) -> &mut Self { - // SAFETY: `Self` is a transparent wrapper around `[f64; N]`, and the alignment invariant is - // the caller's contract. + // SAFETY: repr(transparent) preserves the array's layout and validity. The input reference + // supplies initialized storage and exclusive access, while the caller supplies the stronger + // f64x8 alignment. The cast retains the pointer and mutable-borrow lifetime. Therefore the + // result is a valid exclusive aligned view. unsafe { &mut *ptr::from_mut(value).cast::() } } @@ -588,7 +597,9 @@ impl AlignedDVecN { return None; } - // SAFETY: the early return above rejects unaligned input. + // SAFETY: from_ref_unchecked requires f64x8 alignment. The preceding check establishes it + // for this array's starting address. Therefore the shared array borrow satisfies the + // constructor's contract. unsafe { Some(Self::from_ref_unchecked(value)) } } @@ -603,7 +614,9 @@ impl AlignedDVecN { return None; } - // SAFETY: the early return above rejects unaligned input. + // SAFETY: from_mut_unchecked requires f64x8 alignment. The preceding check establishes it + // for this array's starting address. Therefore the exclusive array borrow satisfies the + // constructor's contract. unsafe { Some(Self::from_mut_unchecked(value)) } } @@ -625,8 +638,7 @@ impl AlignedDVecN { /// /// The split is [`AlignedVecN::lanes`](super::AlignedVecN::lanes) at double precision: group /// `i` holds components `8 · i` through `8 · i + 7`, and the remainder holds the trailing `N % - /// 8` components. The type's alignment invariant guarantees no misaligned prefix exists, so no - /// components precede the groups. + /// 8` components. The type's alignment invariant excludes a misaligned prefix. #[inline] #[must_use] pub fn lanes(&self) -> (&[f64x8], &[f64]) { @@ -642,7 +654,7 @@ impl AlignedDVecN { /// Returns the components as mutable aligned 8-lane groups plus a mutable scalar remainder. /// - /// The split is the same as [`lanes`](Self::lanes); writes through either slice update the + /// The split is the same as [`lanes`](Self::lanes). Writes through either slice update the /// vector in place. #[inline] #[must_use] @@ -657,11 +669,10 @@ impl AlignedDVecN { (lanes, suffix) } - // Arithmetic kernels over the lane view. Alignment is part of the type, so every group - // loads and stores as one aligned `f64x8` and the remainder follows in order. Fold shapes - // match the `DVecN` kernels exactly: the aligned allocation splits at the same 8-lane - // boundary, so both types reduce identical inputs to identical bits. + // matching DVecN's eight-component groups keeps the same accumulation expressions for aligned + // and ordinary storage + /// Accumulates the dot product with the grouping used by [`DVecN::dot`]. #[inline] #[must_use] fn dot_impl(&self, other: &Self) -> f64 { @@ -694,9 +705,9 @@ impl AlignedDVecN { /// Returns the dot product when the result is finite. /// - /// Returns [`None`] when the reduced value is not finite. A non-finite value entering the - /// fold can only produce a non-finite accumulator, so checking the result covers every - /// component and intermediate. + /// A non-finite value entering this multiply-add fold can only produce a non-finite + /// accumulator. Every component and computed intermediate contributes to the final reduction. + /// Returning [`None`] for a non-finite result therefore covers those inputs and intermediates. #[inline] pub(crate) fn checked_dot(&self, other: &Self) -> Option { DFinite::new(self.dot_impl(other)) @@ -716,9 +727,8 @@ impl AlignedDVecN { /// Returns the squared Euclidean length when the result is finite. /// - /// The self-dot, with the fold shape and finiteness refusal of - /// [`checked_dot`](Self::checked_dot). A sum of squares is non-negative, so the reading - /// carries that domain. + /// Uses the fold shape and finiteness refusal of [`checked_dot`](Self::checked_dot). A finite + /// sum of squares is non-negative. #[inline] pub(crate) fn checked_norm_squared(&self) -> Option { DNonNegative::new(self.dot_impl(self)) @@ -749,9 +759,9 @@ impl AlignedDVecN { /// Returns the largest component magnitude, or `0.0` for the empty vector. /// - /// The maximum follows IEEE-754 `maxNum` exactly as [`DVecN::max_abs`]: the fold ignores NaN - /// components in favor of any finite magnitude, so callers that must reject NaN check - /// [`is_finite`](Self::is_finite) first. + /// Follows IEEE-754 `maxNum` semantics exactly as [`DVecN::max_abs`], ignoring NaN components + /// in favor of finite magnitudes. If you must reject NaN, check [`is_finite`](Self::is_finite) + /// first. #[inline] #[must_use] pub fn max_abs(&self) -> f64 { @@ -770,10 +780,16 @@ impl AlignedDVecN { scale } + /// The raw scaled two-pass norm behind [`stable_l2`](Self::stable_l2). + /// + /// For a finite vector with nonzero maximum magnitude, dividing by that magnitude bounds every + /// ratio to `[-1, 1]`. This method divides every component by [`max_abs`](Self::max_abs) before + /// squaring. Therefore no square overflows on that domain. A zero scale returns `0.0` for a + /// finite vector and NaN otherwise. fn stable_l2_impl(&self) -> f64 { let scale = self.max_abs(); if scale == 0.0 { - // maxNum ignores NaN, so a zero scale still needs the finiteness check. + // maxNum can give a zero scale for a mixture of zeros and NaNs return if self.is_finite() { 0.0 } else { f64::NAN }; } @@ -797,21 +813,22 @@ impl AlignedDVecN { scale * sum_squares.sqrt() } - /// Returns the Euclidean norm through the scaled two-pass sum of squares of - /// [`DVecN::stable_l2`], over the lane view. + /// Computes the scaled Euclidean norm as a nonnegative finite value. + /// + /// Uses the evaluation described by [`DVecN::stable_l2`]. You must establish that the + /// components and the computed norm are finite. Use [`Self::checked_stable_l2`] when those + /// conditions need validation. #[inline] #[must_use] pub(crate) fn stable_l2(&self) -> DNonNegative { DNonNegative::new_unchecked(self.stable_l2_impl()) } - /// Returns the Euclidean norm through the scaled two-pass sum of squares of - /// [`DVecN::stable_l2`], over the lane view, when the result is finite. + /// Computes the scaled Euclidean norm when it is finite. /// - /// Subnormal-only vectors keep their norm and magnitudes near [`f64::MAX`] stay finite where - /// naive squared accumulation would not. The norm of the empty and the all-zero vector is - /// `0.0`. Returns [`None`] when a component or the result is not finite. A norm is - /// non-negative, so the reading carries that domain. + /// Uses the evaluation and numerical limits described by [`DVecN::stable_l2`]. Returns [`None`] + /// when any component or the computed norm is non-finite. Empty and all-zero vectors return + /// zero. #[inline] pub(crate) fn checked_stable_l2(&self) -> Option { DNonNegative::new(self.stable_l2_impl()) @@ -898,10 +915,10 @@ impl AlignedDVecN { /// Adds `factor` times an aligned working-precision vector, component-wise. /// - /// Each `f32` component of `direction` widens to `f64` exactly, as [`DVecN::add_scaled`]. - /// Both operands load as aligned lane groups, and the group boundaries coincide (eight - /// components per group on either side), so the fold shape matches the unaligned kernel - /// bit for bit. + /// Each `f32` component of `direction` widens to `f64` exactly, as in [`DVecN::add_scaled`]. + /// The aligned loads retain that method's eight-component groups and scalar remainder, with the + /// same fused expressions on corresponding components. The fold shape matches the unaligned + /// kernel bit for bit. #[inline] pub fn add_scaled(&mut self, direction: &AlignedVecN, factor: f64) { let scale = f64x8::splat(factor); @@ -1010,9 +1027,9 @@ where /// An owned `N`-dimensional vector in a heap allocation aligned for [`f64x8`]. /// -/// The buffer is allocated with `align_of::()` alignment regardless of `N`, so dereferencing -/// always yields an [`AlignedDVecN`]. This is the storage for double-precision optimizer state - -/// parameter and gradient vectors whose dimension is far too large for the stack. +/// The buffer has `align_of::()` alignment regardless of `N`. Use [`Self::zero`] to +/// initialize large vectors directly on the heap and [`AlignedDVecN`] methods to update them in +/// place. pub(crate) struct BoxedDVecN { ptr: NonNull, alloc: A, @@ -1020,6 +1037,10 @@ pub(crate) struct BoxedDVecN { impl BoxedDVecN { /// Copies the vector into a new aligned allocation in the global allocator. + /// + /// # Panics + /// + /// Panics if `N` components cannot be represented by the aligned allocation layout. #[inline] #[must_use] pub(crate) fn new(value: &DVecN) -> Self { @@ -1029,7 +1050,11 @@ impl BoxedDVecN { /// Creates the zero vector in a new aligned allocation in the global allocator. /// /// Every component is `0.0` and the buffer is valid for in-place filling through - /// [`as_array_mut`](AlignedDVecN::as_array_mut). + /// [`AlignedDVecN::as_array_mut`]. + /// + /// # Panics + /// + /// Panics if `N` components cannot be represented by the aligned allocation layout. #[inline] #[must_use] pub(crate) fn zero() -> Self { @@ -1038,9 +1063,14 @@ impl BoxedDVecN { } impl BoxedDVecN { - /// The allocation layout: `N` components, padded to the alignment of [`f64x8`]. + /// Computes the layout of `N` components with [`f64x8`] alignment. + /// + /// Allocation and deallocation must use the same layout, whose byte size does not round up when + /// its alignment increases. /// - /// Allocation and deallocation must agree on this. + /// # Panics + /// + /// Panics if the aligned size exceeds the allocation layout's `isize::MAX` limit. #[inline] fn layout() -> Layout { Layout::array::(N) @@ -1050,8 +1080,11 @@ impl BoxedDVecN { /// Creates the zero vector in a new aligned allocation in `alloc`. /// - /// [`handle_alloc_error`](std::alloc::handle_alloc_error) aborts the process when the - /// allocator cannot provide the buffer. + /// Invokes [`alloc::alloc::handle_alloc_error`] when the allocator cannot provide the buffer. + /// + /// # Panics + /// + /// Panics if `N` components cannot be represented by the aligned allocation layout. #[inline] #[must_use] pub(crate) fn zero_in(alloc: A) -> Self { @@ -1069,8 +1102,12 @@ impl BoxedDVecN { /// Copies the vector into a new aligned allocation in `alloc`. /// - /// [`handle_alloc_error`](std::alloc::handle_alloc_error) aborts the process when the - /// allocator cannot provide the buffer. + /// Invokes [`alloc::alloc::handle_alloc_error`] when the allocator cannot provide the buffer. + /// Use [`Self::try_new_in`] to handle allocation failure. + /// + /// # Panics + /// + /// Panics if `N` components cannot be represented by the aligned allocation layout. #[inline] #[must_use] pub(crate) fn new_in(value: &DVecN, alloc: A) -> Self { @@ -1085,16 +1122,23 @@ impl BoxedDVecN { /// /// # Errors /// - /// Returns [`AllocError`] when the allocator cannot provide the buffer. The error path leaks - /// no memory. + /// Returns [`AllocError`] when the allocator cannot provide the buffer. + /// + /// # Panics + /// + /// Panics if `N` components cannot be represented by the aligned allocation layout. #[inline] pub(crate) fn try_new_in(value: &DVecN, alloc: A) -> Result { let layout = Self::layout(); let allocation = alloc.allocate(layout)?; let ptr = allocation.cast::(); - // SAFETY: `allocate` returned a fresh buffer of at least `N` components, so it cannot - // overlap the borrowed source. + // SAFETY: copy_nonoverlapping is an untyped copy that preserves initialization state. Its + // aligned pointers must be valid for the N-component read and write ranges, without + // overlap. The source array reference supplies N initialized f64 components. allocate + // returns fresh storage for the checked N-component layout, with f64x8 alignment even when + // N is zero. Distinct nonempty live allocations cannot overlap, and an empty copy accesses + // no bytes. Therefore copying N components initializes all components of the owned buffer. unsafe { ptr::copy_nonoverlapping(value.as_array().as_ptr(), ptr.as_ptr(), N); } @@ -1107,17 +1151,23 @@ const impl Deref for BoxedDVecN { type Target = AlignedDVecN; fn deref(&self) -> &Self::Target { - // SAFETY: `ptr` owns an initialized buffer of `N` components for as long as `self` lives, - // allocated with the alignment of `f64x8` by `layout`. + // SAFETY: The array reference requires initialized components and sufficient alignment, and + // from_ref_unchecked additionally requires f64x8 alignment. Every constructor initializes N + // components in the checked layout and retains its allocating instance. The allocator + // provides a non-null aligned pointer even for N = 0. This shared borrow prevents + // destruction or mutation of the buffer. Therefore the aligned view is valid for the + // borrow's lifetime. unsafe { AlignedDVecN::from_ref_unchecked(&*self.ptr.as_ptr().cast::<[f64; N]>()) } } } const impl DerefMut for BoxedDVecN { fn deref_mut(&mut self) -> &mut Self::Target { - // SAFETY: `ptr` owns an initialized buffer of `N` components for as long as `self` lives, - // allocated with the alignment of `f64x8` by `layout`; the exclusive borrow of `self` - // guards the exclusive reference. + // SAFETY: The mutable array reference requires initialized, aligned storage and exclusive + // access. Constructors initialize N components with f64x8 alignment, including an aligned + // non-null pointer for N = 0. This exclusive Self borrow excludes all other access and + // bounds the result's lifetime. Therefore both the array reference and from_mut_unchecked + // satisfy their contracts. unsafe { AlignedDVecN::from_mut_unchecked(&mut *self.ptr.as_ptr().cast::<[f64; N]>()) } } } @@ -1129,11 +1179,14 @@ impl Clone for BoxedDVecN { } fn clone_from(&mut self, source: &Self) { - // Both buffers share the same layout for a given `N`, so `clone_from` reuses the existing - // allocation instead of reallocating. - // - // SAFETY: both pointers own initialized buffers of `N` components, and two live boxes - // cannot alias. + // fixed N permits reuse of the existing destination allocation. + // SAFETY: copy_nonoverlapping is an untyped copy that preserves initialization state. It + // requires aligned pointers valid for disjoint N-component read and write ranges. Each + // owner has an N-component allocation with the same layout, and the source array reference + // supplies initialized f64 values. The exclusive destination borrow prevents aliasing the + // source for a nonempty copy. Both pointers remain aligned and non-null for an empty copy, + // which accesses no bytes. Therefore the copy preserves valid initialized destination + // components. unsafe { ptr::copy_nonoverlapping(source.as_array().as_ptr(), self.ptr.as_ptr(), N); } @@ -1185,18 +1238,23 @@ const impl PartialEq for BoxedDVecN { impl Drop for BoxedDVecN { #[inline] fn drop(&mut self) { - // SAFETY: every constructor allocates `ptr` from `alloc` with `Self::layout()`, the - // layout passed here, and nothing has deallocated it since. + // SAFETY: Deallocation requires the original allocator, a live pointer and a matching + // layout. Every constructor stores the allocating instance and pointer for Self::layout(), + // and no method transfers or frees that ownership. Therefore Drop may deallocate the buffer + // exactly once with this layout. unsafe { self.alloc.deallocate(self.ptr.cast::(), Self::layout()); } } } -// SAFETY: the buffer is exclusively owned and its `f64` components are `Send` and `Sync`; the -// allocator's own thread-safety carries the bound. +// SAFETY: Send permits transferring ownership between threads. The vector exclusively owns its f64 +// buffer, and A: Send permits moving the allocating instance with it. Borrowed views prevent moving +// the owner while in use. Therefore the initialized buffer and its deallocation capability may be +// transferred together. unsafe impl Send for BoxedDVecN {} -// SAFETY: shared access only exposes `&[f64; N]`, which is `Sync`; the allocator's own -// thread-safety carries the bound. +// SAFETY: Sync requires shared access to avoid data races. Shared vector methods expose immutable +// f64 components without interior mutation, and A: Sync covers sharing the allocator. Mutation and +// destruction require exclusive access. Therefore shared vector references are safe across threads. unsafe impl Sync for BoxedDVecN {} diff --git a/libs/@local/graph/atlas/src/math/dvecn/tests.rs b/libs/@local/graph/atlas/src/math/dvecn/tests.rs index 6d4758ea089..4fae81ab9c4 100644 --- a/libs/@local/graph/atlas/src/math/dvecn/tests.rs +++ b/libs/@local/graph/atlas/src/math/dvecn/tests.rs @@ -13,7 +13,11 @@ use crate::math::{BoxedDVecN, DVecN}; #[test] fn max_and_sum_match_scalar_folds_across_chunk_sizes() { - // 0, remainder-only, exact-chunk, and chunk-plus-remainder lengths. + /// Compares the maximum and sum with scalar folds. + /// + /// # Panics + /// + /// Panics if the maxima differ or the sum's absolute difference is not below `1e-12`. fn check(components: [f64; N]) { let vec = DVecN::new(components); @@ -27,6 +31,8 @@ fn max_and_sum_match_scalar_folds_across_chunk_sizes() { ); } + // these lengths cover the empty, remainder-only, exact four-lane chunk, and + // chunk-plus-remainder paths used by `max` and `sum`. check([]); check([-3.5]); check([0.5, -1.25, 2.0]); @@ -168,9 +174,7 @@ fn add_widened_matches_scalar_reference() { #[test] fn div_assign_divides_every_component() { - // N = 11 crosses one full 8-lane group plus a remainder, so both the batched body and the - // remainder divide; the operation is a plain IEEE division either way, so the results are - // bit-equal. + // N = 11 exercises one eight-lane SIMD group and a three-component scalar remainder let components = core::array::from_fn::(|index| { f64::from(u8::try_from(index).expect("test sizes are small")).mul_add(0.75, -4.0) }); @@ -204,7 +208,7 @@ fn add_squared_deviation_matches_scalar_reference() { assert_eq!(accumulator.as_array(), &expected); } -/// Deterministic, sign-varying components crossing multiple 8-lane chunks. +/// Generates a repeating sign-varying sequence plus `offset`. #[expect(clippy::integer_division_remainder_used)] fn scattered(offset: f64) -> [f64; N] { core::array::from_fn(|index| { @@ -216,6 +220,11 @@ fn scattered(offset: f64) -> [f64; N] { #[test] fn dot_matches_a_plain_reference_across_chunk_sizes() { + /// Compares a dot product with the scalar product sum. + /// + /// # Panics + /// + /// Panics if the absolute difference exceeds `expected.abs().mul_add(1e-12, 1e-12)` or is NaN. fn check() { let left: [f64; N] = scattered(0.5); let right: [f64; N] = scattered(-1.25); @@ -260,15 +269,11 @@ fn norm_squared_and_abs_sum_match_plain_references() { // (two full chunks plus a remainder of three) and `add_scaled`'s 8-lane // boundary (one chunk plus three). -/// Logits bounded to `-50..50`, where `exp` is well-conditioned. -/// -/// The example-based tests above pin the shifted form's stability under logits large enough to -/// overflow a naive `exp`. +/// Generates eleven finite logits in `-50..50`. fn logits_strategy() -> impl Strategy { proptest::array::uniform11(-50.0_f64..50.0) } -/// Softmax outputs are probabilities: each lies in `[0, 1]` and they sum to one up to rounding. #[property_test] fn softmax_outputs_form_a_distribution(#[strategy = logits_strategy()] logits: [f64; 11]) { let probabilities = DVecN::new(logits).softmax(); @@ -338,18 +343,31 @@ fn add_scaled_matches_a_scalar_reference_loop( prop_assert_eq!(actual.as_array(), &expected); } -/// A small test index as an exact double, per the house cast discipline. +/// Converts a test index to an exactly representable double. +/// +/// # Panics +/// +/// Panics if `index` exceeds `u8::MAX`. fn coordinate(index: usize) -> f64 { f64::from(u8::try_from(index).expect("test sizes are small")) } -/// An aligned copy of the same components, for cross-type bit agreement. +/// Allocates an aligned copy of the components. +/// +/// # Panics +/// +/// Panics if [`BoxedDVecN::new`] cannot represent the aligned layout. fn aligned(components: &[f64; N]) -> BoxedDVecN { BoxedDVecN::new(DVecN::from_ref(components)) } #[test] fn max_abs_matches_the_scalar_fold_across_chunk_sizes() { + /// Compares each vector type's maximum magnitude with a scalar fold. + /// + /// # Panics + /// + /// Panics if a result differs from the reference or [`aligned`] rejects the layout. fn check(components: [f64; N]) { let expected = components .iter() @@ -381,6 +399,11 @@ fn max_abs_ignores_nan_in_favor_of_finite_magnitudes() { #[test] fn stable_l2_matches_exact_norms_on_both_types() { + /// Asserts the finite scaled norm equals `expected` on both vector types. + /// + /// # Panics + /// + /// Panics if either result differs from `expected`. fn check(components: [f64; N], expected: f64) { assert_eq!(DVecN::new(components).stable_l2(), expected); assert_eq!(aligned(&components).stable_l2(), expected); @@ -394,6 +417,7 @@ fn stable_l2_matches_exact_norms_on_both_types() { check([4.0, 0.0, -3.0], 5.0); } +/// Inserting zeros into a scalar-tail norm preserves the nonzero accumulation order. #[test] fn stable_l2_zero_components_contribute_exactly_nothing() { let dense = [0.3, -1.7, 2.9]; @@ -427,13 +451,17 @@ fn stable_l2_survives_huge_components() { fn stable_l2_propagates_non_finite_components() { assert!(DVecN::new([1.0, f64::NAN]).stable_l2().is_nan()); assert!(!DVecN::new([f64::INFINITY, 1.0]).stable_l2().is_finite()); - // A NaN alongside only zeros hides from the maxNum scale, so the zero-scale finiteness check - // catches it. + // a NaN alongside only zeros gives a zero maxNum scale despite the non-finite component assert!(DVecN::new([0.0, f64::NAN, 0.0]).stable_l2().is_nan()); } #[test] fn stable_l2_agrees_between_types_across_chunk_sizes() { + /// Asserts equal finite scaled norms from both vector types. + /// + /// # Panics + /// + /// Panics if the results differ. fn check(components: [f64; N]) { assert_eq!( DVecN::new(components).stable_l2(), @@ -452,6 +480,11 @@ fn stable_l2_agrees_between_types_across_chunk_sizes() { #[test] fn is_finite_detects_non_finite_components_in_lanes_and_remainder() { + /// Checks both vector types against the expected finiteness result. + /// + /// # Panics + /// + /// Panics if either result differs from `expected` or [`aligned`] rejects the layout. fn check(components: [f64; N], expected: bool) { assert_eq!(DVecN::new(components).is_finite(), expected, "over {N}"); assert_eq!( @@ -475,6 +508,11 @@ fn is_finite_detects_non_finite_components_in_lanes_and_remainder() { #[test] fn mul_add_matches_the_componentwise_fused_multiply_add() { + /// Compares fused updates with scalar multiply-adds bit for bit. + /// + /// # Panics + /// + /// Panics if either result differs from the scalar reference or [`aligned`] rejects the layout. fn check(base: [f64; N], direction: [f64; N], factor: f64) { let mut updated = DVecN::new(base); updated.mul_add(DVecN::from_ref(&direction), factor); @@ -588,7 +626,7 @@ fn scalar_multiply_and_divide_assign_scale_every_component() { let mut scaled = aligned(&components); *scaled *= 2.0; for (&result, &input) in scaled.as_array().iter().zip(&components) { - // Doubling is exact in binary floating point. + // these small binary fractions double without overflow or rounding assert_eq!(result, input * 2.0); } @@ -600,6 +638,13 @@ fn scalar_multiply_and_divide_assign_scale_every_component() { #[test] fn aligned_reductions_agree_with_unaligned_bits_across_chunk_sizes() { + /// Compares the unaligned and aligned vector reductions. + /// + /// The dot and squared norm use numeric equality. The absolute sum compares result bits. + /// + /// # Panics + /// + /// Panics if a comparison fails or [`aligned`] rejects the layout. fn check(components: [f64; N], other: [f64; N]) { let unaligned = DVecN::new(components); let unaligned_other = DVecN::new(other); @@ -637,9 +682,9 @@ fn aligned_reductions_agree_with_unaligned_bits_across_chunk_sizes() { /// The interleaved reductions visit a third lane group and the remainder in one call. /// -/// Twenty-five components split as three 8-lane groups plus one remainder component, so the -/// two-accumulator fold revisits accumulator zero at group index two. Signed integer components -/// keep every partial sum exact. +/// Twenty-five components exercise three 8-lane groups and one remainder component. The third group +/// reuses accumulator zero in the two-accumulator fold. Signed integer components keep every +/// partial sum exact. #[test] fn abs_sum_interleaves_three_lane_groups() { let components = core::array::from_fn::(|index| { @@ -669,7 +714,6 @@ fn stable_l2_interleaves_three_lane_groups() { assert_eq!(DVecN::new(components).stable_l2(), 5.0); } -/// Negation flips the aligned lane groups of guaranteed-aligned storage. #[test] fn negate_flips_the_aligned_lane_groups() { let mut boxed = BoxedDVecN::from([1.0, -2.0, 3.0, -4.0, 5.0, -6.0, 7.0, -8.0]); @@ -682,18 +726,17 @@ fn negate_flips_the_aligned_lane_groups() { ); } -/// Hashes one value with the std default hasher. +/// Hashes a value with [`DefaultHasher`]. +/// +/// # Panics +/// +/// Propagates panics from the value's [`Hash`] implementation. fn hash_of(value: impl Hash) -> u64 { let mut hasher = DefaultHasher::new(); value.hash(&mut hasher); hasher.finish() } -/// The tests the `miri` nextest profile selects. -/// -/// Each test here wraps or boxes double-precision component storage and checks the alignment those -/// views require. The profile selects by module path, so moving a test in or out of this module is -/// the whole edit. mod miri { use super::{hash_of, scattered}; use crate::math::{AlignedDVecN, BoxedDVecN, DVecN, test_alloc::CountingAllocator}; @@ -771,7 +814,6 @@ mod miri { assert_eq!(zero.norm_squared().into_raw(), 0.0); } - /// The checking wrapper admits aligned storage and refuses an offset view of it. #[test] fn aligned_from_mut_checks_alignment() { let mut boxed = BoxedDVecN::from([7.0_f64; 9]); @@ -788,8 +830,6 @@ mod miri { assert!(AlignedDVecN::from_mut(tail).is_none()); } - /// The checking wrapper admits aligned storage and refuses an offset view of it, through a - /// shared reference. #[test] fn aligned_from_ref_checks_alignment() { let boxed = BoxedDVecN::from([7.0_f64; 9]); @@ -804,7 +844,6 @@ mod miri { assert!(AlignedDVecN::from_ref(tail).is_none()); } - /// `clone_from` reuses the target's existing allocation instead of reallocating. #[test] fn boxed_dvecn_clone_from_reuses_the_allocation() { let source = BoxedDVecN::from([9.0_f64; 8]); @@ -821,7 +860,6 @@ mod miri { ); } - /// `Hash` follows the components and `Debug` prints them. #[test] fn boxed_hash_and_debug_follow_the_components() { let low = BoxedDVecN::from([0.5, 1.5]); @@ -832,7 +870,6 @@ mod miri { assert_eq!(format!("{low:?}"), "AlignedDVecN([0.5, 1.5])"); } - /// Dropping a box returns its buffer to the allocator that provided it. #[test] fn boxed_drop_returns_the_buffer_to_its_allocator() { let alloc = CountingAllocator::new(); diff --git a/libs/@local/graph/atlas/src/math/error.rs b/libs/@local/graph/atlas/src/math/error.rs index cc5052f3dee..3fc3d6e4064 100644 --- a/libs/@local/graph/atlas/src/math/error.rs +++ b/libs/@local/graph/atlas/src/math/error.rs @@ -1,19 +1,11 @@ -//! The refusal a finiteness scan hands back, shared by every consumer that proves a point set. -//! -//! [`FinitePointField::new`] owns the scan and refuses the first offender; every boundary that -//! propagates a refusal names the same error type, so refusals stay interchangeable wherever a -//! point set is proven. -//! -//! [`FinitePointField::new`]: super::FinitePointField::new - use core::{error::Error, fmt}; use hashql_core::id::Id; -/// A refused point, a NaN or infinite component at the named id. +/// A point containing a NaN or infinite coordinate, identified by its row ID. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct NonFinitePoint { - /// The first id whose point is non-finite. + /// The first ID whose point is non-finite. pub id: I, } diff --git a/libs/@local/graph/atlas/src/math/field.rs b/libs/@local/graph/atlas/src/math/field.rs index 662b18be10b..378025e630a 100644 --- a/libs/@local/graph/atlas/src/math/field.rs +++ b/libs/@local/graph/atlas/src/math/field.rs @@ -1,10 +1,15 @@ -//! A point slice proven finite at construction, and the statistics defined over it. +//! Finite point fields and their geometric statistics. //! -//! A consumer that needs a finite field takes the field instead of scanning the slice itself, -//! so the finiteness proof lives in one constructor and the consuming arithmetic restates -//! nothing. The statistics accumulate in double precision over fixed chunk boundaries with -//! ordered folds, so every reading is bit-deterministic under any thread schedule and a caller -//! may persist it and replay it exactly. +//! [`FinitePointField`] retains a finiteness check across borrowed and owned point storage. For +//! points pᵢ ∈ ℝ² and count n > 0, the centroid is μ = Σpᵢ/n, the squared-deviation sum about c ∈ +//! ℝ² is S(c) = Σ‖pᵢ − c‖², and RMS spread is √(S(μ)/n). Extent is the greatest absolute +//! coordinate, maxᵢ max(|pᵢₓ|, |pᵢᵧ|). +//! +//! Coordinates widen exactly from finite `f32` to `f64`. Sums, squared distances and normalization +//! still round, including conversion of counts above 2⁵³. Fixed point chunks and a fixed +//! combination tree keep the grouping independent of Rayon scheduling. This preserves the reduction +//! order within a build, without specifying bitwise agreement across builds or SIMD +//! implementations. use alloc::alloc::Allocator; use core::{ @@ -25,12 +30,16 @@ use super::{ /// Points per parallel chunk in the point-statistics reductions. pub(super) const POINT_CHUNK: NonZero = NonZero::new(4096).unwrap(); -/// Folds per-chunk partials over a midpoint-split tree, in a fixed combination order. +/// Folds point chunks over a midpoint-split tree in a fixed combination order. +/// +/// Leaves have at most [`POINT_CHUNK`] points, including an empty leaf for empty input. Every split +/// is at a chunk boundary and divides the chunk count at its midpoint. [`rayon::join`] preserves +/// the left and right result positions regardless of execution order. Therefore deterministic +/// `leaf` and `combine` callbacks give a schedule-independent result. /// -/// The leaves are [`POINT_CHUNK`]-sized chunks and every split occurs at a chunk boundary at -/// the chunk count's midpoint, so the combination tree depends only on the point count and -/// the fold is bit-deterministic under any thread schedule. [`rayon::join`] parallelizes the -/// halves while the combine positions stay fixed by the tree. +/// # Panics +/// +/// Propagates a panic from either callback. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -74,7 +83,8 @@ fn chunk_coordinate_sum(points: &[Vec2]) -> DVec2 { /// Accumulates one chunk's squared distances to the centre in double precision. /// -/// The batches ride SIMD lanes and a scalar tail closes the chunk. +/// Complete four-point batches accumulate per-lane squared deviations before reduction. Remaining +/// points add their separately rounded scalar distances afterward. fn chunk_squared_deviations(points: &[Vec2], centre: DVec2) -> f64 { let (batches, rest) = points.iter_transposed_wide(); @@ -93,13 +103,11 @@ fn chunk_squared_deviations(points: &[Vec2], centre: DVec2) -> f64 { sum } -/// A view of a point slice whose every coordinate is finite, proven at construction. +/// A typed point slice whose coordinates are finite. /// -/// The constructor owns the finiteness scan, four points at a time on SIMD lanes, and a -/// consumer holding a field divides, squares, and folds without re-checking. Reads flow -/// through the slice's own API, and every write path carries the `_unchecked` suffix - -/// [`as_raw_mut_unchecked`](Self::as_raw_mut_unchecked) and its siblings - where the caller -/// keeps the proof. +/// [`new`](Self::new) validates the initial points and [`copy_from`](Self::copy_from) validates +/// replacements before writing them. Writes through the `_unchecked` methods must preserve +/// finiteness. Indexing selects a row by its ID and panics outside the slice's bounds. #[derive(Debug, PartialEq, zerocopy::IntoBytes, zerocopy::Immutable, zerocopy::KnownLayout)] #[repr(transparent)] pub(crate) struct FinitePointField(IdSlice); @@ -120,8 +128,20 @@ where /// /// # Errors /// - /// Returns the smallest index whose point has a NaN or infinite component. + /// Returns [`NonFinitePoint`] with the smallest ID whose point has a NaN or infinite component. + /// + /// # Panics + /// + /// If a non-finite point is found, panics when an index needed by [`IdSlice::iter_enumerated`] + /// is outside `I`'s range. Before scanning for the offender, that enumeration checks the + /// slice's last index because raw typed-slice construction does not establish ID + /// representability. pub(crate) fn new(points: &IdSlice) -> Result<&Self, NonFinitePoint> { + // On the measured arm64 Apple-silicon host, the math_kernels finite_scan benchmark's serial + // four-point scan beat Rayon's per-point search at sampled counts from 2¹² through 2²⁰ + // points, by over 100× at 2¹⁴ and over 4× at 2²⁰. Distributing the same batch predicate + // over Rayon chunks was near parity at 2¹² and slower at sampled counts from 2¹⁴ through + // 2²⁰. These wall-time measurements select the serial scan. let (prefix, aligned, suffix) = Vec2x4::from_slice(points.as_raw()); if !prefix.iter().all(|point| point.is_finite()) || !suffix.iter().all(|point| point.is_finite()) @@ -137,19 +157,25 @@ where return Err(NonFinitePoint { id }); } - // SAFETY: `Self` is `repr(transparent)` over `IdSlice`, so the reference - // reinterprets in place at the same layout, and the borrow keeps the input's lifetime. + // SAFETY: repr(transparent) preserves the IdSlice layout and metadata. The source is + // initialized and shared for the returned lifetime, and the scan established the field's + // finiteness invariant. Therefore the cast preserves reference validity and the field + // contract. let this = unsafe { &*((&raw const *points) as *const Self) }; Ok(this) } - /// Validates every point finite and wraps the owned slice, without a copy. + /// Validates every point as finite and retains the owned slice without copying. /// - /// The boxed form of [`new`](Self::new). + /// The boxed form of [`new`](Self::new). An error drops the supplied allocation. /// /// # Errors /// - /// Returns the smallest index whose point has a NaN or infinite component. + /// Returns [`NonFinitePoint`] with the smallest ID whose point has a NaN or infinite component. + /// + /// # Panics + /// + /// Panics under [`Self::new`]'s ID-range condition. pub(crate) fn new_boxed( points: Box, A>, ) -> Result, NonFinitePoint> { @@ -157,8 +183,11 @@ where let (ptr, alloc) = Box::into_raw_with_allocator(points); - // SAFETY: `Self` is `repr(transparent)` over `IdSlice`, so the box pointer - // reinterprets in place at the same layout, in the same allocator. + // SAFETY: Box::from_raw_in requires unique ownership of a valid allocation with the target + // layout. into_raw_with_allocator transfers that ownership and the allocator, and + // repr(transparent) preserves the initialized slice's layout and metadata. The scan + // established finiteness. Therefore the reconstructed box retains the same valid allocation + // and field invariant. let this = unsafe { Box::from_raw_in(ptr as *mut Self, alloc) }; Ok(this) } @@ -166,8 +195,7 @@ where /// Wraps a slice the caller proves finite. /// /// Where the proof is not immediate, [`new`](Self::new) scans instead. - // Correctness, never memory safety: a broken promise yields wrong statistics downstream, - // so the checked-domain claim stays a debug assertion rather than an `unsafe` contract. + // finiteness concerns correctness alone and imposes no memory-safety requirement. #[inline] #[must_use] pub(crate) fn new_unchecked(points: &IdSlice) -> &Self { @@ -176,15 +204,17 @@ where "the caller promised a finite point set", ); - // SAFETY: `Self` is `repr(transparent)` over `IdSlice`, so the reference - // reinterprets in place at the same layout, and the borrow keeps the input's lifetime. + // SAFETY: repr(transparent) preserves the initialized IdSlice's layout and metadata. The + // pointer keeps its provenance and shared borrow lifetime. Finiteness is a separate + // correctness obligation on the caller. Therefore the cast preserves Rust reference + // validity. unsafe { &*((&raw const *points) as *const Self) } } /// Wraps a mutable slice the caller proves finite, and keeps finite. /// /// The mutable form of [`new_unchecked`](Self::new_unchecked): every write through - /// [`as_raw_mut_unchecked`](Self::as_raw_mut_unchecked) must land a finite value. + /// [`as_raw_mut_unchecked`](Self::as_raw_mut_unchecked) must preserve finite coordinates. // Correctness, never memory safety: a broken promise yields wrong statistics downstream. #[inline] #[must_use] @@ -194,16 +224,16 @@ where "the caller promised a finite point set", ); - // SAFETY: `Self` is `repr(transparent)` over `IdSlice`, so the reference - // reinterprets in place at the same layout, and the borrow keeps the input's lifetime - // and exclusivity. + // SAFETY: repr(transparent) preserves the initialized IdSlice's layout and metadata. The + // pointer keeps its provenance and exclusive borrow lifetime. Finiteness is a separate + // correctness obligation on the caller. Therefore the cast preserves Rust reference + // validity. unsafe { &mut *((&raw mut *points) as *mut Self) } } /// Wraps an owned slice the caller proves finite, without a copy. /// - /// The boxed form of [`new_unchecked`](Self::new_unchecked), for an owner that stores the - /// proof beside the points. + /// The boxed form of [`new_unchecked`](Self::new_unchecked). // Correctness, never memory safety: a broken promise yields wrong statistics downstream. #[must_use] pub(crate) fn new_boxed_unchecked( @@ -216,8 +246,11 @@ where let (ptr, alloc) = Box::into_raw_with_allocator(points); - // SAFETY: `Self` is `repr(transparent)` over `IdSlice`, so the box pointer - // reinterprets in place at the same layout, in the same allocator. + // SAFETY: Box::from_raw_in requires unique ownership of a valid allocation with the target + // layout. into_raw_with_allocator transfers that ownership and the allocator, and + // repr(transparent) preserves the initialized slice's layout and metadata. Finiteness + // remains the caller's correctness obligation. Therefore reconstructing the box preserves + // allocation and value validity. unsafe { Box::from_raw_in(ptr as *mut Self, alloc) } } @@ -228,10 +261,9 @@ where &self.0 } - /// Returns the underlying point slice mutably; every write must land a finite value. + /// Borrows the typed points mutably, with finiteness maintained by the caller. /// - /// The mutable form of [`as_slice`](Self::as_slice): the caller keeps the proof, exactly - /// as through [`as_raw_mut_unchecked`](Self::as_raw_mut_unchecked). + /// Every coordinate must be finite when the borrow ends, as for [`Self::as_raw_mut_unchecked`]. #[inline] #[must_use] pub(crate) const fn as_slice_mut_unchecked(&mut self) -> &mut IdSlice { @@ -240,9 +272,8 @@ where /// Gathers the named rows into an owned field over the gather's own row domain. /// - /// Each entry of `rows` names a row of this field, and the returned field reads the - /// gathered points in `rows` order. A gather from a proven-finite field stays finite, so - /// the proof carries over with no scan. + /// Each entry of `rows` names a row of this field. The gather copies the proven-finite points + /// in `rows` order without arithmetic. The returned field is finite without another scan. /// /// # Panics /// @@ -254,10 +285,9 @@ where FinitePointField::new_boxed_unchecked(gathered.into_boxed_slice()) } - /// Returns the raw mutable rows, and the caller keeps every write finite. + /// Borrows the raw points mutably, with finiteness maintained by the caller. /// - /// The write path for a kernel whose own vocabulary is raw rows. The caller holds the - /// finiteness proof, and every value written must be finite when the borrow ends. + /// Every coordinate must be finite when the borrow ends. // Correctness, never memory safety: a non-finite write yields wrong statistics downstream. #[inline] #[must_use] @@ -267,8 +297,8 @@ where /// Returns the largest absolute coordinate component over the whole field. /// - /// Maximum folds are order-independent over a finite set, so the reading is - /// bit-deterministic under any thread schedule. + /// The empty field gives zero. Absolute values make all zeros positive, and maximum over the + /// finite non-negative components is order-independent. #[must_use] pub(crate) fn extent(&self) -> NonNegative { let largest = self @@ -291,16 +321,16 @@ where }) .reduce(|| 0.0_f32, f32::max); - // In domain with no check: a maximum of absolute components of finite points is finite - // and at least zero, and the empty fold's identity is zero. + // Absolute finite f32 coordinates remain finite and non-negative. Each fold selects a + // component or its zero identity. Therefore the maximum satisfies NonNegative's domain. NonNegative::new_unchecked(largest) } /// Returns the centroid in double precision. /// - /// Chunks of [`POINT_CHUNK`] points accumulate four points at a time on SIMD lanes, and - /// the partials combine through [`tree_fold`]'s fixed-shape tree, so the reading is - /// bit-deterministic under any thread schedule and allocates nothing. + /// Approximates μ = Σpᵢ/n using double-precision accumulation and normalization. Chunk + /// boundaries and the combination tree are fixed by the point order and count, independently of + /// Rayon scheduling. /// /// # Panics /// @@ -321,10 +351,11 @@ where total / count } - /// Returns the sum of squared distances from the points to `centre`, in double precision. + /// Accumulates squared distances from `centre` in double precision. /// - /// The reduction is chunked and shaped exactly like [`centroid`](Self::centroid)'s, so it - /// is bit-deterministic under any thread schedule. + /// Approximates S(c) = Σ‖pᵢ − c‖² with the schedule-independent grouping of [`Self::centroid`]. + /// An empty field gives zero. The supplied centre is unrestricted, and nonempty calculations + /// can produce non-finite results for a non-finite or sufficiently large centre. #[must_use] pub(crate) fn squared_deviation_sum(&self, centre: DVec2) -> f64 { tree_fold( @@ -336,8 +367,8 @@ where /// Returns the RMS spread of the points about their centroid, in double precision. /// - /// The centroid pass runs first and the mean-squared-distance pass second, both through - /// the deterministic chunked reductions above. + /// Approximates √(S(μ)/n), using the computed centroid μ followed by a squared-deviation pass. + /// Both passes retain the schedule-independent grouping of [`Self::centroid`]. /// /// # Panics /// @@ -356,7 +387,11 @@ where /// Views the rows below `bound` as a field. /// - /// A prefix of a proven-finite field stays finite, so the proof carries over with no scan. + /// Taking a prefix preserves the finiteness invariant. + /// + /// # Panics + /// + /// Panics if `bound` exceeds the field length. #[inline] #[must_use] pub(crate) fn prefix(&self, bound: I) -> &Self { @@ -365,8 +400,11 @@ where /// Views the rows below `bound` as a mutable field. /// - /// The mutable form of [`prefix`](Self::prefix): writes through the view carry the same - /// keep-it-finite contract as [`as_raw_mut_unchecked`](Self::as_raw_mut_unchecked). + /// Writes through the view must preserve finiteness, as for [`Self::as_raw_mut_unchecked`]. + /// + /// # Panics + /// + /// Panics if `bound` exceeds the field length. #[inline] pub(crate) fn prefix_mut(&mut self, bound: I) -> &mut Self { Self::new_unchecked_mut(self.0.prefix_mut(bound)) @@ -405,17 +443,19 @@ mod tests { hashql_core::id::newtype! { /// The test fields' row domain. + /// #[id(const)] struct RowId(u32) } hashql_core::id::newtype! { /// The gather tests' target domain. + /// #[id(const)] struct DrawId(u32) } - /// Enough points to cover the prefix, batch, and suffix regions of the SIMD split. + /// Generates eleven finite points for slice-alignment tests. fn points() -> Vec { (0..11_u8) .map(|index| Vec2::new(f32::from(index), -f32::from(index))) @@ -482,6 +522,8 @@ mod tests { let _: Box> = field.gather(IdSlice::::from_raw(&rows)); } + // The dyadic rectangle has centroid (2, −1) and four squared deviations of 5. The sum is + // exactly 20, and its RMS spread is the floating-point square root of 5. #[test] fn the_statistics_read_exact_dyadic_values() { // Centroid (2, -1), deviations (∓2, ±1): the sums are exact dyadics. diff --git a/libs/@local/graph/atlas/src/math/kdtree/mod.rs b/libs/@local/graph/atlas/src/math/kdtree/mod.rs index ab3658bbef4..9424fede2bb 100644 --- a/libs/@local/graph/atlas/src/math/kdtree/mod.rs +++ b/libs/@local/graph/atlas/src/math/kdtree/mod.rs @@ -1,63 +1,62 @@ -//! Exact k-nearest-neighbour readouts over a placed 2D frame. +//! Nearest-neighbour readouts with row-ordered distance ties. //! -//! [`KdTree`] indexes a borrowed point slice, the frame, and answers the exact `k` nearest other -//! rows of any frame row, ascending by squared distance with ties resolved by row. The frame -//! arrives as an [`IdSlice`], so every readout names rows in the frame's own id domain and a -//! consumer never rediscovers which domain a raw position meant. Equality with a full scan is -//! the contract: a query selects exactly the rows that sorting every other row's -//! [`Vec2::distance_squared_wide`] reading would select. The tree only accelerates that -//! selection, and no readout depends on its internal shape. A point query -//! ([`KdTree::nearest_point_in`]) answers the exact `k` nearest frame rows of any finite point -//! under the same contract, with no row excluded. +//! [`KdTree`] indexes a borrowed [`FinitePointField`], the frame. Row queries exclude the query +//! row. Point queries accept any finite point and exclude no row. Both return up to `k` entries +//! ordered by squared distance, then row ID. Every published distance uses +//! [`Vec2::distance_squared_wide`]. Duplicated positions retain distinct row identities. //! -//! Readouts are deterministic. A readout is a function of the frame bytes, the query, and `k`, -//! and row ids double as the tie-break identity, so a frame with duplicated positions still -//! orders every readout totally. +//! # Example +//! +//! This in-crate example is ignored because the math API is crate-private. //! //! ```ignore +//! use hashql_core::id::IdSlice; +//! use crate::math::{FinitePointField, KdTree, Vec2, nz}; +//! # hashql_core::id::newtype! { #[id(const)] struct RowId(u32) } //! let points = [Vec2::new(0.0, 0.0), Vec2::new(1.0, 0.0), Vec2::new(0.0, 2.0)]; //! let frame = FinitePointField::new(IdSlice::::from_raw(&points)) //! .expect("the example points are finite"); //! let tree = KdTree::build(frame); //! -//! let neighbours = tree.nearest(RowId::new(0), NonZero::new(2).expect("two is nonzero")); +//! let neighbours = tree.nearest(RowId::new(0), nz!(2)); //! assert_eq!(neighbours[0].row, RowId::new(1)); //! assert_eq!(neighbours[1].row, RowId::new(2)); //! ``` //! -//! # Engine +//! # Selection //! -//! The index is kiddo's immutable kd-tree over the frame's `f32` coordinates, with the -//! Eytzinger stem layout and soft-bucketed arena leaves, so a run of co-located rows wider than -//! a bucket becomes one over-full leaf instead of refusing construction. The engine computes in -//! `f64` over its `f32` storage, and its readings are bit-identical to -//! [`Vec2::distance_squared_wide`]: it widens each coordinate exactly before subtracting, and -//! its squared distance rounds once per multiply and once per add, x before y. That identity is -//! what lets the engine's radius selection decide membership under the one metric. +//! A nearest-`k` engine query can choose arbitrary members of an equal-distance class at its +//! boundary. Selecting the entire boundary class before sorting makes the row ID decide ties. The +//! first walk probes for a distance boundary. The second performs an inclusive radius query at that +//! boundary. Each candidate is re-read through [`Vec2::distance_squared_wide`], ordered by +//! `(distance_squared, row)`, and retained only if it is among the first `k`. //! -//! The engine resolves equal readings in traversal order, so a single k-query cannot honour the -//! `(reading, row)` tie contract when a tie class straddles the boundary: which co-located rows -//! enter the result would depend on the tree's internal shape. A readout therefore composes two -//! walks. The first probes for the k-th smallest reading, the boundary. The second selects -//! every row reading at most the boundary, which admits each boundary tie class whole; the -//! readout then re-reads every candidate through [`Vec2::distance_squared_wide`], orders by -//! `(reading, row)`, and keeps the first `k`. +//! With complete radius selection and the correct probed boundary, this composition equals sorting +//! a full scan. It includes every closer row and every boundary tie before applying the row +//! ordering. A row query probes one extra entry to account for its excluded zero-distance row. //! //! # Precision //! -//! Coordinates widen exactly from `f32` to `f64` before any arithmetic, and squared distances -//! accumulate in `f64` ([`Vec2::distance_squared_wide`]). A consumer that compares its own -//! readings against the tree's computes them through that one metric, so tie sets never depend -//! on the call site. +//! Finite `f32` coordinates widen exactly to `f64` before subtraction. The engine's mixed-precision +//! leaf metric squares each rounded difference separately and adds x before y, matching +//! [`Vec2::distance_squared_wide`]. Use that method when comparing published readings. +//! +//! Radius membership also depends on the engine's rectangle bounds. Their incremental `f64` updates +//! can round above the point metric at a box corner. The radius query makes no allowance for +//! outward rounding. Re-reading and sorting candidates preserves their published distances and +//! ordering, but cannot recover a row pruned at a rounding-sensitive boundary. //! //! # Complexity //! -//! The build median-splits on alternating axes in parallel above the engine's own threshold, -//! and runs in expected `O(N·log N)` for `N` rows. A readout on a well-spread frame walks the -//! tree twice in expected `O(log N + k)`. Rows sharing one position defeat the spatial pruning -//! and degrade a readout toward the full `O(N)` scan, with the result unchanged. The index owns -//! a copy of the coordinates and one item per row beside the borrowed frame, 16 bytes per row -//! for a 64-bit id. +//! Construction copies the coordinates and one item per row, using soft buckets that permit +//! co-located rows to exceed [`BUCKET_ROWS`]. The point payload is 8 bytes plus the ID size per +//! row. Stem storage, leaf extents, alignment padding and spare capacity add to that payload. +//! +//! A readout performs two tree walks and sorts `m` radius candidates, where `m` can be the entire +//! frame even for a small `k`. Sorting costs O(m log m) comparisons in the worst case. The returned +//! vector retains the candidate allocation after truncation. Its initial capacity is +//! `k.saturating_add(1)`, independent of frame size. The supplied allocator controls this vector. +//! Multi-entry engine probes use separate allocations. #![expect( clippy::min_ident_chars, @@ -78,28 +77,13 @@ use super::{FinitePointField, scalar::DNonNegative, vec2::Vec2}; #[cfg(test)] mod tests; -/// A frame row as the engine stores it against a point. -/// -/// The engine keeps items in fixed-size leaf arrays that it initialises before filling, so its -/// item type must have a default for the unused tail of a partly-filled leaf. A row id has none: -/// every id names a real row. This wrapper supplies the one the engine needs, and the wrapper -/// exists so that the requirement is stated here rather than forced onto the id domain. -/// -/// The padding is [`Id::MIN`], which is also a real row. Nothing distinguishes the two by value, -/// and nothing needs to: a query reads only within a leaf's extent, and every item inside an -/// extent is written at construction. The alternative, an extra inhabitant through [`Option`], -/// would make padding loud at a cost of double the item storage, because a row id wraps a plain -/// integer and leaves no niche for one. Items are the per-point storage the engine scans, so -/// that is the wrong trade. -/// -/// The ordering is the id's own and exists because the engine's query bound asks for one; -/// consumers order distance ties by row, and the engine's k-nearest queries never consult it. +/// A frame-row identity for the engine's stored items. #[derive(Copy, Clone, Debug, PartialEq, Eq, PartialOrd, Ord)] #[repr(transparent)] struct Leaf(N); impl Leaf { - /// Wraps the frame row. + /// Creates an engine item naming `row`. pub(crate) const fn new(row: N) -> Self { Self(row) } @@ -110,6 +94,9 @@ impl Leaf { } } +// kiddo's Content bound requires Default, including for a nearest-one query's initial best item. +// VecOfArenas copies populated items only. MIN is a provisional item value, not leaf padding or an +// extra ID inhabitant. impl Default for Leaf where N: Id, @@ -120,12 +107,15 @@ where } /// A frame row together with its squared distance to the query. +/// +/// Equality and ordering compare `(distance_squared, row)`. #[derive(Debug, Copy, Clone)] pub(crate) struct KdNeighbour { /// The neighbouring frame row. pub row: I, - /// The row's squared Euclidean distance to the query, per - /// [`Vec2::distance_squared_wide`]. + /// The row's squared distance to the query. + /// + /// Computed by [`Vec2::distance_squared_wide`]. pub distance_squared: DNonNegative, } @@ -161,10 +151,10 @@ where } } -/// The engine's leaf bucket capacity, in rows. +/// The engine's target leaf size, exceeded when a split cannot separate coordinates. const BUCKET_ROWS: usize = 32; -/// Kiddo's immutable tree over the frame's widened coordinates, one [`Leaf`] item per row. +/// The engine over stored `f32` coordinates and frame-row identities. type Engine = kiddo::kd_tree::KdTree< f32, Leaf, @@ -174,15 +164,15 @@ type Engine = kiddo::kd_tree::KdTree< BUCKET_ROWS, >; -/// An exact k-nearest-neighbour index over a borrowed 2D frame. +/// A nearest-neighbour index over a borrowed finite 2D frame. /// -/// Building validates the frame once and borrows it for the tree's lifetime. Frame rows are the -/// identities: a query names a row, and readouts name rows. The module documentation states the -/// exactness, determinism, and complexity guarantees. +/// Construction borrows the validated field for the tree's lifetime. Readouts use the frame's row +/// IDs. See the [module documentation](crate::math::kdtree) for the selection model, precision +/// limits and allocation costs. pub(crate) struct KdTree<'frame, I> { /// The borrowed frame, indexed by row. points: &'frame FinitePointField, - /// The engine over the frame's widened coordinates. + /// The engine over the frame's stored coordinates. engine: Engine, } @@ -190,15 +180,15 @@ impl<'frame, I> KdTree<'frame, I> where I: Id, { - /// Builds the index over `points`. + /// Builds an index whose row IDs address `points`. /// - /// The field becomes the frame, so row `r` is `points[r]` and readouts name these rows. - /// Finiteness arrives proven with the field, so construction cannot refuse. + /// The field supplies the finite-coordinate invariant. Soft buckets permit repeated positions. + /// Construction uses the engine's adaptive serial/parallel policy. /// /// # Panics /// - /// This panics when the frame holds more rows than `I` addresses, which the frame's - /// constructor is contracted to prevent. + /// Panics if a stored row index cannot be represented by `I` or if an engine allocation exceeds + /// its capacity limits. pub(crate) fn build(points: &'frame FinitePointField) -> Self { let engine = Engine::new_from_source_parallel( points.as_raw(), @@ -213,16 +203,17 @@ where Self { points, engine } } - /// Returns the `k` nearest other rows of `row`, allocating the readout in `alloc`. + /// Selects up to `k` other rows, allocating candidates in `alloc`. /// - /// The readout ascends by `(distance_squared, row)`: exactly the first `k` entries of the - /// sorted full scan over every other row. A bump allocator makes the readout free to create - /// and abandon per query: allocate a reading loop's readouts in a scratch arena and reset it - /// between readings. + /// The readout ascends by `(distance_squared, row)`, subject to the module's radius-selection + /// precision limits. The result retains its candidate capacity after truncation. Multi-entry + /// engine probes allocate separately. /// /// # Panics /// - /// This panics when `row` is not a frame row. + /// Panics if `row` is outside the frame, if the one-past-end frame index is not representable + /// by `I`, or if the requested candidate capacity exceeds the vector's limits. The initial + /// reservation uses `k.saturating_add(1)` even for a smaller nonempty frame. #[must_use] pub(crate) fn nearest_in( &self, @@ -238,28 +229,29 @@ where self.readout_in(self.points[row], Some(row), k, alloc) } - /// Returns the `k` nearest other rows of `row`: [`nearest_in`](Self::nearest_in) in the - /// global allocator. + /// Selects up to `k` other rows using the global allocator. + /// + /// See [`Self::nearest_in`] for ordering, precision and allocation behavior. /// /// # Panics /// - /// This panics when `row` is not a frame row. + /// Panics under the same conditions as [`Self::nearest_in`]. #[must_use] pub(crate) fn nearest(&self, row: I, k: NonZero) -> Vec> { self.nearest_in(row, k, Global) } - /// Returns the `k` nearest frame rows of `point`, allocating the readout in `alloc`. + /// Selects up to `k` rows near `point`, allocating candidates in `alloc`. /// - /// The query point needs no frame membership, and no row is excluded: a frame row co-located - /// with `point` is a candidate at distance zero. The readout holds one entry per frame row up - /// to `k`, so a frame with fewer than `k` rows returns them all, ordered as - /// [`nearest_in`](Self::nearest_in) states. + /// The point needs no frame membership. No row is excluded, including a row at the same + /// position. An empty frame returns an empty vector. Ordering and radius-selection precision + /// follow [`Self::nearest_in`]. /// /// # Panics /// - /// This panics when `point` has a NaN or infinite component, which would break the pruning's - /// soundness argument. + /// Panics if `point` has a NaN or infinite component, or if the requested candidate capacity + /// exceeds the vector's limits. The initial reservation uses `k.saturating_add(1)` even for a + /// smaller nonempty frame. #[must_use] pub(crate) fn nearest_point_in( &self, @@ -275,27 +267,29 @@ where self.readout_in(point, None, k, alloc) } - /// Returns the `k` nearest frame rows of `point`: - /// [`nearest_point_in`](Self::nearest_point_in) in the global allocator. + /// Selects up to `k` rows near `point` using the global allocator. + /// + /// See [`Self::nearest_point_in`] for ordering, precision and allocation behavior. /// /// # Panics /// - /// This panics when `point` has a NaN or infinite component. + /// Panics under the same conditions as [`Self::nearest_point_in`]. #[must_use] pub(crate) fn nearest_point(&self, point: Vec2, k: NonZero) -> Vec> { self.nearest_point_in(point, k, Global) } - /// Selects the exact `k`-set of `query` under `(reading, row)`, the two-walk composition. + /// Probes a boundary, gathers its radius candidates and applies row-ordered truncation. + /// + /// `query` must be finite. If `exclude` is present, it must name a frame row at `query`. For N + /// frame rows, the probe requests min(k + 1, N) entries for exclusion and min(k, N) otherwise, + /// with saturating addition. Including the excluded zero-distance row in this count preserves + /// the desired boundary even when the probe chooses other rows from the same tie class. The + /// module's selection argument requires complete engine radius membership. + /// + /// # Panics /// - /// The probe walk asks the engine for the `k` smallest readings, `k + 1` when `exclude` - /// names a frame row, because that row's own zero reading occupies one slot. The largest - /// probed reading is the boundary: the multiset of the `k` smallest readings is a function - /// of the frame alone, whichever tied rows the engine kept. The selection walk then admits - /// every row reading at most the boundary, inclusively, so each boundary tie class arrives - /// whole and the engine's traversal order decides nothing. Re-reading the candidates through - /// [`Vec2::distance_squared_wide`] makes the published readings the one metric's by - /// construction. The engine's bit-identical readings only steered the selection. + /// Panics if the candidate reservation exceeds the vector's capacity limits. fn readout_in( &self, query: Vec2, @@ -316,10 +310,9 @@ where let mut scratch = QueryScratch::new(); - // A single-reading probe is its own boundary, and the single-item query returns without - // the result vector a top-k probe buffers into. The engine has no visitor for a top-k: - // a bounded k-set evicts rows while the walk runs, so it only exists once the walk ends, - // and the executed vector is that collection itself. + // a nearest-one probe returns its boundary without a result vector. Larger probes retain a + // bounded collection until traversal has finished, then reduce its distances to the + // boundary. let boundary = if probe_size == NonZero::::MIN { self.engine .query(query.as_array()) @@ -347,8 +340,8 @@ where boundary }; - // Without boundary ties the selection returns exactly the probed rows, so `k + 1` is the - // readout's usual size and a wider tie class is the one case that grows the buffer. + // the probe bounds its size by the frame length, but this reservation uses k directly. + // Radius ties can grow the candidate vector beyond this initial capacity. let mut readout = Vec::with_capacity_in(k.get().saturating_add(1), alloc); self.engine .query(query.as_array()) @@ -370,6 +363,7 @@ where readout } + /// Returns the point slice the tree indexes, in row order. pub(crate) const fn points(&self) -> &'frame IdSlice { self.points } diff --git a/libs/@local/graph/atlas/src/math/kdtree/tests.rs b/libs/@local/graph/atlas/src/math/kdtree/tests.rs index 9d335029149..d6b596bae61 100644 --- a/libs/@local/graph/atlas/src/math/kdtree/tests.rs +++ b/libs/@local/graph/atlas/src/math/kdtree/tests.rs @@ -17,18 +17,26 @@ use super::{BUCKET_ROWS, KdNeighbour, KdTree}; use crate::math::{DNonNegative, FinitePointField, Vec2}; hashql_core::id::newtype! { - /// The test frames' row domain. + /// A row identity in a test frame. + /// #[id(const)] struct RowId(u32) } -/// Views a finite point slice as a proven frame over the test row domain. +/// Views points as a test frame without validating coordinates. +/// +/// Every component must be finite. fn frame(points: &[Vec2]) -> &FinitePointField { FinitePointField::new_unchecked(IdSlice::from_raw(points)) } -/// Selects the `k` nearest rows by sorting every other row's reading, the reference a readout -/// must equal. +/// Sorts the full frame by distance and row, excluding `row`. +/// +/// Every component must be finite. +/// +/// # Panics +/// +/// Panics if `row` is outside the frame or if a frame index is outside the test ID domain. fn full_scan(frame: &IdSlice, row: RowId, k: usize) -> Vec> { let query = frame[row]; let mut readings: Vec> = frame @@ -44,7 +52,14 @@ fn full_scan(frame: &IdSlice, row: RowId, k: usize) -> Vec, point: Vec2, k: usize) -> Vec> { let mut readings: Vec> = frame .ids() @@ -76,7 +97,7 @@ fn full_scan_point(frame: &IdSlice, point: Vec2, k: usize) -> Vec Vec { let mut rng = Xoshiro256PlusPlus::seed_from_u64(seed); iter::repeat_with(|| { @@ -91,7 +112,7 @@ fn scattered(seed: u64, rows: usize) -> Vec { #[test] fn readouts_equal_the_full_scan_on_scattered_frames() { - // Frame lengths straddle the leaf bucket, and the longest reaches several split levels. + // frame lengths straddle the leaf bucket, and the longest reaches several split levels. for rows in [ 1, 2, @@ -111,8 +132,8 @@ fn readouts_equal_the_full_scan_on_scattered_frames() { #[test] fn duplicated_positions_resolve_ties_by_row() { - // Sixty-four rows over nine distinct positions guarantee co-located tie classes wider than - // most of the tested k values, so the cut lands inside a class and the row tie-break decides. + // 64 rows drawn from nine positions give at least one co-located class of eight or more. Counts + // 1, 3 and 6 cut within that class for a query at its position. let mut rng = Xoshiro256PlusPlus::seed_from_u64(7); let positions: Vec = (0..3_u16) .flat_map(|x| (0..3_u16).map(move |y| Vec2::new(f32::from(x), f32::from(y)))) @@ -126,7 +147,8 @@ fn duplicated_positions_resolve_ties_by_row() { #[test] fn a_fully_co_located_frame_orders_by_row_alone() { - // Forty rows exceed one leaf bucket, so construction takes the over-full soft-bucket path. + // 40 identical positions exceed the target bucket size and cannot be separated by an axis + // split. let points = vec![Vec2::new(2.5, -3.5); BUCKET_ROWS + 8]; let tree = KdTree::build(frame(&points)); @@ -144,7 +166,7 @@ fn a_fully_co_located_frame_orders_by_row_alone() { #[test] fn an_integer_lattice_cuts_inside_a_tie_class() { - // An interior lattice row has four neighbours at distance² 1, so k = 3 cuts inside that + // an interior lattice row has four neighbours at squared distance 1. k = 3 cuts within that // exact tie class. let points: Vec = (0..5_u16) .flat_map(|x| (0..5_u16).map(move |y| Vec2::new(f32::from(x), f32::from(y)))) @@ -181,7 +203,6 @@ fn k_at_least_the_frame_returns_every_other_row() { #[test] fn the_query_row_is_never_a_readout_while_its_co_located_rows_are() { let mut points = scattered(13, 20); - // Rows 3 and 17 sit exactly on the query row 3's position. points[17] = points[3]; let tree = KdTree::build(frame(&points)); @@ -234,7 +255,7 @@ fn point_readouts_equal_the_full_scan() { for rows in [1, 2, BUCKET_ROWS, BUCKET_ROWS + 1, 100, 333] { for seed in [29, 31] { let points = scattered(seed, rows); - // Off-frame query points from an independent stream, plus every frame position. + // differently seeded queries extend coverage beyond the frame positions. let queries: Vec = scattered(seed ^ 0xBEEF, 24) .into_iter() .chain(points.iter().copied()) @@ -266,7 +287,6 @@ fn a_point_query_excludes_no_row() { let tree = KdTree::build(frame); let k = NonZero::new(2).expect("two is nonzero"); - // A row query from row 1 excludes row 1 itself. let neighbours = tree.nearest(RowId::new(1), k); assert!( neighbours @@ -274,7 +294,6 @@ fn a_point_query_excludes_no_row() { .all(|neighbour| neighbour.row != RowId::new(1)) ); - // A point query from row 1's position keeps it, at distance zero and ahead of every other. let neighbours = tree.nearest_point(frame[RowId::new(1)], k); assert_eq!( neighbours[0], diff --git a/libs/@local/graph/atlas/src/math/kernel/bench.rs b/libs/@local/graph/atlas/src/math/kernel/bench.rs index adc475de4ad..3c3a1760bea 100644 --- a/libs/@local/graph/atlas/src/math/kernel/bench.rs +++ b/libs/@local/graph/atlas/src/math/kernel/bench.rs @@ -1,14 +1,12 @@ -//! Measurement seam for the vendored transcendental kernels. +//! Benchmark entry points for production and candidate transcendental kernels. //! -//! The `math_kernels` benchmark target measures each production wrapper exactly as production calls -//! it, so a rewrite of the vendored kernels shows up as an instruction-count or cycle change -//! against the saved per-event baselines. Nothing here is API for consumers of the crate. +//! The exponential and power functions expose the production wrappers. The table functions expose +//! alternative exponential implementations for comparison on the same inputs. Inlining these entry +//! points permits the benchmark caller to optimize the surrounding expression. use core::simd::{f32x4, f32x8, f64x4}; -/// Base-e exponential of each `f64` lane. -/// -/// As the production wrapper computes it. +/// Approximates each lane's exponential through [`super::exp_f64x4`]. #[expect( clippy::inline_always, reason = "the seam must measure the wrapper as production calls it: transparently inlined, \ @@ -20,9 +18,7 @@ pub fn exp_f64x4(values: f64x4) -> f64x4 { super::exp_f64x4(values) } -/// Base-e exponential of each `f32` lane. -/// -/// As the production wrapper computes it. +/// Approximates each lane's exponential through [`super::exp_f32x8`]. #[expect( clippy::inline_always, reason = "the seam must measure the wrapper as production calls it: transparently inlined, \ @@ -34,10 +30,9 @@ pub fn exp_f32x8(values: f32x8) -> f32x8 { super::exp_f32x8(values) } -/// Base-e exponential of each `f32` lane, table-based alternative in its portable gather form. +/// Approximates each lane's exponential with portable table gathers. /// -/// Not a production wrapper: this entry keeps the 16-entry hi/lo-table candidate measured against -/// [`exp_f32x8`]. +/// This exposes the 16-entry split-table candidate for comparison with [`exp_f32x8`]. #[expect( clippy::inline_always, reason = "the seam must measure the candidate as a production wrapper would call it: \ @@ -49,10 +44,10 @@ pub fn exp_f32x8_table_gather(values: f32x8) -> f32x8 { super::exp_table::exp_f32(values) } -/// Base-e exponential of each `f32` lane, table-based alternative through paired `TBL4` lookups. +/// Approximates each lane's exponential with target-selected table lookups. /// -/// Not a production wrapper: this entry keeps the aarch64 lookup form of the table candidate -/// measured against [`exp_f32x8`]. +/// On little-endian aarch64, the candidate uses paired `TBL4` lookups when NEON is enabled and +/// portable gathers otherwise. Compare with [`exp_f32x8`]. #[cfg(all(target_arch = "aarch64", target_endian = "little"))] #[expect( clippy::inline_always, @@ -65,9 +60,10 @@ pub fn exp_f32x8_table_tbl4(values: f32x8) -> f32x8 { super::exp_table::exp_f32x8(values) } -/// Lanewise power for strictly positive bases. +/// Approximates lanewise powers through [`super::pow_f32x4`]. /// -/// As the production wrapper computes it. +/// Bases must be strictly positive and finite. The production wrapper's range and precision limits +/// apply. #[expect( clippy::inline_always, reason = "the seam must measure the wrapper as production calls it: transparently inlined, \ diff --git a/libs/@local/graph/atlas/src/math/kernel/exp_table.rs b/libs/@local/graph/atlas/src/math/kernel/exp_table.rs index 587e4a9ed22..56c240b663e 100644 --- a/libs/@local/graph/atlas/src/math/kernel/exp_table.rs +++ b/libs/@local/graph/atlas/src/math/kernel/exp_table.rs @@ -1,25 +1,25 @@ -//! Table-based `exp` for `f32` lanes: 16-entry hi/lo table of `2^(j/16)`, degree-3 tail, -//! hi/lo-corrected reconstruction. +//! Table-based exponential approximation for single-precision SIMD lanes. //! //! An alternative to the polynomial [`exp_f32`](super::sleef::exp_f32) with the same edge-case //! contract, measured against it under the `math_kernels` benchmark target. //! //! # Design //! -//! Reduction: `n = round(x · 16/ln 2)`, split as `n = 16q + j`, so `e^x = 2^q · 2^(j/16) · e^r` -//! with `|r| ≤ ln(2)/32 ≈ 0.0217`. The table stores each entry `T = 2^(j/16)` as an f32 pair -//! `(T_hi, T_lo)` with `T_lo = round(T - T_hi)`, and the kernel reconstructs the product `T · e^r` -//! as +//! For finite x in the unclamped interval [−104, 100], choose integer n near x · 16/ln 2 and split +//! n = 16q + j with 0 ≤ j < 16. The identity eˣ = 2ᑫ · 2^(j/16) · eʳ uses residual r = x − n · +//! ln(2)/16. Ideal nearest-integer selection bounds |r| by ln(2)/32 ≈ 0.0217. The implementation +//! rounds the selection and residual in `f32`. //! -//! ```text -//! T_hi + fma(T_hi, expm1(r), T_lo) -//! ``` +//! For table value T = 2^(j/16), define Tₕ = round₃₂(T) and Tₗ = round₃₂(T − Tₕ), where round₃₂ +//! rounds to nearest `f32`, ties to even. A cubic polynomial approximates eʳ − 1. Reconstruction +//! approximates T · eʳ with Tₕ + fma(Tₕ, expm1(r), Tₗ), using the table's rounded high and low +//! parts. //! -//! which keeps the table's rounding error out of the result. The only half-ulp-scale rounding -//! left is the final add. Error budget: 0.5 (final add) + 0.078 (tail fit, measured in exact -//! arithmetic) + ≈0.03 (small-scale roundings + reduction residual). The Cody-Waite split keeps -//! `n · LN2_16_HI` exact for `n < 4096` (12-bit significand times `|n| ≤ 2402`), and both -//! [`scale_by_pow2_f32`] multiplies stay exact powers of two as in `exp_f32`. +//! The low part corrects the high part's table-rounding error before the final addition. +//! The recorded design budget assigns 0.5 ULP to that addition, 0.078 ULP to the tail fit in +//! exact arithmetic, and about 0.03 ULP to the smaller rounding terms and reduction residual. +//! This budget concerns the unscaled reconstruction. Subnormal results can round again during +//! [`scale_by_pow2_f32`]. The tests measure the complete result against `f64` libm. //! //! # Lookup portability //! @@ -36,15 +36,19 @@ use std::simd::StdFloat as _; use super::sleef::scale_by_pow2_f32; -/// `16 / ln(2)` (= 23.083120346069336). +/// The rounded reduction multiplier 16/ln(2), equal to 23.083120346069336. const INVLN2_16: f32 = f32::from_bits(0x41B8_AA3B); -/// `ln(2)/16` with the low 12 mantissa bits zeroed: a 12-bit significand, so `n · LN2_16_HI` is -/// exact for `|n| < 4096` (the reduction produces `|n| ≤ 2402`). +/// The coarse part of `ln(2)/16`, with twelve low significand bits cleared. +// A product is exact when its significand and exponent fit the destination. This constant retains +// at most twelve bits in its significand, and the unclamped reduction integer satisfies |n| ≤ 2402, +// at most twelve bits. Therefore their product fits f32 exactly. The low-part FMA still rounds the +// residual. const LN2_16_HI: f32 = { let base = core::f32::consts::LN_2 / 16.; // exact: power-of-two divide f32::from_bits(base.to_bits() & !0xFFF) }; +/// The rounded low part of ln(2)/16 after subtracting [`LN2_16_HI`]. #[expect( clippy::cast_possible_truncation, reason = "the cast is the derivation's rounding step: the remainder is correctly rounded into \ @@ -52,12 +56,16 @@ const LN2_16_HI: f32 = { )] const LN2_16_LO: f32 = (core::f64::consts::LN_2 / 16. - (LN2_16_HI as f64)) as f32; -/// Degree-3 tail of `e^r - 1 = r + r^2 (C2 + C3 r)` over `|r| ≤ ln(2)/32`; near-minimax fit -/// (Chebyshev projection, coefficients rounded jointly), 0.078 ulp intrinsic error. +/// The quadratic coefficient in the cubic approximation of eʳ − 1. +/// +/// The tail r + r²(C₂ + C₃r) approximates eʳ − 1 on |r| ≤ ln(2)/32. The jointly rounded +/// Chebyshev-fit coefficients have a recorded intrinsic-error budget of 0.078 ULP. The module's +/// reconstruction budget accounts for separate rounding terms. const C2: f32 = f32::from_bits(0x3F00_00A4); // 0.5000097751617432 +/// Cubic coefficient of the degree-3 tail, fitted jointly with [`C2`]. const C3: f32 = f32::from_bits(0x3E2A_AB2E); // 0.1666686236858368 -/// `2^(j/16)` rounded to f32. +/// The high parts Tₕ of the table split defined in the module model. const EXP16_HI: [f32; 16] = [ f32::from_bits(0x3F80_0000), // 1.0 f32::from_bits(0x3F85_AAC3), // 1.0442737340927124 @@ -77,7 +85,7 @@ const EXP16_HI: [f32; 16] = [ f32::from_bits(0x3FF5_257D), // 1.9152065515518188 ]; -/// `round(2^(j/16) - EXP16_HI[j])`: the sub-half-ulp remainder of each entry. +/// The low parts Tₗ of the table split defined in the module model. const EXP16_LO: [f32; 16] = [ f32::from_bits(0x0000_0000), // 0.0 f32::from_bits(0x334F_9891), // 4.8334701574503924e-8 @@ -112,9 +120,9 @@ fn tbl4_lookup(table: &[f32; 16], index: Simd) -> Simd { let byte_index = (index << Simd::splat(2)) * Simd::splat(0x0101_0101) + Simd::splat(0x0302_0100); - // SAFETY: NEON is mandatory on aarch64; the table is 64 contiguous, initialized bytes, which - // is exactly what `vld1q_u8_x4` reads (no alignment requirement). LLVM hoists the table load - // out of loops. + // SAFETY: These intrinsics require NEON, and `vld1q_u8_x4` accepts 64 readable bytes at any + // alignment. This function's cfg requires NEON enabled, and the shared table borrow keeps those + // initialized bytes readable. Therefore the load and NEON operations are safe. unsafe { let entries = vld1q_u8_x4(table.as_ptr().cast()); let indices = vreinterpretq_u8_u32(uint32x4_t::from(byte_index)); @@ -122,7 +130,12 @@ fn tbl4_lookup(table: &[f32; 16], index: Simd) -> Simd { } } -/// Shared tail: polynomial, hi/lo reconstruction, scaling, clamps. +/// Reconstructs the scaled exponential and applies the final range masks. +/// +/// `reduced`, `quotient` and the table pair must come from the module's range reduction of +/// `values`. Exceptional and out-of-range lanes may produce intermediate values outside the scaling +/// helper's numerical contract. The final masks set values below −104 to zero and above 100 to +/// infinity. NaN lanes propagate through the polynomial arithmetic. #[inline] fn finish( values: Simd, @@ -145,10 +158,9 @@ fn finish( .select(Simd::splat(f32::INFINITY), result) } -/// Table-based counterpart of [`exp_f32`](super::sleef::exp_f32), portable form. +/// Evaluates lanes with portable gather lookups. /// -/// Semantically identical on every target; lookup speed is target-dependent (see the module -/// docs). Prefer the `exp_f32x4_table` form on aarch64. +/// Use [`exp_f32x4`] or [`exp_f32x8`] for fixed lane counts to select NEON lookups where enabled. #[inline] pub(crate) fn exp_f32(values: Simd) -> Simd { let nearest = (values * Simd::splat(INVLN2_16)).round_ties_even(); @@ -240,21 +252,23 @@ mod tests { } } - // The stride is odd, so consecutive samples differ in exponent/mantissa phase. Full-bit-range - // iteration covers negative inputs, subnormals, both zeros, both infinities, and NaN payloads - // without listing them. + // the odd stride samples different exponent and significand bit patterns across both signs. + // It omits some special encodings, which `exp_f32_specials` supplies directly. + /// The sampling stride through the `u32` bit space. const F32_STRIDE: usize = 641; /// Allowed kernel-to-reference distance in representation steps. /// - /// The kernel's 1.0-ulp accuracy tier plus half a step for the reference's own - /// correctly-rounded narrowing, rounded up to whole steps. + /// The budget is ⌈1.0 + 0.5⌉ = 2, adding a nominal half-step narrowing allowance to the design + /// tier. The libm reference is itself approximate. This encoding-distance check can also accept + /// adjacent finite/infinite or infinite/NaN encodings. const U10_F32_TOLERANCE: u64 = 2; - /// Position of a value in the ordered sequence of representable `f32`s. + /// Maps a single-precision encoding to a signed representation-step index. /// - /// Adjacent representable values differ by one across the whole line, including zeros, - /// subnormals, and infinities, so one distance bound holds without per-class cases. + /// Adjacent distinct non-NaN values differ by one, including the subnormal and infinity + /// boundaries. Both zeros map to zero. NaN encodings have indices beyond the corresponding + /// infinity, without numerical-distance semantics. fn ordered_f32(value: f32) -> i64 { let bits = value.to_bits(); if bits & 0x8000_0000 == 0 { @@ -266,10 +280,8 @@ mod tests { /// Strided samples of the full input bit range track scalar libm inside the step tolerance. /// - /// The agreement tests in this module compare entry points that share every constant and - /// every reconstruction step, so drift in that shared arithmetic moves all of them - /// identically and only an external reference can pin it. `ulp_sweep.rs` holds the - /// exhaustive `#[ignore]` form of this check. + /// Compares the generic table kernel with a wider-precision libm reference. Entry-point + /// agreement alone cannot check the arithmetic shared by all lookup implementations. #[test] #[expect( clippy::cast_possible_truncation, @@ -306,11 +318,6 @@ mod tests { } } - /// Edge-case results are exact. - /// - /// The contract matches [`exp_f32`](super::super::sleef::exp_f32). Zero yields exactly one - /// and negative infinity exactly zero. An infinity lane stays infinite and a NaN lane stays - /// NaN. The assertions compare bit patterns, so a merely-close value fails. #[test] fn edge_cases_are_exact() { let output = exp_f32::<4>(Simd::from_array([ @@ -327,8 +334,8 @@ mod tests { /// Every named entry point agrees with the generic kernel bit for bit. /// - /// On aarch64 the entry points are the TBL4 forms, elsewhere the passthrough fallbacks, so - /// the sweep pins the agreement on every target this module compiles for. + /// Little-endian aarch64 with NEON enabled uses TBL4. Other configurations use portable gather + /// lookups. #[test] fn entry_points_agree_with_the_generic_kernel() { let mut lanes = [0.0_f32; 8]; diff --git a/libs/@local/graph/atlas/src/math/kernel/mod.rs b/libs/@local/graph/atlas/src/math/kernel/mod.rs index fbceae85aa1..c375094c99f 100644 --- a/libs/@local/graph/atlas/src/math/kernel/mod.rs +++ b/libs/@local/graph/atlas/src/math/kernel/mod.rs @@ -1,17 +1,13 @@ -//! Shared portable-SIMD operations. +//! Shared fused arithmetic and approximate transcendental functions for SIMD lanes. //! -//! The multiply-add wrappers keep the fusion semantic, rounding once per lane to the result IEEE -//! 754 defines on every target, so content-hashed artifacts reproduce across platforms by -//! construction. The vendored SLEEF kernels in [`self::sleef`] vectorize the transcendentals. Each -//! wrapper documents its own accuracy bound, picked per kernel by measuring the consumers' -//! requirements against instruction counts, from the 1.0-ulp `u10` tier down to compositions of the -//! cheaper 3.5-ulp `u35` tier. These wrappers are the crate's single seam onto the vendored -//! kernels. Every consumer routes through here, and the `math::kernel` tests bound each wrapper's -//! error against scalar libm. -// `StdFloat` also exposes vector `exp`/`ln`, but the compiler lowers them -// to one libm call per lane on every current target (verified against the -// emitted assembly), which is why the bodies below call the vendored -// kernels instead. +//! The multiply-add wrappers round each product-plus-sum once. The vendored kernels in +//! [`self::sleef`] evaluate lane arithmetic with fixed coefficients and evaluation order, avoiding +//! a scalar transcendental call per lane. These choices specify arithmetic operations rather than +//! instruction counts or cross-platform NaN payloads. +//! +//! SLEEF's accuracy tiers distinguish 1.0-ULP (`u10`) and 3.5-ULP (`u35`) approximations. A +//! composition such as [`pow_f32x4`] does not inherit either bound. The tests compare samples +//! against scalar libm, with the reference precision and coverage described in [`self::sleef`]. use core::simd::{f32x4, f32x8, f64x4, f64x8}; @@ -23,17 +19,16 @@ mod sleef; #[cfg(test)] mod ulp_sweep; -/// Refuses to run on a CPU below the x86-64-v3 baseline the crate is compiled for. +/// Checks selected CPU feature bits required by the crate's x86-64 build. /// -/// The check reads the processor's own feature bits, so it stays live in a build that already -/// assumes the baseline, and a mismatched machine reports what it lacks instead of faulting on its -/// first vector instruction. Call it before anything else in `main`, ahead of argument parsing. On -/// targets whose baseline needs no runtime support (aarch64) it compiles to nothing. +/// Reads AVX, AVX2, FMA, BMI2 and OSXSAVE directly from CPUID. Call it before other work in `main` +/// to report missing features early. The function performs no check on other architectures. /// /// # Panics /// -/// This panics when the CPU reports no AVX, AVX2, FMA or BMI2 support, or no operating-system XSAVE -/// support. Whether the operating system has enabled YMM register state is outside the check. +/// On x86-64, this panics for `target_env = "sgx"` before reading CPUID. It also panics when the +/// CPU reports no AVX, AVX2, FMA or BMI2 support, or no operating-system XSAVE support. Whether the +/// operating system has enabled YMM register state is outside the check. #[cfg_attr( not(target_arch = "x86_64"), expect( @@ -52,9 +47,9 @@ pub(crate) fn verify_cpu_baseline() { let max_basic_leaf = __cpuid_count(0, 0).eax; let processor_info = __cpuid_count(1, 0); let extended_features = __cpuid_count(7, 0); - // The AVX and OSXSAVE bits guard the AVX2 reading: SKL052 leaves BMI bits set on Skylake - // parts with AVX disabled in firmware, and AVX2 is meaningless without OS-enabled extended - // state. + // Some Skylake parts that lack AVX falsely report BMI1/BMI2 support (SKL052). Checking AVX + // alongside BMI2 rejects that combination. AVX2 also requires operating-system support for + // extended state. let avx = processor_info.ecx & (1 << 28) != 0; let osxsave = processor_info.ecx & (1 << 27) != 0; let fma = processor_info.ecx & (1 << 12) != 0; @@ -68,11 +63,11 @@ pub(crate) fn verify_cpu_baseline() { } } -/// Fused multiply-add, correctly rounded on every target. +/// Computes a fused multiply-add in each lane. /// -/// The fusion is semantic. Each lane rounds once, so every lane matches scalar [`f32::mul_add`] bit -/// for bit. Both baselines the crate builds for lower it in hardware (aarch64 FMLA, x86-64 the -/// workspace's x86-64-v3 FMA), and lowering changes speed, never bits. +/// Each lane rounds the exact product-plus-sum once, as [`f32::mul_add`] does. This is an +/// arithmetic guarantee, including when the target implements fusion in software. NaN payloads +/// are not part of the guarantee. #[inline(always)] pub(crate) fn mul_add_f32x4(lhs: f32x4, rhs: f32x4, accumulator: f32x4) -> f32x4 { use std::simd::StdFloat as _; @@ -80,9 +75,9 @@ pub(crate) fn mul_add_f32x4(lhs: f32x4, rhs: f32x4, accumulator: f32x4) -> f32x4 lhs.mul_add(rhs, accumulator) } -/// Fused multiply-add, correctly rounded on every target. +/// Computes a fused multiply-add in each double-precision lane. /// -/// The `f64x4` counterpart of [`mul_add_f32x4`]. +/// The four-lane counterpart of [`mul_add_f32x4`], with the same rounding contract. #[inline(always)] pub(crate) fn mul_add_f64x4(lhs: f64x4, rhs: f64x4, accumulator: f64x4) -> f64x4 { use std::simd::StdFloat as _; @@ -90,9 +85,9 @@ pub(crate) fn mul_add_f64x4(lhs: f64x4, rhs: f64x4, accumulator: f64x4) -> f64x4 lhs.mul_add(rhs, accumulator) } -/// Fused multiply-add, correctly rounded on every target. +/// Computes a fused multiply-add in each double-precision lane. /// -/// The `f64x8` counterpart of [`mul_add_f32x4`]. +/// The eight-lane counterpart of [`mul_add_f32x4`], with the same rounding contract. #[inline(always)] pub(crate) fn mul_add_f64x8(lhs: f64x8, rhs: f64x8, accumulator: f64x8) -> f64x8 { use std::simd::StdFloat as _; @@ -100,7 +95,10 @@ pub(crate) fn mul_add_f64x8(lhs: f64x8, rhs: f64x8, accumulator: f64x8) -> f64x8 lhs.mul_add(rhs, accumulator) } -/// Exponential of each lane, accurate to 1.0 unit in the last place. +/// Approximates the base-e exponential of each double-precision lane. +/// +/// Uses [`sleef::exp_f64`]. Its sampled agreement with scalar libm does not establish a +/// worst-case ULP bound over every input. #[expect( clippy::inline_always, reason = "SIMD values cross non-inlined call boundaries through memory; the wrapper must be \ @@ -111,9 +109,10 @@ pub(crate) fn exp_f64x4(values: f64x4) -> f64x4 { sleef::exp_f64(values) } -/// Exponential of each lane, accurate to 1.0 unit in the last place. +/// Approximates the base-e exponential of each single-precision lane. /// -/// A zero lane yields exactly one, and a negative-infinity lane yields exactly zero. +/// Uses the `u10`-tier [`sleef::exp_f32`]. A zero lane yields exactly one, and a negative-infinity +/// lane yields exactly zero. #[expect( clippy::inline_always, reason = "SIMD values cross non-inlined call boundaries through memory; the wrapper must be \ @@ -124,29 +123,25 @@ pub(crate) fn exp_f32x8(values: f32x8) -> f32x8 { sleef::exp_f32(values) } -/// Raises each lane of `base` to the matching lane of `exponent`, for strictly positive bases. +/// Approximates each lane's power for strictly positive finite bases. +/// +/// Evaluates exp₂(p · log₂ b) for base b and exponent p through `u35`-tier stages. If the logarithm +/// has absolute error δₗ and multiplication contributes δₘ, the exponential's argument error is Δz +/// = pδₗ + δₘ. For a finite positive normal result, an exp₂ relative error δₑ gives composed +/// relative error 2^Δz · (1 + δₑ) − 1. For small errors this is approximately ln(2) · Δz + δₑ. The +/// exponent can amplify logarithm error before exp₂ is evaluated. /// -/// This evaluates the power as `exp2(exponent · log2(base))` through the vendored SLEEF 3.5-ulp -/// stages. The relative error grows with the magnitude of the result's binary exponent, from a few -/// units in the last place for results near one to the order of `1e-4` at the edges of the normal -/// range. Gradient kernels tolerate far more. For 1-ulp powers, take the scalar [`f32::powf`] per -/// lane instead. +/// This composition has no fixed ULP bound inherited from its stages. The tests allow relative +/// error `2e-4` on their finite sample grid. Scalar [`f32::powf`] is an alternative with +/// platform-dependent accuracy. /// -/// A `base` of zero yields zero for positive exponents, infinity for negative exponents, and NaN -/// when the exponent is also zero. Negative bases yield NaN. -// Measured on an M5 Max (per 4-lane call, criterion via darwin-kperf, fused ladders): this -// composition 68 instructions / 18 cycles, four scalar libm `powf` calls 311 / 36. The scalar -// near-tie in standalone cycles vanishes under load. Embedded in the attraction-coefficient -// arithmetic, the composition occupies idle issue slots while the scalar bodies compete for them. -// The tie also does not generalize across machines. It needs an out-of-order engine wide and deep -// enough to overlap four independent libm bodies (IPC ≈8.6 here) and Apple's branch-free `powf`. -// Production Linux targets have neither, and glibc's `powf` is a different, branchier function with -// different rounding. The composition's cost is the same wherever the binary runs, because the same -// vendored code produces bit-identical results on every platform, and that also keeps -// content-hashed fits reproducible across dev and prod. Inside the fused gradient kernels the -// instruction count also becomes the shared resource, and the composition leaves three quarters of -// the issue slots to the surrounding batch arithmetic while staying in vector registers. `StdFloat` -// offers no vector `pow`, and its `exp2`/`log2` scalarize to one libm call per lane. +/// A `base` of zero yields zero for positive finite exponents, infinity for negative finite +/// exponents, and NaN when the exponent is also zero. Negative bases yield NaN. Infinite bases or +/// exponents follow the intermediate logarithm and product, including NaN for an infinite base +/// raised to zero. +// measured on an M5 Max per four-lane call with Criterion and darwin-kperf: this composition used +// 68 instructions / 18 cycles, versus 311 / 36 for four scalar libm powf calls. The isolated +// comparison does not measure the complete gradient or predict another CPU's cost. #[expect( clippy::inline_always, reason = "SIMD values cross non-inlined call boundaries through memory; the wrapper must be \ @@ -168,14 +163,12 @@ mod tests { use super::{exp_f32x8, exp_f64x4, mul_add_f32x4, mul_add_f64x4, mul_add_f64x8, pow_f32x4}; - /// `mul_add_f32x4` rounds once per lane, against scalar [`f32::mul_add`] as the reference. + /// Distinguishes fused cancellation from separately rounded multiplication. /// - /// Lane 0 discriminates the fusion, so an unfused `lhs * rhs + accumulator` body fails here: `a - /// = b = 1 + 2⁻¹²` and `c = -(1 + 2⁻¹¹)` make `a·b` round to `1 + 2⁻¹¹` under tie-to-even - /// before the add (the exact product carries a `2⁻²⁴` term, exactly half the `f32` ulp at this - /// magnitude), so `a * b + c` gives `0.0` where the fused result is `2⁻²⁴`. Lane 1 is its - /// sign-negated twin, mirroring the same tie around zero. Lanes 2 and 3 are ordinary non-dyadic - /// values. + /// For a = b = 1 + 2⁻¹² and c = −(1 + 2⁻¹¹), the exact product contains a 2⁻²⁴ term, half the + /// binary32 ULP at this magnitude. Ties-to-even rounds a · b to 1 + 2⁻¹¹ before a separate + /// addition. Therefore the separate result is zero while the fused result is 2⁻²⁴. Lane 1 + /// negates the product and addend, giving the corresponding negative residual. #[test] fn mul_add_f32x4_rounds_once_per_lane() { let factor = 1.0_f32 + (-12.0_f32).exp2(); @@ -198,13 +191,11 @@ mod tests { } } - /// `mul_add_f64x4` rounds once per lane, against scalar [`f64::mul_add`] as the reference. + /// Distinguishes fused cancellation in double precision. /// - /// Lane 0 discriminates the fusion, so an unfused `lhs * rhs + accumulator` body fails here: `a - /// = b = 1 + 2⁻²⁷` and `c = -(1 + 2⁻²⁶)` make the exact product's `2⁻⁵⁴` term round away (a - /// quarter of the `f64` ulp at this magnitude) before the add, so `a * b + c` gives `0.0` where - /// the fused result is `2⁻⁵⁴`. Lane 1 is its sign-negated twin. Lanes 2 and 3 are ordinary - /// non-dyadic values. + /// For a = b = 1 + 2⁻²⁷ and c = −(1 + 2⁻²⁶), the exact product's 2⁻⁵⁴ term is one quarter of a + /// binary64 ULP and rounds away before a separate addition. Therefore the separate result is + /// zero while the fused result is 2⁻⁵⁴. Lane 1 gives the sign-negated residual. #[test] fn mul_add_f64x4_rounds_once_per_lane() { let factor = 1.0_f64 + (-27.0_f64).exp2(); @@ -227,10 +218,9 @@ mod tests { } } - /// `mul_add_f64x8` rounds once per lane, against scalar [`f64::mul_add`] as the reference. + /// Extends the double-precision cancellation fixture to eight lanes. /// - /// Lanes 0 and 1 repeat the `f64x4` fusion-discriminating pair and its sign-negated twin; the - /// remaining six lanes are ordinary non-dyadic values. + /// Lanes 0 and 1 use the ±2⁻⁵⁴ residuals derived in [`mul_add_f64x4_rounds_once_per_lane`]. #[test] fn mul_add_f64x8_rounds_once_per_lane() { let factor = 1.0_f64 + (-27.0_f64).exp2(); @@ -307,10 +297,8 @@ mod tests { continue; } - // The composed error scales with the result's binary - // exponent; 2e-4 relative covers the extreme corner of the - // sample grid (3.4e37 squared) with margin, and results - // near one land far inside it. + // the tolerance applies to this finite sample grid. Overflowing powers, + // including 3.4e37 squared, took the classification branch above. assert!( (vectorized - reference).abs() <= reference.abs() * 2e-4, "pow({base}, {exponent}): sleef {vectorized} vs libm {reference}", @@ -318,8 +306,7 @@ mod tests { } } - // A zero exponent is exact for any positive base: the exponent - // product is zero and exp2(0) is one. + // A finite logarithm multiplied by zero gives zero, and exp2(0) is exactly one. assert_eq!( pow_f32x4(Simd::splat(7.5), Simd::splat(0.0)).to_array(), [1.0; 4] @@ -348,9 +335,6 @@ mod tests { } } - // Exact special points: the smooth-kNN kernel encodes "at or - // below rho" as an adjusted distance of zero and padding lanes - // as negative infinity, so these must not merely be close. assert_eq!(exp_f32x8(Simd::splat(0.0)).to_array(), [1.0; 8]); assert_eq!( exp_f32x8(Simd::splat(f32::NEG_INFINITY)).to_array(), @@ -358,13 +342,13 @@ mod tests { ); } - /// The distance to the next representable `f64` above `value`. + /// Returns the spacing above the finite magnitude of `value`. fn ulp_f64(value: f64) -> f64 { let bits = value.abs().to_bits(); f64::from_bits(bits + 1) - f64::from_bits(bits) } - /// The distance to the next representable `f32` above `value`. + /// Returns the spacing above the finite magnitude of `value`. fn ulp_f32(value: f32) -> f32 { let bits = value.abs().to_bits(); f32::from_bits(bits + 1) - f32::from_bits(bits) diff --git a/libs/@local/graph/atlas/src/math/kernel/sleef.rs b/libs/@local/graph/atlas/src/math/kernel/sleef.rs index ac0bc34a1ef..54d2427833c 100644 --- a/libs/@local/graph/atlas/src/math/kernel/sleef.rs +++ b/libs/@local/graph/atlas/src/math/kernel/sleef.rs @@ -4,20 +4,13 @@ //! of a portable-SIMD vector without a libm call. Range reduction splits the input into an integer //! power of two and a small residual. A short minimax polynomial approximates the function on the //! residual. Reconstruction then applies the power of two through direct exponent-field arithmetic. -//! Each function documents its own error bound, and each bound comes from the SLEEF accuracy tier -//! its kernel derives from (`u10` is within 1.0 ULP, `u35` within 3.5). +//! The f32 entry points document SLEEF accuracy tiers (`u10` is within 1.0 ULP, `u35` within 3.5). //! -//! # Reproducibility contract +//! # Arithmetic //! -//! Content-hashed fit artifacts require bit-identical results from these kernels on every target -//! the crate builds for. Unconditional fusion and plain lane arithmetic give that guarantee: -//! -//! - Every multiply-accumulate is a fused [`mul_add`](std::simd::StdFloat::mul_add). A fused -//! multiply-add has exactly one correctly rounded result, defined by IEEE 754 independently of -//! how a target lowers it, and both baselines the crate builds for lower it in hardware (aarch64 -//! FMLA, x86-64 the v3 baseline's FMA). Targets differ in speed, never in bits. -//! - Every step is plain `f32`/`f64` lane arithmetic, bit shifts, and lane selects, with one -//! rounding per operation as IEEE 754 requires. No step depends on a target-specific instruction. +//! Every multiply-accumulate uses fused [`mul_add`](std::simd::StdFloat::mul_add) to round once +//! instead of rounding the product separately. Portable-SIMD arithmetic does not guarantee +//! identical NaN payloads across targets. //! //! # Provenance and divergences //! @@ -25,26 +18,22 @@ //! from the `sleef` crate, version 0.3.3 (MIT OR Apache-2.0), a pure-Rust port of the SLEEF //! vector math library (Naoki Shibata and contributors, Boost Software License 1.0): //! . The entry points correspond to upstream's -//! `f32x::exp_u10`, `f32x::exp2_u35`, `f32x::log2_u35`, and `f64x::exp_u10`. This module diverges -//! from upstream in form, never in result bits: +//! `f32x::exp_u10`, `f32x::exp2_u35`, `f32x::log2_u35`, and `f64x::exp_u10`. The evaluation differs +//! in these respects: //! //! - This module fuses every multiply-accumulate unconditionally. Upstream selects fusion per //! target under `cfg!(target_feature = "fma")`, an x86-only cfg string, and rounds twice per step -//! where it is false. This module's ladders round once everywhere, and their result bits are -//! their own contract, verified against libm by the tests below. +//! where it is false. Fused evaluation avoids that extra product rounding. //! - `exp_f64` keeps the coefficient set of upstream's non-FMA branch, evaluated fused. Upstream's -//! FMA branch carries a different degree-10 set, so the fused ladder here matches neither -//! upstream branch bit-for-bit; one coefficient set on every architecture is what keeps content -//! hashes reproducible. +//! FMA branch uses a different degree-10 set. Combining fused evaluation with the non-FMA +//! coefficients matches neither upstream branch bit-for-bit. The coefficient set is the same on +//! every architecture. //! - Nearest-integer rounding uses [`round_ties_even`](std::simd::StdFloat::round_ties_even) //! directly. Upstream predates the portable-SIMD API and computes the same round-half-to-even -//! through an add-subtract trick against `2^23` (`2^52` for `f64`) plus sign restoration. The -//! intrinsic returns the identical value in every rounding regime, including the pass-through -//! above `2^23` where the trick's guard bit runs out. -//! - Lane suppression uses mask selects against zero where upstream masks the raw bits through a -//! sign-extended integer AND; names, the `Poly`/`Sign` trait helpers (flattened to explicit -//! Estrin steps and plain functions), and constant spellings (shortest round-trip literals, -//! `core` constants where the value is exactly a named one) follow this crate's conventions. +//! through an add-subtract trick against 2²³ (2⁵² for `f64`) plus sign restoration. The intrinsic +//! specifies ties-to-even directly, including for inputs already integral at their precision. +//! - Lane suppression selects zero with masks. Polynomial evaluation uses explicit Horner or Estrin +//! steps, and constant spellings preserve the source values. //! //! # Verification //! @@ -57,15 +46,16 @@ use core::{f32, f64, f128, simd::prelude::*}; use std::simd::StdFloat as _; -// ln 2 split into a coarse part whose low mantissa bits are zero and -// the correctly rounded remainder against the next-wider constant. -// Multiplying the coarse part by `nearest` is exact through the -// masked width (2^9 for f32, 2^12 for f64, beyond both exp kernels' -// reduction ranges), so the range reduction `x - nearest · ln 2` -// loses no bits to the subtraction; the remainder repays the split's -// truncation. +// A product is exact when its significand fits the destination precision and its exponent is in +// range. Clearing nine low bits leaves at most 15 significand bits in the f32 coarse part, and +// clearing twelve leaves at most 41 in f64. The reduction integers need at most nine and twelve +// bits respectively. Therefore their coarse products are exact. The low part corrects the +// truncation, but its stored value and the residual FMA still carry rounding error. +/// The low `f32` mantissa bits zeroed in the coarse part of `ln 2`: nine bits. const F32_MASK: u32 = 0x1FF; +/// The coarse part of `ln 2` in `f32`, exact under multiplication by integers below `2⁹`. const LN2_HI_F32: f32 = f32::from_bits(f32::consts::LN_2.to_bits() & !F32_MASK); +/// The correctly rounded `f32` remainder `ln 2 - LN2_HI_F32`, computed in `f64`. #[expect( clippy::cast_possible_truncation, reason = "the cast is the derivation's rounding step: the remainder is correctly rounded into \ @@ -73,40 +63,47 @@ const LN2_HI_F32: f32 = f32::from_bits(f32::consts::LN_2.to_bits() & !F32_MASK); )] const LN2_LO_F32: f32 = (f64::consts::LN_2 - (LN2_HI_F32 as f64)) as f32; +/// The low `f64` mantissa bits zeroed in the coarse part of `ln 2`: twelve bits. const F64_MASK: u64 = 0xFFF; +/// The coarse part of `ln 2` in `f64`, exact under multiplication by integers below `2¹²`. const LN2_HI_F64: f64 = f64::from_bits(f64::consts::LN_2.to_bits() & !F64_MASK); +/// The correctly rounded `f64` remainder `ln 2 - LN2_HI_F64`, computed in `f128`. const LN2_LO_F64: f64 = (f128::consts::LN_2 - (LN2_HI_F64 as f128)) as f64; -/// Raises two to the power in each lane of `exponent`, directly in the result's exponent field. +/// Constructs single-precision power-of-two exponent fields. /// -/// Exact for exponents where the result is a normal `f32`; the callers keep exponents in that range -/// by splitting (see [`scale_by_pow2_f32`]). +/// For integer n in [−126, 127], the result is exactly 2ⁿ. Other exponents need not encode that +/// power. Use [`scale_by_pow2_f32`] for split scaling. #[inline] fn pow2_f32(exponent: Simd) -> Simd { // 0x7F is the f32 exponent bias, and 23 the mantissa width. Simd::from_bits(((exponent + Simd::splat(0x7F)) << Simd::splat(23)).cast()) } -/// Raises two to the power in each lane of `exponent`, as `f64`. +/// Constructs double-precision power-of-two exponent fields. /// -/// The `f64` counterpart of [`pow2_f32`], exact for exponents where the result is a normal `f64`. +/// For integer n in [−1022, 1023], the result is exactly 2ⁿ. Other exponents need not encode that +/// power. #[inline] fn pow2_f64(exponent: Simd) -> Simd { - // 0x3FF is the f64 exponent bias, and the field starts 20 bits into the upper half of the word, - // so the cast widens the biased value into the upper 32 bits before the shift moves it into - // place there. + // 0x3FF is the f64 exponent bias, and the field starts at bit 52, twenty bits into the upper + // 32-bit half. The shifts split that offset as 32 + 20, with widening to i64 before either + // shift. let biased = Simd::splat(0x3FF) + exponent; let upper = biased.cast::() << Simd::splat(32); Simd::from_bits((upper << Simd::splat(20)).cast()) } -/// Scales each lane by two raised to `exponent`, in two half-steps. +/// Scales each lane by an integer power of two in two steps. +/// +/// The split exponents are ⌊n/2⌋ and n − ⌊n/2⌋ for exponent n. When both constructed powers and the +/// first scaled value are normal, the first multiplication is exact. The second multiplication +/// supplies the final rounding if the result is subnormal or overflows. Splitting permits +/// reconstruction even when a single scale factor would already be zero or infinite. /// -/// Applying `2^(exponent/2)` twice keeps each factor a normal number for the exponent range the -/// reconstruction step produces, where a single factor could overflow or flush to zero before the -/// scaled value lands back in range. Both multiplies are by powers of two with normal intermediate -/// results, so the scaling is exact except for the single rounding when the final result is -/// subnormal. +/// Outside those conditions, this helper has no general power-of-two scaling guarantee. The +/// transcendental entry points also evaluate non-finite and out-of-range lanes here. Their final +/// masks and NaN arithmetic determine the results for those lanes. #[inline] pub(super) fn scale_by_pow2_f32( values: Simd, @@ -125,10 +122,11 @@ fn scale_by_pow2_f64(values: Simd, exponent: Simd( values: Simd, @@ -137,31 +135,30 @@ fn scale_by_pow2_direct_f32( Simd::from_bits((values.to_bits().cast() + (exponent << Simd::splat(23))).cast()) } -/// The unbiased binary exponent of each lane, read from the exponent field. +/// Extracts each exponent field and subtracts the single-precision bias. /// -/// For a normal lane this is `floor(log2(|lane|))`. The caller scales subnormal lanes into the -/// normal range first. +/// For a normal lane x, this equals ⌊log₂|x|⌋. A zero exponent field yields −127, and an all-ones +/// field yields 128. The helper does not validate normality. #[inline] fn binary_exponent_f32(values: Simd) -> Simd { let field = (values.to_bits().cast::() >> Simd::splat(23)) & Simd::splat(0xFF); field - Simd::splat(0x7F) } -/// Base-e exponential of each lane, accurate to the u10 tier (1.0 ULP). +/// Approximates the base-e exponential with the u10 accuracy target. /// -/// Measured faithfully rounded: 0.988 ULP maximum over an exhaustive sweep of the domain `|x| ≤ -/// 110` (2.24e9 inputs), zero misclassified specials, monotone across the reduction boundaries. +/// The recorded exhaustive sweep of `|x| ≤ 110` (2.24e9 inputs) measured a maximum error of 0.988 +/// ULP against `f64` libm. This measurement uses a wider reference whose exactness it does not +/// establish. #[inline] pub(crate) fn exp_f32(values: Simd) -> Simd { - // Range reduction: with n = round(x / ln 2), exp(x) = 2^n · exp(r) - // for r = x - n · ln 2, accumulated in two exact steps against the - // split constants. + // choose n near x / ln 2 and approximate r = x − n · ln 2. The identity + // exp(x) = 2ⁿ · exp(r) separates reconstruction from the small-residual approximation. + // The two FMAs subtract the split constant without separately rounding their products. let nearest = (values * Simd::splat(core::f32::consts::LOG2_E)).round_ties_even(); let exponent = nearest.cast::(); - // `nearest` is integral, and every lane the backstops leave alive - // holds it within ±152, where it equals `exponent` exactly; the - // reduction uses it directly instead of round-tripping the integer - // back to float. + // for finite lanes in [−104, 100], nearest is integral within [−150, 144] and equals exponent + // exactly. Use it directly rather than converting the integer back to float. let reduced = nearest.mul_add(-Simd::splat(LN2_HI_F32), values); let reduced = nearest.mul_add(-Simd::splat(LN2_LO_F32), reduced); @@ -179,11 +176,10 @@ pub(crate) fn exp_f32(values: Simd) -> Simd { let result = scale_by_pow2_f32(poly, exponent); - // Backstops only: the natural path rounds correctly through the - // overflow boundary (ln(f32::MAX) ≈ 88.72) and the underflow-to- - // zero boundary (≈ -103.97); the clamps guard the region beyond, - // where the saturating cast and the exponent-field scaling break - // down. + // reconstruction handles the neighbourhoods of overflow (x ≈ 88.72) and rounding to zero (x ≈ + // −103.97). These masks enforce the more distant results and discard out-of-range + // exponent-field calculations. Within [−104, 100], both split power-of-two factors remain + // normal. let result = values .simd_lt(Simd::splat(-104.)) .select(Simd::splat(0.), result); @@ -192,23 +188,22 @@ pub(crate) fn exp_f32(values: Simd) -> Simd { .select(Simd::splat(f32::INFINITY), result) } -/// Base-2 exponential of each lane, accurate to the u35 tier (3.5 ULP). +/// Approximates the base-2 exponential with the u35 accuracy target. /// -/// Measured far inside the tier, faithfully rounded: 0.885 ULP maximum over an exhaustive sweep of -/// the domain `|x| ≤ 160` (2.25e9 inputs). The reduction `x - round(x)` is exact, so the polynomial -/// fit dominates the error budget. +/// The recorded exhaustive sweep of `|x| ≤ 160` (2.25e9 inputs) measured a maximum error of +/// 0.885 ULP against `f64` libm. In this range the reduction `x - round(x)` is exact. Polynomial +/// evaluation and reconstruction still round. #[inline] pub(crate) fn exp2_f32(values: Simd) -> Simd { - // Range reduction is exact: 2^x = 2^n · 2^f for n = round(x) and - // f = x - n, |f| ≤ 1/2. + // for finite lanes in [−150, 128), n = round(x) and f = x − n give an exact subtraction with + // |f| ≤ 1/2. The identity is 2ˣ = 2ⁿ · 2ᶠ. let nearest = values.round_ties_even(); let exponent = nearest.cast::(); let fraction = values - nearest; - // Degree-6 minimax polynomial for 2^f, in Horner form: the Taylor - // coefficients are ln(2)^k / k! (ln(2)^6/6! = 1.536e-4 down to - // ln(2)^2/2! = 0.240), minimax-nudged in the low digits; the last - // two steps add the exact k = 1 and k = 0 terms, ln(2) · f and 1. + // degree-6 minimax polynomial for 2ᶠ in Horner form. Its coefficients approximate ln(2)ᵏ/k!, + // from ln(2)⁶/6! ≈ 1.540e−4 to ln(2)²/2! ≈ 0.240, adjusted in the low digits. The final steps + // add the linear term using rounded ln(2), then the constant 1. let poly = Simd::splat(0.000_153_592_09) .mul_add(fraction, Simd::splat(0.001_339_262_7)) .mul_add(fraction, Simd::splat(0.009_618_385)) @@ -227,36 +222,38 @@ pub(crate) fn exp2_f32(values: Simd) -> Simd { .select(Simd::splat(0.), result) } -/// Base-2 logarithm of each lane, accurate to the u35 tier (3.5 ULP). +/// Approximates the base-2 logarithm with the u35 accuracy target. /// -/// Measured 3.07 ULP maximum over an exhaustive sweep of all finite positive inputs, with every -/// case above 2 ULP inside `[0.5, 1.5)`, where the result cancels toward zero; outside that band -/// the maximum is 1.6 ULP. The dominant error terms are the roundings of `m + 1` and of the -/// division, amplified when `|result|` is small; sub-ULP accuracy would need a double-float -/// ratio, not a better polynomial. +/// The recorded exhaustive sweep of finite positive inputs measured 3.07 ULP maximum against +/// `f64` libm. Every measured case above 2 ULP was inside `[0.5, 1.5)`, where the logarithm +/// approaches zero. Outside that interval the measured maximum was 1.6 ULP. Rounding in the +/// reduced ratio contributes error before polynomial evaluation begins. +/// +/// Either zero yields negative infinity. Negative inputs and NaN yield NaN, and positive infinity +/// yields positive infinity. #[inline] pub(crate) fn log2_f32(values: Simd) -> Simd { - // The select below multiplies subnormal lanes by 2^64, which brings them into the normal range - // so the exponent-field read is exact, and the exponent subtraction afterwards repays the - // factor. Zero, negative, and NaN lanes compute whatever the arithmetic yields, and the - // selects at the end overwrite them. + // multiplication by 2⁶⁴ brings finite subnormal lanes into the normal range exactly. + // Subtracting 64 from the recovered exponent compensates for that scale. Final masks supply the + // zero, negative, infinite and NaN results. let is_subnormal = values.is_subnormal(); let scaled = is_subnormal.select(values * Simd::splat(1.844_674_4e19), values); - // The 1/0.75 bias centers the mantissa split on [0.75, 1.5), so - // the ratio below stays small and symmetric around zero. + // the 1/0.75 bias targets a mantissa near [0.75, 1.5), keeping the ratio below near zero. For + // sufficiently large finite inputs the biased product overflows. Its all-ones exponent field + // yields 128, the exponent needed to scale those inputs into this interval. let exponent = binary_exponent_f32(scaled * Simd::splat(1. / 0.75)); let mantissa = scale_by_pow2_direct_f32(scaled, -exponent); let exponent = is_subnormal.select(exponent - Simd::splat(64), exponent); - // With r = (m-1)/(m+1), the atanh identity gives ln(m) = 2 atanh(r) = 2 (r + r^3/3 + r^5/5 + - // ...), so log2(m) is a series in odd powers of r. + // with r = (m − 1)/(m + 1), the identity ln(m) = 2 atanh(r) = 2(r + r³/3 + r⁵/5 + ...) + // expresses log₂(m) as a series in odd powers of r. let ratio = (mantissa - Simd::splat(1.)) / (mantissa + Simd::splat(1.)); let ratio_squared = ratio * ratio; - // The r^3, r^5, and r^7 coefficients, minimax-nudged from the - // series' 2/(k ln 2); the final mul_add below adds the exact - // leading term 2/ln(2) · r and the integer exponent. + // the r³, r⁵ and r⁷ coefficients approximate the series terms 2/(k · ln 2). + // The final FMA combines the polynomial correction with the leading term, whose + // coefficient 2/ln(2) is rounded to f32, and the integer exponent. let poly = Simd::splat(0.437_408_83) .mul_add(ratio_squared, Simd::splat(0.576_484_4)) .mul_add(ratio_squared, Simd::splat(0.961_802_4)); @@ -277,17 +274,15 @@ pub(crate) fn log2_f32(values: Simd) -> Simd { .select(Simd::splat(f32::NEG_INFINITY), result) } -/// Base-e exponential of each lane, accurate to the u10 tier (1.0 ULP). +/// Approximates the base-e exponential in each lane. /// -/// Measured faithfully rounded (1.0 ULP maximum) over 242e6 samples including every double -/// adjacent to a reduction boundary `k · ln(2) / 2`, the overflow window around `ln(f64::MAX)`, -/// and the subnormal-output region, where the two-step power-of-two scaling keeps the error at -/// one rounding. +/// Reconstruction applies its power-of-two scale in two steps to avoid overflowing or underflowing +/// the scale factor before multiplication. #[inline] pub(crate) fn exp_f64(values: Simd) -> Simd { - // Range reduction: with n = round(x / ln 2), exp(x) = 2^n · exp(r) - // for r = x - n · ln 2, accumulated in two exact steps against the - // split constants. + // use exp(x) = 2ⁿ · exp(r), with n near x / ln 2 and r ≈ x − n · ln 2. + // The coarse product is exact in the unclamped range. The low-part FMA corrects the + // split's truncation and rounds the residual once. let nearest = (values * Simd::splat(core::f64::consts::LOG2_E)).round_ties_even(); let exponent = nearest.cast::(); let reduced = nearest.mul_add(-Simd::splat(LN2_HI_F64), values); @@ -333,14 +328,11 @@ pub(crate) fn exp_f64(values: Simd) -> Simd { let result = scale_by_pow2_f64(poly, exponent); - // Backstops only: the natural path rounds correctly through the - // overflow boundary (ln(f64::MAX) ≈ 709.7827) and far past the - // underflow-to-zero boundary (≈ -745.13); the clamps guard the - // region beyond, where the saturating cast and the exponent-field - // scaling break down (near |x| = 1421 the biased half-exponent - // leaves the normal range). Any upper constant ∈ [710, 1421) is - // correct, and one at or below ln(f64::MAX) misclassifies the finite doubles immediately under - // the boundary as infinite. + // reconstruction handles the neighbourhoods of overflow (x ≈ 709.7827) and rounding to zero (x + // ≈ −745.13). The masks enforce more distant results. Finite lanes retained in [−1000, 710] + // have reduction integers in [−1443, 1024], whose two half-exponents remain in the normal + // power-of-two range. The upper mask leaves finite results near the overflow boundary to + // reconstruction. let result = values .simd_gt(Simd::splat(710.)) .select(Simd::splat(f64::INFINITY), result); @@ -360,26 +352,31 @@ mod tests { use super::{exp_f32, exp_f64, exp2_f32, log2_f32}; - // The strides are odd, so consecutive samples differ in exponent/mantissa phase. Full-bit-range - // iteration covers negative inputs, subnormals, both zeros, both infinities, and NaN payloads - // without listing them. + // odd strides sample different exponent and significand bit patterns across both signs. + // They do not visit every special encoding. Separate tests supply selected boundary cases. + /// The sampling stride through the `u32` bit space. const F32_STRIDE: usize = 641; + /// Odd stride through the `u64` bit space with the same coverage for `f64` inputs. const F64_STRIDE: usize = 0x0400_0000_000D; - // Each bound is the kernel's accuracy tier plus half a - // representation step for the reference's own correctly-rounded - // narrowing, rounded up to whole steps: 1.0 + 0.5 -> 2 and - // 3.5 + 0.5 -> 4. A result past the bound is a behavior change, - // not measurement noise: a wrong constant or a swapped - // coefficient moves results by orders of magnitude. + /// Representation-step budget for the `u10` single-precision sample test. + /// + /// The budget is ⌈1.0 + 0.5⌉ = 2, adding a nominal half-step narrowing allowance to the tier. + /// The wider libm reference is itself approximate. This comparison budget is not a certified + /// error bound against the exact function. const U10_F32_TOLERANCE: u64 = 2; + /// Representation-step budget for the `u35` single-precision sample tests. + /// + /// The budget is ⌈3.5 + 0.5⌉ = 4, using the same reference model as [`U10_F32_TOLERANCE`]. const U35_F32_TOLERANCE: u64 = 4; + /// Allowed representation-step distance from the scalar f64 libm reference. const U10_F64_TOLERANCE: u128 = 2; - /// Position of a value in the ordered sequence of representable `f32`s. + /// Maps a single-precision encoding to a signed representation-step index. /// - /// Adjacent representable values differ by one across the whole line, including zeros, - /// subnormals, and infinities, so one distance bound holds without per-class cases. + /// Adjacent distinct non-NaN values differ by one, including the subnormal and infinity + /// boundaries. Both zeros map to zero. NaN encodings have indices beyond the corresponding + /// infinity, without numerical-distance semantics. fn ordered_f32(value: f32) -> i64 { let bits = value.to_bits(); if bits & 0x8000_0000 == 0 { @@ -389,7 +386,7 @@ mod tests { } } - /// Position of a value in the ordered sequence of representable `f64`s. + /// Maps a double-precision encoding to a signed representation-step index. fn ordered_f64(value: f64) -> i128 { let bits = value.to_bits(); if bits & 0x8000_0000_0000_0000 == 0 { @@ -401,7 +398,12 @@ mod tests { /// Checks one `f32` lane against its reference. /// - /// NaN must map to NaN. Every other pair must lie within `tolerance` representation steps. + /// A NaN reference requires a NaN output. Every other pair is compared by encoding distance, + /// including across the finite/infinite and infinite/NaN boundaries. + /// + /// # Panics + /// + /// Panics if a NaN reference has a non-NaN output or the encoding distance exceeds `tolerance`. #[track_caller] fn assert_lane_f32(name: &str, at: f32, kernel: f32, reference: f32, tolerance: u64) { if reference.is_nan() { @@ -495,9 +497,9 @@ mod tests { /// Overflow classification at the `f32` boundary matches libm exactly. /// - /// The ordered-step tolerance forgives an infinity one step from `MAX`, so the sweeps above - /// cannot see a misclassified overflow boundary; this scan pins the class over every - /// representable input around `ln(f32::MAX)`. + /// The ordered-step tolerance cannot distinguish infinity from an adjacent `MAX`. This scan + /// compares the class over every representable input around `ln(f32::MAX)`, independently of + /// that distance tolerance. #[test] fn exp_f32_overflow_boundary_is_class_exact() { let mut bits = 88.5_f32.to_bits(); @@ -524,10 +526,9 @@ mod tests { /// Overflow classification at the `f64` boundary matches libm exactly. /// - /// This window is doubly invisible to the strided sweep: the stride jumps over it, and the - /// ordered-step tolerance would forgive an infinity one step from `MAX` anyway. The scan - /// covers every double from below the retired conservative threshold through `ln(f64::MAX)` - /// and asserts both the class and the u10 distance. + /// The strided sweep can skip the overflow transition, and its representation-step tolerance + /// permits infinity beside `MAX`. This scan checks classification and distance at every + /// representable input in `[709.782711, 709.782713]`, which straddles that transition. #[test] fn exp_f64_overflow_boundary_is_class_exact() { let mut bits = 709.782_711_f64.to_bits(); diff --git a/libs/@local/graph/atlas/src/math/kernel/ulp_sweep.rs b/libs/@local/graph/atlas/src/math/kernel/ulp_sweep.rs index 2d27cba2630..84f9b67a711 100644 --- a/libs/@local/graph/atlas/src/math/kernel/ulp_sweep.rs +++ b/libs/@local/graph/atlas/src/math/kernel/ulp_sweep.rs @@ -1,16 +1,17 @@ -//! Exhaustive ULP verification for the vendored transcendental kernels. +//! Exhaustive finite-input error measurements for the single-precision kernels. //! -//! Each sweep compares a kernel against an `f64` libm reference on every `f32` bit pattern the -//! backstop clamps do not decide, threaded across all cores. Every test here carries `#[ignore]`, -//! so run them in release mode, where the full set takes tens of seconds: +//! Each sweep compares every `f32` bit pattern in its stated intervals against an `f64` libm +//! reference, distributed across worker threads. These intervals contain billions of inputs. +//! Every test here carries `#[ignore]`. Run them in release mode: //! //! ```text //! cargo test --release -p hash-graph-atlas --features bench ulp_sweep -- --ignored --nocapture //! ``` //! -//! An `f64` reference carries far more precision than one `f32` ULP, so it measures the `f32` -//! kernels exactly. It cannot certify `exp_f64` to sub-ULP. [`sleef`](super::sleef)'s tests pin -//! that kernel's overflow classes exhaustively, and the strided sweep there bounds its distance. +//! The measurements use a wider scalar libm reference, which resolves much smaller differences than +//! one `f32` ULP but is itself approximate, and do not certify correct or faithful rounding against +//! the exact real function. [`sleef`](super::sleef)'s separate `f64` tests sample +//! representation-step distances and scan a narrow interval around its overflow transition. #![expect( clippy::cast_possible_truncation, clippy::cast_precision_loss, @@ -27,6 +28,7 @@ use super::{ sleef::{exp_f32, exp2_f32, log2_f32}, }; +/// Lanes per kernel call in a sweep. const LANES: usize = 8; /// Error statistics accumulated over one sweep. @@ -42,6 +44,7 @@ struct Accumulator { } impl Accumulator { + /// Folds another thread's statistics into this one, keeping the larger worst case. fn merge(&mut self, other: &Self) { if other.max_ulp > self.max_ulp { self.max_ulp = other.max_ulp; @@ -54,6 +57,11 @@ impl Accumulator { self.misclassified += other.misclassified; } + /// Scores one argument: `kernel` against the `f64` `reference`, in ULPs at the reference. + /// + /// NaN references require a NaN result. References that round to infinity require that same + /// infinity, and an infinite kernel result against a finite rounded reference counts as a + /// misclassification. Finite pairs contribute their distance in ULPs at the reference. fn record(&mut self, bits: u32, kernel: f32, reference: f64) { self.total += 1; if reference.is_nan() { @@ -80,6 +88,7 @@ impl Accumulator { } } + /// Prints the sweep's statistics under `name`. fn report(&self, name: &str) { println!( "{name}: n={} max={:.4} ulp at x={:?} ({:#010x}) not-correctly-rounded={} ({:.3}%) \ @@ -100,7 +109,7 @@ impl Accumulator { /// Spacing of `f32` at the magnitude of `reference`. /// /// For `|reference| = m · 2^e` with `m ∈ [1, 2)`, one ULP is `2^(e - 23)`, clamped to the subnormal -/// spacing `2^-149` below the normal range. +/// spacing `2⁻¹⁴⁹` below the normal range. fn ulp32_at(reference: f64) -> f64 { let magnitude = reference.abs(); if magnitude < f64::from(f32::MIN_POSITIVE) { @@ -110,8 +119,13 @@ fn ulp32_at(reference: f64) -> f64 { 2_f64.powi(exponent - 23) } -/// Sweeps every bit pattern in each inclusive range through `kernel`, lane-wise against -/// `reference`, spread across all cores. +/// Measures each bit pattern in the inclusive ranges against a scalar reference. +/// +/// Each range must have its lower endpoint at or below its upper endpoint. +/// +/// # Panics +/// +/// Panics if a worker running `kernel` or `reference` panics, or if a worker cannot be spawned. fn sweep( ranges: &[(u32, u32)], kernel: fn(Simd) -> Simd, diff --git a/libs/@local/graph/atlas/src/math/matrixn/mod.rs b/libs/@local/graph/atlas/src/math/matrixn/mod.rs index 752285ea463..17b6835b4d6 100644 --- a/libs/@local/graph/atlas/src/math/matrixn/mod.rs +++ b/libs/@local/graph/atlas/src/math/matrixn/mod.rs @@ -1,4 +1,4 @@ -//! An owned row-major matrix with SIMD-aligned rows. +//! Row-major matrices whose rows support aligned SIMD access. use alloc::alloc::Global; use core::{ @@ -15,20 +15,30 @@ use super::AlignedVecN; #[cfg(test)] mod tests; -/// An owned `T x N` matrix of `f32` components in one heap allocation aligned for [`f32x8`]. +/// An owned row-major `f32` matrix with SIMD-aligned rows. /// -/// The row width `N` is a nonzero multiple of 8, so one row's `N · 4` bytes are a multiple of -/// `align_of::()` and every row begins at an alignment boundary. [`rows`](Self::rows) views -/// the matrix as [`AlignedVecN`] rows that satisfy the alignment invariant by construction, and -/// [`BoxedVecN`](super::BoxedVecN) gives one vector the same guarantee. A width that is not a -/// multiple of 8 fails to compile. +/// The row width `N` is a nonzero multiple of 8. Each row occupies a whole number of aligned +/// [`f32x8`] groups in the allocation, preserving alignment at every row start. +/// [`rows`](Self::rows) exposes these as [`AlignedVecN`] views. Constructing a matrix with zero +/// width or a width not divisible by 8 fails to compile. [`BoxedVecN`](super::BoxedVecN) provides +/// owned storage for one vector. /// /// The caller picks the row count at runtime. [`zeroed`](Self::zeroed) is the constructor, and rows -/// fill in place through [`rows_mut`](Self::rows_mut). +/// fill in place through [`rows_mut`](Self::rows_mut). Indexing selects a row and panics when the +/// index is at least the row count. Cloning copies the complete buffer into a separate allocation +/// using a clone of the retained allocator. /// -/// # Examples +/// Allocation failure during construction or cloning is handled by +/// [`handle_alloc_error`](alloc::alloc::handle_alloc_error). These operations panic if the required +/// layout cannot be represented. +/// +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::matrixn::MatrixN; +/// /// let mut matrix = MatrixN::<32>::zeroed(2); /// matrix.rows_mut()[1].as_array_mut()[0] = 1.0; /// @@ -43,24 +53,28 @@ pub(crate) struct MatrixN { } impl MatrixN { - /// Creates the zero matrix of `rows` rows in a new aligned allocation in the global allocator. + /// Creates a zero-filled matrix in the global allocator. + /// + /// Every component is `0.0`. Allocation failure is handled by + /// [`handle_alloc_error`](alloc::alloc::handle_alloc_error). + /// + /// # Panics /// - /// Every component is `0.0` and the buffer is valid for in-place filling through - /// [`rows_mut`](Self::rows_mut). + /// Panics if the matrix layout cannot be represented. See [`Self::zeroed_in`]. #[inline] #[must_use] pub(crate) fn zeroed(rows: usize) -> Self { Self::zeroed_in(rows, Global) } - /// Creates the matrix whose rows copy the iterator's, in order, in the global allocator. + /// Copies rows in iterator order into a matrix in the global allocator. + /// + /// Allocation failure is handled by [`handle_alloc_error`](alloc::alloc::handle_alloc_error). /// /// # Panics /// - /// This panics when the iterator yields fewer rows than its exact length declares, so a short - /// iterator cannot leave silently zeroed rows behind, and when the declared length's matrix - /// layout cannot fit `isize`, since the matrix is allocated from that declared length before - /// any row is consumed. + /// Panics if the declared row count's layout cannot be represented, or if the iterator yields + /// fewer rows than it declares. See [`Self::from_rows_in`]. #[inline] #[must_use] pub(crate) fn from_rows<'row>( @@ -71,14 +85,15 @@ impl MatrixN { } impl MatrixN { - /// Create the layout of the allocation. + /// Computes the row-major component layout with SIMD alignment. /// - /// The allocation layout: `rows · N` components, padded to the alignment of [`f32x8`]. - /// Allocation and deallocation must agree on this. + /// Raising alignment preserves the byte size of the `rows · N` components. Allocation and + /// deallocation must use this same layout. /// /// # Panics /// - /// This panics when the component count overflows the address space. + /// Panics if the component count overflows [`usize`] or if the aligned layout exceeds the + /// layout size limit. #[inline] fn layout(rows: usize) -> Layout { const { @@ -97,10 +112,13 @@ impl MatrixN { ) } - /// Creates the zero matrix of `rows` rows in a new aligned allocation in `alloc`. + /// Creates a zero-filled matrix with `rows` rows in `alloc`. + /// + /// Allocation failure is handled by [`handle_alloc_error`](alloc::alloc::handle_alloc_error). /// - /// This aborts the process through [`handle_alloc_error`](std::alloc::handle_alloc_error) when - /// the allocator cannot provide the buffer. + /// # Panics + /// + /// Panics if the component count or aligned allocation layout cannot be represented. #[inline] #[must_use] pub(crate) fn zeroed_in(rows: usize, alloc: A) -> Self { @@ -117,14 +135,15 @@ impl MatrixN { } } - /// Creates the matrix whose rows copy the iterator's, in order, in `alloc`. + /// Copies rows in iterator order into a matrix in `alloc`. + /// + /// The iterator's declared length sets the allocation size before copying. Allocation failure + /// is handled by [`handle_alloc_error`](alloc::alloc::handle_alloc_error). /// /// # Panics /// - /// This panics when the iterator yields fewer rows than its exact length declares, so a short - /// iterator cannot leave silently zeroed rows behind, and when the declared length's matrix - /// layout cannot fit `isize`, since the matrix is allocated from that declared length before - /// any row is consumed. + /// Panics if the declared row count's layout cannot be represented, or if the iterator yields + /// fewer rows than it declares. #[inline] #[must_use] pub(crate) fn from_rows_in<'row>( @@ -163,8 +182,11 @@ impl MatrixN { #[inline] #[must_use] pub(crate) const fn as_components(&self) -> &[f32] { - // SAFETY: `ptr` owns an initialized buffer of `rows · N` components for as long as `self` - // lives. + // SAFETY: from_raw_parts requires one initialized, aligned allocation valid for the + // borrowed range. layout checked rows * N and its byte size, and zeroed_in initialized and + // retained that buffer, including an aligned non-null pointer for zero rows. The dimensions + // never change and the shared borrow prevents deallocation or mutation. Therefore the slice + // is valid for this borrow of self. unsafe { slice::from_raw_parts(self.ptr.as_ptr(), self.rows * N) } } @@ -172,8 +194,11 @@ impl MatrixN { #[inline] #[must_use] pub(crate) const fn as_components_mut(&mut self) -> &mut [f32] { - // SAFETY: `ptr` owns an initialized buffer of `rows · N` components for as long as `self` - // lives. The exclusive borrow of `self` guards the exclusive reference. + // SAFETY: from_raw_parts_mut requires one initialized, aligned allocation exclusively + // accessible for the borrowed range. layout checked rows * N and its byte size, and + // zeroed_in initialized and retained that buffer, including an aligned non-null pointer for + // zero rows. The dimensions never change and the exclusive borrow prevents other access. + // Therefore the mutable slice is valid for this borrow of self. unsafe { slice::from_raw_parts_mut(self.ptr.as_ptr(), self.rows * N) } } @@ -195,10 +220,9 @@ impl MatrixN { /// Views the matrix as one flat slice of aligned 8-lane groups. /// - /// The lanes run row-major over the whole storage. Row `i` occupies the `N / 8` consecutive - /// lanes from `i · N / 8`, and no lane straddles two rows, so whole-matrix elementwise kernels - /// iterate one slice without per-row dispatch. No scalar remainder exists, because the row - /// width is a multiple of the lane width by construction. + /// Row `i` occupies the `N / 8` consecutive groups from `i · N / 8`. Whole-group row widths + /// leave no group straddling two rows and no scalar remainder. The flat view supports + /// whole-matrix elementwise operations without per-row dispatch. #[inline] #[must_use] pub(crate) fn lanes(&self) -> &[f32x8] { @@ -213,8 +237,8 @@ impl MatrixN { /// Views the matrix as one flat slice of aligned 8-lane groups, mutably. /// - /// The split is the same as [`lanes`](Self::lanes); writes through the slice update the matrix - /// in place. + /// The grouping is the same as [`lanes`](Self::lanes). Writes through the slice update the + /// matrix in place. #[inline] #[must_use] pub(crate) fn lanes_mut(&mut self) -> &mut [f32x8] { @@ -232,8 +256,10 @@ impl Clone for MatrixN { fn clone(&self) -> Self { let clone = Self::zeroed_in(self.rows, self.alloc.clone()); - // SAFETY: both pointers own initialized buffers of `rows · N` components, and a fresh - // allocation cannot overlap its source. + // SAFETY: copy_nonoverlapping requires readable source components, a writable destination + // and no overlap for a nonzero copy. Both layouts cover the same checked component count, + // and zeroed_in supplies a separate allocation. Empty buffers still have non-null aligned + // pointers. Therefore the copy preserves the source and initializes the independent clone. unsafe { ptr::copy_nonoverlapping(self.ptr.as_ptr(), clone.ptr.as_ptr(), self.rows * N); } @@ -271,8 +297,10 @@ impl fmt::Debug for MatrixN { impl Drop for MatrixN { #[inline] fn drop(&mut self) { - // SAFETY: `zeroed_in` allocated `ptr` from `alloc` with the same layout, and nothing has - // deallocated it since. + // SAFETY: deallocate requires a currently allocated pointer and a matching allocator + // layout. zeroed_in retains the allocator and its buffer, and no operation transfers or + // releases that buffer. The dimensions remain unchanged. Therefore Drop releases it exactly + // once through the original allocator and layout. unsafe { self.alloc .deallocate(self.ptr.cast::(), Self::layout(self.rows)); @@ -280,10 +308,13 @@ impl Drop for MatrixN { } } -// SAFETY: the matrix owns its buffer exclusively; sending it moves the unique owner, exactly as -// `Box<[f32]>` is `Send`. +// SAFETY: Send permits transferring ownership between threads. The matrix exclusively owns its f32 +// buffer, whose components are Send, and A: Send permits moving the retained allocator. Therefore +// the buffer and its eventual deallocation can transfer with the matrix. unsafe impl Send for MatrixN {} -// SAFETY: shared access hands out only `&[f32]`-shaped views of the owned buffer; there is no -// interior mutability, exactly as `Box<[f32]>` is `Sync`. +// SAFETY: Sync requires shared access to avoid unsynchronized mutation. Shared matrix methods +// expose immutable f32 components, and A: Sync permits shared allocator access. Buffer mutation and +// deallocation require exclusive ownership. Therefore sharing the matrix introduces no mutable +// buffer aliases. unsafe impl Sync for MatrixN {} diff --git a/libs/@local/graph/atlas/src/math/matrixn/tests.rs b/libs/@local/graph/atlas/src/math/matrixn/tests.rs index a23e842fdd9..cd87aad289a 100644 --- a/libs/@local/graph/atlas/src/math/matrixn/tests.rs +++ b/libs/@local/graph/atlas/src/math/matrixn/tests.rs @@ -1,8 +1,3 @@ -/// The tests the `miri` nextest profile selects. -/// -/// Each test here drives the row-major matrix buffer: row alignment and offsets, clone -/// independence, and the return of the buffer to its allocator. The profile selects by module path, -/// so moving a test in or out of this module is the whole edit. mod miri { use core::simd::f32x8; @@ -75,7 +70,6 @@ mod miri { assert_eq!(matrix, matrix.clone()); } - /// `Debug` prints the rows. #[test] fn debug_prints_the_rows() { let mut matrix = MatrixN::<8>::zeroed(1); @@ -87,7 +81,6 @@ mod miri { ); } - /// Dropping a matrix returns its buffer to the allocator that provided it. #[test] fn drop_returns_the_buffer_to_its_allocator() { let alloc = CountingAllocator::new(); diff --git a/libs/@local/graph/atlas/src/math/mod.rs b/libs/@local/graph/atlas/src/math/mod.rs index e2db347425d..21170bbe119 100644 --- a/libs/@local/graph/atlas/src/math/mod.rs +++ b/libs/@local/graph/atlas/src/math/mod.rs @@ -1,15 +1,18 @@ -//! SIMD-native math primitives for fitting and serving 2D maps of embeddings. +//! Geometry and numerical kernels for fitting and querying 2D maps of embeddings. //! -//! Everything here serves one pipeline that has to be fast and correct. The pipeline places -//! high-dimensional embedding vectors on a 2D map and goes on transforming, aligning, and verifying -//! that map. The types are `f32` throughout, batch four-wide where hot loops iterate, and every -//! performance claim in their docs traces to emitted assembly or a hardware-counter measurement. +//! The module provides vector arithmetic, coordinate transformations and fitting operations. +//! Single-precision storage keeps point and embedding arrays compact. Double-precision arithmetic +//! supplies wider accumulation and range where individual kernels need it. Validated fields and +//! scalar domains make input conditions explicit. //! -//! The module is crate-internal. Its examples carry `ignore` and spell each call as an in-crate -//! caller writes it. +//! # Example +//! +//! This in-crate example is ignored because the math API is crate-private. It maps a layout's +//! extent onto a viewport. //! //! ```ignore -//! // Gather a layout's extent and map its points onto a viewport. +//! use crate::math::{Bounds2, Vec2}; +//! //! let points = [ //! Vec2::new(-2.0, 0.0), //! Vec2::new(6.0, 4.0), @@ -22,61 +25,65 @@ //! assert_eq!(mapped[2], Vec2::new(5.0, 5.0)); //! ``` //! -//! # The types, by role -//! -//! 2D geometry: [`Vec2`] is the scalar point/vector. [`Vec2x4`](vec2::Vec2x4) (natural order) -//! and -//! [`Vec2x4T`] (transposed order) batch four of them for SIMD, staging and computing respectively. -//! [`Bounds2`] is the validated bounding box, with serial, SIMD, and parallel construction. -//! -//! Transforms, most constrained first: [`Rotation`] (angle only, exact inverse), -//! [`Translation`](translation::Translation) (offset only, exact inverse), [`Similarity`] -//! (uniform scale + rotation + translation, total inverse, fitted from weighted point -//! correspondences), [`Transform`] (general affine, fallible inverse). -//! Prefer the most constrained type that models the job; each widens into -//! [`Transform`] via [`From`], and composition is always `a.then(b)`, -//! reading in application order. -//! -//! Embeddings: [`VecN`] is the `N`-dimensional `f32` vector with the distance kernels; -//! [`BoxedVecN`] owns SIMD-aligned heap storage and hands out [`AlignedVecN`] references. [`DVecN`] -//! is the double-precision twin for the few consumers whose algorithms need it. -//! -//! Dense solves: [`DSquareMatrix`] is the runtime-order square `f64` matrix; -//! [`DSquareMatrix::cholesky`] factors it deterministically into the -//! [`DCholeskyFactor`](dsquare::DCholeskyFactor) that answers symmetric positive-definite -//! linear systems. -//! -//! Exact neighbours: [`KdTree`] indexes a placed 2D frame and answers exact k-nearest-neighbour -//! readouts equal to a full scan, with `f64` squared-distance readings and ties resolved by row. -//! -//! Layout fitting: [`AffinityCurve`] evaluates the affinity curve of UMAP-style layouts and its -//! attraction/repulsion gradients over batches. Its parameters come from [`AffinityCurve::fit`]. -//! -//! Scalar helpers: [`softplus`] and the checked narrowing [`narrow_f32`]. The Huber penalty and -//! the logistic function live on [`NonNegative`] as [`huber`](NonNegative::huber) and -//! [`sigmoid`](NonNegative::sigmoid). -//! -//! Unclaimed folds: [`Derivation`] is a data-dependent fold's raw value bound for its -//! validated [`Domain`](derivation::Domain), claiming nothing until -//! [`finish`](Derivation::finish). [`Diverged`] returns the raw evidence -//! of a refused claim. -//! -//! # Precision policy -//! -//! `f32` is the working precision: coordinates, transforms, gradients, and distances take and -//! return `f32`. Long reductions accumulate in `f64` internally and round once at the end, which -//! the kernel docs state as an accuracy guarantee rather than exposing in signatures. A signature -//! takes `f64` only where a consumer's algorithm demands it, such as classifier logits on -//! [`DVecN`]. +//! # Types by role +//! +//! Geometry: [`Vec2`] is a single-precision point or vector. [`Vec2x4`](vec2::Vec2x4) keeps four +//! points in natural order, while [`Vec2x4T`] groups their x and y components for axis-parallel +//! arithmetic. [`DVec2`] and [`DVec2x4T`] provide double-precision counterparts. +//! [`FinitePointField`] validates a row-indexed slice's coordinates, and [`Bounds2`] describes a +//! finite ordered bounding box. +//! +//! Transformations: [`Rotation`] models an angle, [`Translation`](translation::Translation) an +//! offset, [`Similarity`] a positive uniform scale with rotation and translation, and [`Transform`] +//! a general affine map. Prefer the most constrained type that models the operation. Each converts +//! into [`Transform`] through [`From`], and `a.then(b)` composes in application order. Inverse +//! methods compute floating-point approximations subject to their documented range conditions. +//! [`Similarity::fit`] and [`Transform::fit_uniform`] estimate maps from corresponding points. +//! +//! Embeddings: [`VecN`] provides fixed-width single-precision vectors and distance kernels. +//! [`BoxedVecN`] owns aligned heap storage exposed through [`AlignedVecN`]. [`DVecN`], +//! [`BoxedDVecN`] and [`AlignedDVecN`] provide double-precision storage and reductions. [`MatrixN`] +//! stores rows with a fixed embedding width. +//! +//! Dense solves: [`DSquareMatrix`] is a runtime-order double-precision square matrix. +//! [`DSquareMatrix::cholesky`] computes a Cholesky factor for symmetric positive-definite systems, +//! rejecting nonpositive or non-finite computed pivots. Its +//! [`DCholeskyFactor`](dsquare::DCholeskyFactor) performs triangular solves. Rounding can make a +//! mathematically positive-definite input fail factorization. +//! +//! Neighbours: [`KdTree`] indexes a finite point field and orders selected neighbours by +//! double-precision squared distance, breaking ties by row. Its [selection model](kdtree) explains +//! the two walks and the precision limits of radius pruning. +//! +//! Affinities: [`AffinityCurve`] evaluates a distance-based affinity and attractive/repulsive +//! gradients. [`AffinityCurve::fit`] fits its parameters to a sampled target curve. Clipping, +//! regularization and numerical stopping conditions are part of those operations' contracts. +//! +//! Scalar domains: [`Finite`], [`Positive`] and related types express value ranges. [`softplus`], +//! [`NonNegative::huber`] and [`NonNegative::sigmoid`] provide common scalar functions. +//! [`narrow_f32`] checks the result of a double-to-single-precision conversion. [`Derivation`] +//! carries raw intermediate arithmetic toward a destination [`Domain`](derivation::Domain), and +//! [`Derivation::finish`] validates the final value or returns [`Diverged`]. +//! +//! # Precision +//! +//! Single-precision storage does not imply single-precision arithmetic throughout. Wide distance +//! methods return `f64`, and fitting and reductions often accumulate in `f64` before any final +//! narrowing. Each arithmetic step can round at its working precision. Widening does not make a sum +//! exact, recover distinctions already lost from stored coordinates, or prevent cancellation. +//! +//! Individual kernels state their input domains, handling of special-case results and reduction +//! order. Parallel grouping can change acceptance decisions or final bits. An inverse or algebraic +//! identity in the real-valued model does not by itself promise an exact floating-point round trip. //! //! # Batching //! -//! Hot loops work in [`Vec2x4T`]. Convert `[Vec2; 4]` once at the loop boundary (paying one -//! shuffle), then run axis-parallel arithmetic inside and write back with [`Vec2x4T::from_lanes`]. -//! Batch types align for full-width vector loads, and conversions to [`Simd`] compile to single -//! load and store instructions. -//! -//! [`Simd`]: core::simd::Simd +//! Convert an array of four [`Vec2`] points into [`Vec2x4T`] for a sequence of axis-parallel +//! operations. [`Vec2x4T::into_lanes`] exposes separate x and y vectors, and +//! [`Vec2x4T::from_lanes`] combines them. [`Vec2x4T::transpose`] returns natural point order for +//! access through [`Vec2x4::as_array`](vec2::Vec2x4::as_array). Alignment and lane layout support +//! SIMD access, while instruction selection and conversion cost depend on the target and +//! optimization context. #![expect(unsafe_code)] #![expect( dead_code, diff --git a/libs/@local/graph/atlas/src/math/rotation/mod.rs b/libs/@local/graph/atlas/src/math/rotation/mod.rs index 54fa911f6ff..10b52de7484 100644 --- a/libs/@local/graph/atlas/src/math/rotation/mod.rs +++ b/libs/@local/graph/atlas/src/math/rotation/mod.rs @@ -1,4 +1,6 @@ -//! Rotations about the origin, stored in decomposed form. +//! Rotations about the origin with arithmetic composition. +//! +//! [`Rotation`] retains cosine and sine to compose rotations without recovering their angles. use core::simd::Simd; @@ -10,24 +12,32 @@ use super::{ #[cfg(test)] mod tests; -/// A rotation about the origin, stored as the unit vector `(cos, sin)`. +/// A rotation about the origin represented by approximate cosine and sine. /// -/// The decomposed representation is the contract of this type. It computes the angle's cosine and -/// sine once, or accepts them directly, and every later operation is plain arithmetic on them: -/// composing two rotations multiplies the unit vectors, which adds the angles without any -/// trigonometric calls, and inverting negates the sine, which is exact. +/// For stored components c and s, application uses the matrix R = [[c, −s], [s, c]]. A rotation +/// requires finite components with c² + s² ≈ 1. [`from_radians`](Self::from_radians) computes them +/// from a finite angle, and [`from_cos_sin`](Self::from_cos_sin) accepts a caller-established pair. +/// Byte construction does not validate this numerical condition. +/// +/// Composition multiplies the represented complex numbers, adding their angles in real arithmetic +/// without trigonometric calls. [`inverse`](Self::inverse) conjugates the pair by exactly negating +/// its finite sine. Application and composition still round in `f32`. /// /// Angles follow the mathematical convention: radians, counterclockwise, with `x` growing right and /// `y` growing up. In a `y`-down space (such as screen coordinates) the visual direction of /// rotation reverses. /// -/// Note that long composition chains accumulate rounding in the stored vector, letting it drift -/// off the unit circle by about one unit in the last place per composition. Renormalizing -/// periodically corrects the vector's length and leaves the angle alone. +/// Composition can accumulate error in both length and angle. [`renormalize`](Self::renormalize) +/// corrects length drift up to rounding, but cannot recover an angle lost through earlier rounding. +/// The stored pair's length scales every applied vector in the real-arithmetic model. +/// +/// # Example /// -/// # Examples +/// This example is ignored because [`Rotation`] is crate-private. /// /// ```ignore +/// use crate::math::{Rotation, Vec2}; +/// /// let quarter = Rotation::from_radians(core::f32::consts::FRAC_PI_2); /// /// let rotated = quarter.apply(Vec2::new(1.0, 0.0)); @@ -54,8 +64,8 @@ impl Rotation { /// Creates a rotation from an angle in radians. /// - /// This is the only constructor that calls into trigonometry. Every later operation reuses the - /// resulting cosine and sine. + /// `radians` must be finite. The stored components are the approximations returned by + /// [`f32::sin_cos`]. #[inline] #[must_use] pub(crate) fn from_radians(radians: f32) -> Self { @@ -66,41 +76,45 @@ impl Rotation { /// Creates a rotation directly from its cosine and sine. /// - /// The pair must lie on the unit circle: `cos · cos + sin · sin = 1` up to rounding. This is - /// useful when the pair is already available, for example from normalizing a direction vector, - /// and avoids round-tripping through an angle. + /// For a rotation, `cos` and `sin` must be finite with cos² + sin² ≈ 1. This avoids recovering + /// an angle when a normalized direction is already available. A finite non-unit pair can + /// instead be rescaled with [`renormalize`](Self::renormalize), subject to that method's + /// numerical conditions. #[inline] #[must_use] pub(crate) const fn from_cos_sin(cos: f32, sin: f32) -> Self { Self(Vec2::new(cos, sin)) } - /// Returns the cosine of the rotation angle. + /// Returns the stored cosine component. #[inline] #[must_use] pub(crate) const fn cos(self) -> f32 { self.0.x() } - /// Returns the sine of the rotation angle. + /// Returns the stored sine component. #[inline] #[must_use] pub(crate) const fn sin(self) -> f32 { self.0.y() } - /// Returns the rotation angle in radians, in `(-pi, pi]`. + /// Returns the angle of the stored pair in radians. + /// + /// For a finite nonzero pair, [`f32::atan2`] returns an approximation in [−π, π], with either + /// endpoint possible according to the sine's sign, including signed zero. #[inline] #[must_use] pub(crate) fn radians(self) -> f32 { self.sin().atan2(self.cos()) } - /// Returns the rotation equivalent to applying `self` first, then `next`. + /// Composes `self` followed by `next`. /// - /// Rotations commute, so the order only matters for consistency with the other transform types. - /// The composition adds the two angles by multiplying the stored unit vectors; no trigonometric - /// calls occur. + /// In real arithmetic the pair is (c₁c₂ − s₁s₂, s₁c₂ + c₁s₂), and rotations commute. Rounding + /// these products and sums in `f32` can make application of the composed pair differ from + /// sequential application. #[inline] #[must_use] pub(crate) const fn then(self, next: Self) -> Self { @@ -110,13 +124,15 @@ impl Rotation { )) } - /// Rescales the stored vector back onto the unit circle. + /// Rescales the stored pair to approximately unit length. + /// + /// This multiplies both components by an approximation to 1 / √(c² + s²), using one square root + /// and reciprocal. A positive common factor preserves the pair's direction in real arithmetic. + /// The final products round separately. /// - /// Composition accumulates rounding in the vector's length at about one unit in the last place - /// per [`then`](Self::then); a drifted length scales every vector passed to - /// [`apply`](Self::apply) by that factor. Renormalizing divides the drift out at the cost of - /// one square root, leaving the angle unchanged up to rounding. Calling it once every few - /// hundred compositions keeps the error invisible in `f32`. + /// The components must be finite, and the computed squared length and reciprocal length must be + /// finite and positive. Pairs near unit length satisfy these conditions. A zero pair or an + /// extreme non-unit pair can produce NaNs or infinities instead of a normalized rotation. #[inline] #[must_use] pub(crate) fn renormalize(self) -> Self { @@ -129,18 +145,20 @@ impl Rotation { Self(Vec2::new(self.cos() * scale, self.sin() * scale)) } - /// Returns the rotation by the negated angle. + /// Conjugates the stored pair to represent the negated angle. + /// + /// Negating a finite sine introduces no rounding. /// - /// This negates the stored sine, which is exact: applying a rotation and then its inverse - /// reproduces the rounding of the forward and backward applications only, never of the - /// inversion itself. + /// The represented matrices satisfy `RᵀR = (c² + s²)I` in real arithmetic: any length drift + /// remains in an inverse round trip, in addition to application rounding. This is an + /// approximate inverse for a near-unit pair. #[inline] #[must_use] pub(crate) const fn inverse(self) -> Self { Self(Vec2::new(self.cos(), -self.sin())) } - /// Rotates a single vector about the origin. + /// Applies the stored rotation matrix with separate `f32` products and sums. #[inline] #[must_use] pub(crate) const fn apply(self, vec: Vec2) -> Vec2 { @@ -150,12 +168,12 @@ impl Rotation { ) } - /// Rotates four vectors at once, entirely in SIMD registers. + /// Applies the stored rotation matrix to four vectors with SIMD arithmetic. /// - /// On targets with native FMA the fused multiply-adds round once where [`apply`](Self::apply) - /// rounds after each multiply and each add. Results differ by at most a few units in the last - /// place of the intermediate products; where the products cancel, that absolute difference - /// spans many units in the last place of the small result. + /// Each axis uses one rounded product and one fused multiply-add. Fusion rounds its product and + /// addition once, independently of native FMA availability. This differs from + /// [`apply`](Self::apply)'s separate operations, especially near cancellation or overflow. No + /// uniform result-relative ULP bound relates the two paths. #[inline] #[must_use] pub(crate) fn apply_x4(self, batch: Vec2x4T) -> Vec2x4T { diff --git a/libs/@local/graph/atlas/src/math/rotation/tests.rs b/libs/@local/graph/atlas/src/math/rotation/tests.rs index c093487bab8..5542193d80c 100644 --- a/libs/@local/graph/atlas/src/math/rotation/tests.rs +++ b/libs/@local/graph/atlas/src/math/rotation/tests.rs @@ -26,8 +26,6 @@ fn rotation_composition_adds_angles() { fn rotation_apply_turns_a_positive_angle_counterclockwise() { let quarter_turn = Rotation::from_radians(core::f32::consts::FRAC_PI_2); - // Every other test in this file passes under a global sign flip of `apply`; these hand - // values are the direction witness. assert_vec2_close(quarter_turn.apply(Vec2::new(1.0, 0.0)), Vec2::new(0.0, 1.0)); assert_vec2_close( quarter_turn.apply(Vec2::new(0.0, 1.0)), @@ -62,7 +60,7 @@ fn rotation_apply_x4_matches_apply() { fn renormalize_removes_composition_drift() { let step = Rotation::from_radians(1e-3); - // Walk once around the circle in small steps to accumulate drift. + // 6283 steps of approximately 10⁻³ radians make nearly one full turn let mut chained = Rotation::IDENTITY; for _ in 0..6283 { chained = chained.then(step); @@ -80,25 +78,22 @@ fn renormalize_removes_composition_drift() { "renormalizing must not move the vector further off the unit circle", ); assert!((norm(renormalized) - 1.0).abs() < 4.0 * f32::EPSILON); - // Renormalizing preserves the angle, so after ~2 pi the rotation is close to the identity. + // 6.283 radians is about 0.000185 radians short of 2π assert!((renormalized.radians()).abs() < 1e-2); } -/// An angle within a few turns of zero, where `sin_cos` is well-conditioned. +/// Generates finite angles within a few turns of zero. fn angle() -> impl Strategy { -16.0_f32..16.0 } -/// A vector with coordinates bounded to the well-conditioned `-1e5..1e5` range. +/// Generates vectors with coordinates in `-1e5..1e5`. /// -/// The rotation laws are about algebra, not overflow. +/// This range keeps the tested products and squared lengths finite. fn vec2_strategy() -> impl Strategy { (-1e5_f32..1e5, -1e5_f32..1e5).prop_map(|(x, y)| Vec2::new(x, y)) } -/// Rotation preserves length. -/// -/// `|apply(v)| == |v|` up to rounding scaled by the vector's magnitude. #[property_test] fn apply_preserves_length( #[strategy = angle()] radians: f32, @@ -116,9 +111,6 @@ fn apply_preserves_length( ); } -/// Composition distributes over application. -/// -/// `a.then(b).apply(v) == b.apply(a.apply(v))` up to rounding scaled by the vector's magnitude. #[property_test] fn then_matches_sequential_application( #[strategy = angle()] first_radians: f32, @@ -141,9 +133,6 @@ fn then_matches_sequential_application( ); } -/// The inverse undoes the rotation. -/// -/// `inverse().apply(apply(v)) == v` up to rounding scaled by the vector's magnitude. #[property_test] fn inverse_undoes_apply( #[strategy = angle()] radians: f32, @@ -163,10 +152,10 @@ fn inverse_undoes_apply( ); } -/// Renormalizing preserves the angle. +/// Compares the pair before and after renormalizing a bounded composition chain. /// -/// The cosine and sine keep their direction (compared componentwise rather than through `atan2`, -/// which wraps at pi), even after enough compositions to accumulate drift. +/// Componentwise comparison avoids the angular branch cut at ±π. It bounds the change to the stored +/// pair, rather than certifying an exact recovered angle. #[property_test] fn renormalize_preserves_the_angle( #[strategy = angle()] radians: f32, @@ -180,9 +169,7 @@ fn renormalize_preserves_the_angle( let renormalized = chained.renormalize(); - // The drifted vector's length stays within a couple of ulps of - // one per composition, so dividing it out moves each component by - // at most that relative amount. + // the absolute tolerance grows with the number of rounded compositions let tolerance = 8.0 * f32::EPSILON * f32::from(u8::try_from(compositions + 2).expect("bounded below 66")); prop_assert!( @@ -196,12 +183,10 @@ fn renormalize_preserves_the_angle( ); } -/// Renormalization rescales both components even far from unit length. +/// Renormalizes a pair whose length exceeds one by a quarter. /// -/// The stored pair `(0.75, 1.0)` has norm `1.25` exactly, so the rescale is a quarter of the -/// magnitude and each mis-scaled component misses its target by far more than the tolerance. -/// Drift-sized inputs cannot see that: near unit length, multiplying and dividing by the scale -/// land within drift of each other. +/// The squared length of (3/4, 1) is 9/16 + 1 = 25/16. Its length is exactly 5/4, and multiplying +/// by 4/5 gives the normalized pair (3/5, 4/5), compared with a tolerance for rounding. #[test] fn renormalize_rescales_a_quarter_off_unit_pair() { let renormalized = Rotation::from_cos_sin(0.75, 1.0).renormalize(); diff --git a/libs/@local/graph/atlas/src/math/scalar/d_non_negative.rs b/libs/@local/graph/atlas/src/math/scalar/d_non_negative.rs index a2109571d82..07d025d18d1 100644 --- a/libs/@local/graph/atlas/src/math/scalar/d_non_negative.rs +++ b/libs/@local/graph/atlas/src/math/scalar/d_non_negative.rs @@ -30,21 +30,25 @@ pub(crate) use d_non_negative; /// A finite, non-negative `f64`, valid by construction. /// -/// The double-precision twin of [`NonNegative`]. Zero passes, so the type carries tolerances and -/// floors that may legitimately switch a check off, and measured magnitudes such as distances. +/// The double-precision twin of [`NonNegative`] admits zero. It represents measured magnitudes such +/// as distances and tolerances or floors that may use zero to switch a check off. /// /// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value, /// with `-0.0` and `+0.0` the same value: construction canonicalizes the sign of zero. Values /// sort and key ordered maps like the numbers they hold, with no NaN case. /// -/// Arithmetic whose result provably stays in the domain stays in the type. The square root of a -/// non-negative value is non-negative ([`sqrt`](Self::sqrt)), while subtracting one non-negative -/// value from another leaves the domain yet provably stays finite, so `-` outputs [`DFinite`]. -/// Serialization writes plain numbers and deserialization re-validates. +/// Arithmetic preserves the type when its result provably remains in the domain. The square root of +/// a nonnegative value is nonnegative ([`sqrt`](Self::sqrt)). Subtraction returns [`DFinite`] to +/// admit negative differences while preserving finiteness. Serialization writes plain numbers and +/// deserialization re-validates. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{DNonNegative}; +/// /// assert_eq!( /// DNonNegative::new(0.0) /// .expect("zero disables the floor") @@ -66,8 +70,8 @@ impl DNonNegative { /// Views a slice of readings as raw `f64`s. /// - /// The view is zero-cost: the type is `repr(transparent)` over `f64`, so the slices share one - /// layout. It serves a boundary whose vocabulary is the raw primitive. + /// The `repr(transparent)` representation over `f64` gives both slices the same layout, + /// permitting a zero-copy view. #[inline] pub(crate) fn slice_as_raw(values: &[Self]) -> &[f64] { zerocopy::FromBytes::ref_from_bytes(zerocopy::IntoBytes::as_bytes(values)) @@ -156,10 +160,10 @@ impl DNonNegative { Self(self.0.sqrt()) } - /// Squares into a derivation, claiming nothing. + /// Squares with deferred validation. /// - /// Squaring doubles an unbounded exponent, a fat exit, so the product enters the fold raw - /// and the claim waits for the finish. + /// Squaring can overflow the finite range. The product enters the derivation for validation at + /// the finish. #[inline] pub(crate) const fn square(self) -> Derivation { Derivation::raw(self.0 * self.0) @@ -167,8 +171,8 @@ impl DNonNegative { /// Narrows to working precision with round-to-nearest. /// - /// A value beyond the `f32` range overflows to `+∞` and asserts in debug builds through the - /// constructor. + /// The rounded result must be finite. Values that round to positive infinity lie outside + /// the result type's domain. #[expect( clippy::cast_possible_truncation, reason = "the rounding cast is the operation itself" @@ -181,8 +185,8 @@ impl DNonNegative { /// Converts a count, exactly. /// - /// Every `u16` is non-negative and far inside `f64`'s exact-integer range, so the conversion - /// is total and no re-validation happens. + /// Every `u16` is nonnegative and exactly representable in `f64`. The conversion is total and + /// requires no re-validation. #[inline] #[must_use] pub(crate) const fn from_u16(value: u16) -> Self { @@ -191,8 +195,7 @@ impl DNonNegative { /// Narrows to the strictly positive domain. /// - /// Returns [`None`] exactly at zero, so an `if let` on the result is the zero guard and the - /// positivity witness in one move. + /// Returns [`None`] exactly at zero. #[inline] #[must_use] pub(crate) const fn positive(self) -> Option { @@ -213,14 +216,9 @@ impl DNonNegative { self.0.is_subnormal() } - /// Returns whether the value stayed in domain. - /// - /// Construction admits only finite values and arithmetic escapes to `+∞` on overflow, so a - /// non-finite reading is exactly an escaped one. + /// Returns whether the stored reading is finite. /// - /// The one caller shape is a validation point that rejects escaped readings before acting on - /// a computed value. Anywhere else the query re-checks what construction already proved, and - /// the check itself is the defect. + /// Detects a non-finite result after arithmetic whose range requirements were not met. #[inline] #[must_use] pub(crate) const fn is_finite(self) -> bool { @@ -238,13 +236,10 @@ impl DNonNegative { Self::new(self.0 / rhs.get()) } - /// Raises to a raw power, staying non-negative. + /// Raises to a real power with deferred validation. /// - /// A non-negative base admits no NaN from `powf`. Zero raised to a positive exponent is - /// zero, and zero raised to the zero exponent is one. A positive base stays positive under - /// any exponent. Zero to a negative power and an overflowing result escape to `+∞` - wrong - /// readings rather than soundness breaks, since no unsafe code trusts the domain - and - /// assert in debug builds through the constructor. + /// Overflow and zero raised to a negative exponent produce infinity in the [`Derivation`]. Zero + /// raised to zero is one. Underflow to zero remains nonnegative. #[inline] #[must_use] pub(crate) fn powf(self, exponent: f64) -> Self { @@ -279,7 +274,7 @@ impl fmt::LowerExp for DNonNegative { const impl PartialEq for DNonNegative { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -305,7 +300,7 @@ const impl Ord for DNonNegative { impl Hash for DNonNegative { #[inline] fn hash(&self, state: &mut H) { - // canonical bits: equal values share one bit pattern, so `Hash` agrees with `Eq` + // hashing each value's canonical bit pattern preserves agreement with `Eq` state.write_u64(self.0.to_bits()); } } @@ -315,8 +310,8 @@ const impl core::ops::Sub for DNonNegative { /// Subtracts, into the finite domain. /// - /// The difference of two non-negative finite values is finite, with no re-validation: its - /// magnitude never exceeds the larger operand, so the subtraction cannot overflow. Equal + /// The difference of two nonnegative finite values has a magnitude that never exceeds the + /// larger operand. The subtraction cannot overflow and requires no re-validation. Equal /// operands give `+0.0`. #[inline] fn sub(self, rhs: Self) -> DFinite { @@ -339,10 +334,6 @@ const impl core::ops::Neg for DNonNegative { const impl core::ops::Add for DNonNegative { type Output = f64; - /// Adds a raw offset. - /// - /// The raw operand is arbitrary, so the sum can leave any bounded domain and returns a raw - /// float. #[inline] fn add(self, rhs: f64) -> f64 { self.0 + rhs @@ -352,10 +343,6 @@ const impl core::ops::Add for DNonNegative { const impl core::ops::Sub for DNonNegative { type Output = f64; - /// Subtracts a raw offset. - /// - /// The raw operand is arbitrary, so the difference can leave any bounded domain and returns - /// a raw float. #[inline] fn sub(self, rhs: f64) -> f64 { self.0 - rhs @@ -365,10 +352,6 @@ const impl core::ops::Sub for DNonNegative { const impl core::ops::Add for f64 { type Output = f64; - /// Adds a non-negative offset to a raw `f64`. - /// - /// The raw operand is arbitrary, so the sum can leave any bounded domain and returns a raw - /// float. #[inline] fn add(self, rhs: DNonNegative) -> f64 { self + rhs.0 @@ -378,10 +361,6 @@ const impl core::ops::Add for f64 { const impl core::ops::Sub for f64 { type Output = f64; - /// Subtracts a non-negative offset from a raw `f64`. - /// - /// The raw operand is arbitrary, so the difference can leave any bounded domain and returns - /// a raw float. #[inline] fn sub(self, rhs: DNonNegative) -> f64 { self - rhs.0 @@ -393,9 +372,9 @@ const impl core::ops::Sub for DNonNegative { /// Subtracts a positive value, into the finite domain. /// - /// The difference of two finite values of one sign is finite, with no re-validation: its - /// magnitude never exceeds the larger operand, so the subtraction cannot overflow. The - /// sign is the reading, negative whenever the positive operand exceeds the value. + /// The difference of two finite values of one sign has a magnitude that never exceeds the + /// larger operand. The subtraction cannot overflow and requires no re-validation. The result is + /// negative whenever the positive operand exceeds `self`. #[inline] fn sub(self, rhs: DPositive) -> DFinite { DFinite::new_unchecked(self.0 - rhs.get()) @@ -405,8 +384,11 @@ const impl core::ops::Sub for DNonNegative { const impl core::ops::Add for DNonNegative { type Output = Self; - /// Accumulates a fraction: an open unit fraction is a finite non-negative value, and a sum - /// with a value below one cannot overflow. + /// Adds a fraction without overflowing the finite domain. + /// + /// Binary64 addition is monotone and round(MAX + 1) = MAX, where MAX is [`f64::MAX`]. The + /// operands are bounded by MAX and one, and their rounded sum is at most MAX. Therefore + /// adding an in-domain fraction remains finite and non-negative. #[inline] fn add(self, rhs: OpenUnitFraction) -> Self { Self(self.0 + rhs.get()) @@ -425,10 +407,8 @@ const impl core::ops::Add for DNonNegative { /// Adds. /// - /// A sum of non-negatives is never NaN and never `-0.0`. Overflow escapes to `+∞` - a - /// wrong reading rather than a soundness break, since no unsafe code trusts the domain and - /// a persisted value re-validates at construction - and asserts in debug builds, mirroring - /// integer `+`. + /// The rounded sum must remain finite. A sum of in-domain values is never NaN or `-0.0`, + /// but it can overflow to positive infinity. #[inline] fn add(self, rhs: Self) -> Self { let sum = self.0 + rhs.0; @@ -446,8 +426,10 @@ const impl core::ops::AddAssign for DNonNegative { } const impl core::ops::AddAssign for DNonNegative { - /// Accumulates a fraction: a positive unit fraction is a finite non-negative value, and a - /// sum with a value at most one cannot overflow. + /// Accumulates a positive unit fraction without overflowing. + /// + /// Monotone rounding bounds the sum by round(MAX + 1) = MAX, as for addition of an open unit + /// fraction. #[inline] fn add_assign(&mut self, rhs: PositiveUnitFraction) { *self = *self + Self(rhs.get()); @@ -459,8 +441,8 @@ const impl core::ops::Add for DNonNegative { /// Adds a positive value, into the positive domain. /// - /// Rounding is monotone, so the sum is at least the positive operand and never reaches - /// zero. Overflow escapes to `+∞` and asserts in debug builds through the constructor. + /// The rounded sum must remain finite. Monotone rounding keeps it at least as large as the + /// positive operand, but does not prevent overflow. #[inline] fn add(self, rhs: DPositive) -> DPositive { DPositive::new_unchecked(self.0 + rhs.get()) @@ -470,12 +452,11 @@ const impl core::ops::Add for DNonNegative { const impl core::ops::Mul for DNonNegative { type Output = Derivation; - /// Multiplies. + /// Multiplies with deferred validation of overflow. /// - /// A product of finite non-negatives is never NaN and never negative. Overflow escapes to - /// `+∞` - a wrong reading rather than a soundness break, since no unsafe code trusts the - /// domain - and asserts in debug builds through the constructor. Underflow rounds to zero, - /// inside the domain. + /// For finite non-negative operands the product is never NaN or negative. The derivation + /// carries overflow to positive infinity until its finish. Underflow rounds to zero, inside + /// the target domain. #[inline] fn mul(self, rhs: Self) -> Derivation { Derivation::raw(self.0 * rhs.0) @@ -485,9 +466,10 @@ const impl core::ops::Mul for DNonNegative { const impl core::ops::Mul for DNonNegative { type Output = Derivation; - /// Multiplies by a positive factor, staying non-negative. + /// Multiplies by a positive factor with deferred validation. /// - /// Zero stays exactly zero, and overflow escapes to `+∞` as the in-family product does. + /// A zero operand gives zero. Finite products are non-negative, while overflow produces + /// positive infinity in the derivation. #[inline] fn mul(self, rhs: DPositive) -> Derivation { Derivation::raw(self.0 * rhs.get()) @@ -497,11 +479,11 @@ const impl core::ops::Mul for DNonNegative { const impl core::ops::Div for DNonNegative { type Output = Derivation; - /// Divides by a positive divisor, staying non-negative. + /// Divides by a nonzero divisor with deferred validation of overflow. /// - /// The divisor is never zero and never NaN, so the sign is closed and no NaN can arise. The - /// exponents compose: a large numerator over a small divisor overflows to `+∞`, the fat - /// exit the derivation carries to its finish. Underflow rounds to zero, inside the domain. + /// The divisor is never zero or NaN, and the numerator is finite and non-negative. Their + /// quotient cannot be NaN or negative. A large numerator over a small divisor can overflow to + /// positive infinity. Underflow rounds to zero, inside the target domain. #[inline] fn div(self, rhs: DPositive) -> Derivation { Derivation::raw(self.0 / rhs.get()) @@ -511,11 +493,11 @@ const impl core::ops::Div for DNonNegative { const impl core::ops::Div for DNonNegative { type Output = Derivation; - /// Divides within the family, staying non-negative where the quotient exists. + /// Divides with deferred validation of zero division and overflow. /// - /// A zero divisor sends a positive numerator to `+∞` and zero to NaN, both upward exits - /// the derivation carries to its finish. Overflow escapes the same way as the divisor - /// shrinks, and underflow rounds to zero, inside the domain. + /// A zero divisor gives positive infinity for a positive numerator and NaN for zero. A positive + /// divisor follows the range behavior of division by [`DPositive`]: overflow can produce + /// infinity, while underflow produces an in-domain zero. #[inline] fn div(self, rhs: Self) -> Derivation { Derivation::raw(self.0 / rhs.0) @@ -523,7 +505,6 @@ const impl core::ops::Div for DNonNegative { } const impl PartialEq for DNonNegative { - /// Compares across the scalar family, in one precision with no widening. #[inline] fn eq(&self, other: &DPositive) -> bool { self.0 == other.get() @@ -531,7 +512,6 @@ const impl PartialEq for DNonNegative { } const impl PartialOrd for DNonNegative { - /// Orders across the scalar family, in one precision with no widening. #[inline] fn partial_cmp(&self, other: &DPositive) -> Option { self.0.partial_cmp(&other.get()) @@ -543,7 +523,6 @@ impl proptest::arbitrary::Arbitrary for DNonNegative { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole domain, zero and subnormals included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -554,14 +533,12 @@ impl proptest::arbitrary::Arbitrary for DNonNegative { } impl serde::Serialize for DNonNegative { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f64(self.0) } } impl<'de> serde::Deserialize<'de> for DNonNegative { - /// Deserializes a plain number, refusing values outside the finite non-negative range. fn deserialize>(deserializer: D) -> Result { let value = f64::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { @@ -574,7 +551,6 @@ impl<'de> serde::Deserialize<'de> for DNonNegative { } const impl From for DNonNegative { - /// Widens into the enclosing domain: every positive value is non-negative. #[inline] fn from(value: DPositive) -> Self { Self(value.get()) @@ -582,7 +558,6 @@ const impl From for DNonNegative { } const impl From for DNonNegative { - /// Widens into double precision, exactly: the canonical zero and the domain both survive. #[inline] fn from(value: NonNegative) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. @@ -591,8 +566,6 @@ const impl From for DNonNegative { } const impl From for DNonNegative { - /// Widens into double precision and the enclosing domain, exactly: every positive value - /// is non-negative. #[inline] fn from(value: Positive) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. @@ -601,7 +574,6 @@ const impl From for DNonNegative { } const impl From for DNonNegative { - /// [0, 1] is non-negative. #[inline] fn from(value: UnitFraction) -> Self { Self(value.get()) @@ -620,8 +592,9 @@ const impl core::ops::Mul for UnitFraction { #[inline] fn mul(self, rhs: DNonNegative) -> DNonNegative { - // In domain with no check: a fraction in [0, 1] scales the magnitude toward zero, so the - // product stays finite and non-negative, and a zero product keeps the canonical +0.0. + // A fraction in [0, 1] cannot increase the magnitude of a nonnegative finite value. The + // product remains finite and nonnegative without a check, with canonical +0.0 for a zero + // product. DNonNegative(self.get() * rhs.0) } } @@ -631,8 +604,9 @@ const impl core::ops::Mul for OpenUnitFraction { #[inline] fn mul(self, rhs: DNonNegative) -> DNonNegative { - // In domain with no check: a fraction in (0, 1) scales the magnitude toward zero, so the - // product stays finite and non-negative, and a zero product keeps the canonical +0.0. + // A fraction in (0, 1) cannot increase the magnitude of a nonnegative finite value. The + // product remains finite and nonnegative without a check, with canonical +0.0 for a zero + // product. DNonNegative(self.get() * rhs.0) } } diff --git a/libs/@local/graph/atlas/src/math/scalar/d_positive.rs b/libs/@local/graph/atlas/src/math/scalar/d_positive.rs index 13ff8ca5499..4b1386046cd 100644 --- a/libs/@local/graph/atlas/src/math/scalar/d_positive.rs +++ b/libs/@local/graph/atlas/src/math/scalar/d_positive.rs @@ -16,8 +16,8 @@ use crate::math::Derivation; /// Validates a positive double-precision literal at compile time. /// -/// The expansion is a `const` block over [`DPositive::new`], so a literal outside the domain fails -/// the build instead of a test run. Runtime values keep the checked constructor. +/// A `const` block validates the literal with [`DPositive::new`] during compilation. A literal +/// outside the domain fails the build. Runtime values use the checked constructor. macro_rules! d_positive { ($value:expr) => { const { $crate::math::DPositive::new($value).expect("the literal is finite and positive") } @@ -42,13 +42,15 @@ impl Error for NotPositive {} /// A finite, strictly positive `f64`, valid by construction. /// -/// The double-precision twin of [`Positive`], named as [`DVecN`](crate::math::DVecN) is to -/// [`VecN`](crate::math::VecN): configuration fields that steer double-precision arithmetic -/// carry their domain in the type, and the consuming site validates nothing. +/// Use [`Positive`] when the finite domain needs only single precision. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{DPositive}; +/// /// assert_eq!( /// DPositive::new(1.0e-8) /// .expect("the radius floor is positive") @@ -59,14 +61,18 @@ impl Error for NotPositive {} /// assert_eq!(DPositive::new(f64::INFINITY), None); /// ``` /// -/// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value. -/// The domain excludes NaN and both zeros, so every value owns one bit pattern with no -/// canonicalization step. +/// The domain excludes NaN and both zeros, giving every value one bit pattern without +/// canonicalization. [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow +/// numeric value. #[derive(Copy, Clone, zerocopy::Immutable)] #[repr(transparent)] pub(crate) struct DPositive(f64); impl DPositive { + /// The unit in the last place of one, `2⁻⁵²`. + /// + /// The spacing between one and the next larger `f64`, the unit a tolerance stated in ulps + /// multiplies. pub(crate) const EPSILON: Self = Self::new(f64::EPSILON).unwrap(); /// The value one. pub(crate) const ONE: Self = Self(1.0); @@ -90,8 +96,8 @@ impl DPositive { /// Converts a nonzero count, exactly. /// - /// Every nonzero `u16` is strictly positive and far inside `f64`'s exact-integer range, so - /// the conversion is total and no re-validation happens. + /// Every nonzero `u16` is strictly positive and exactly representable in `f64`. The conversion + /// is total and requires no re-validation. #[inline] #[must_use] pub(crate) const fn from_u16(value: NonZero) -> Self { @@ -100,8 +106,8 @@ impl DPositive { /// Converts a nonzero count, exactly. /// - /// Every nonzero `u32` is strictly positive and inside `f64`'s exact-integer range, so the - /// conversion is total and no re-validation happens. + /// Every nonzero `u32` is strictly positive and exactly representable in `f64`. The conversion + /// is total and requires no re-validation. #[inline] #[must_use] pub(crate) const fn from_u32(value: NonZero) -> Self { @@ -124,15 +130,13 @@ impl DPositive { /// Returns whether `value`'s exact bits are a stored positive value. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: the domain - /// holds no zero of either sign and accepted values store bit for bit, so the bits are - /// valid exactly when [`new`](Self::new) accepts the value. + /// Accepted values retain their bits, and the domain excludes both zeros. This validates + /// persisted bits exactly when [`new`](Self::new) accepts the corresponding value. #[inline] #[must_use] pub(crate) const fn is_canonical(value: f64) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. + // compare with the constructed value to account for normalization Some(accepted) => accepted.0.to_bits() == value.to_bits(), None => false, } @@ -170,7 +174,7 @@ impl DPositive { /// /// The logarithm of a positive value is never NaN and always finite, because the smallest /// positive subnormal's logarithm is only about `-745` and the largest finite value's about - /// `710`. The sign is the reading, so the result carries finiteness alone. + /// `710`. #[inline] #[must_use] pub(crate) fn ln(self) -> DFinite { @@ -180,8 +184,8 @@ impl DPositive { /// Divides, refusing the escape. /// /// The quotient of positives is never NaN and never negative. Returns [`None`] exactly when - /// the quotient leaves the domain, overflowing to `+∞` or underflowing to zero, where the - /// plain division would escape and assert. + /// the quotient leaves the domain, overflowing to positive infinity or underflowing to zero. + /// The division operator instead carries that raw result in a [`Derivation`]. #[inline] #[must_use] pub(crate) const fn checked_div(self, rhs: Self) -> Option { @@ -192,7 +196,7 @@ impl DPositive { const impl PartialEq for DPositive { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -218,7 +222,7 @@ const impl Ord for DPositive { impl Hash for DPositive { #[inline] fn hash(&self, state: &mut H) { - // one bit pattern per value, so `Hash` agrees with `Eq` + // hashing each value's unique bit pattern preserves agreement with `Eq` state.write_u64(self.0.to_bits()); } } @@ -244,11 +248,10 @@ impl fmt::Display for DPositive { const impl core::ops::Div for f64 { type Output = f64; - /// Divides a double-precision measurement by the positive value, staying in `f64`. + /// Divides a raw reading by a finite nonzero divisor. /// - /// The divisor is never zero and never NaN, so a NaN quotient arrives only through the - /// numerator. An infinite quotient arrives through the numerator or through overflow - /// against a small divisor. + /// The divisor is never zero or NaN. The quotient is NaN only for a NaN numerator. An infinite + /// numerator or overflow of a finite quotient produces infinity. #[inline] fn div(self, rhs: DPositive) -> f64 { self / rhs.0 @@ -256,7 +259,6 @@ const impl core::ops::Div for f64 { } const impl core::ops::DivAssign for f64 { - /// Divides a double-precision measurement in place, as the binary form does. #[inline] fn div_assign(&mut self, rhs: DPositive) { *self /= rhs.0; @@ -268,9 +270,9 @@ const impl core::ops::Sub for DPositive { /// Subtracts, into the finite domain. /// - /// The difference of two positive finite values is finite, with no re-validation: its - /// magnitude never exceeds the larger operand, so the subtraction cannot overflow. Equal - /// operands give `+0.0`. + /// The difference of two positive finite values has a magnitude that never exceeds the larger + /// operand. The subtraction cannot overflow and requires no re-validation. Equal operands give + /// `+0.0`. #[inline] fn sub(self, rhs: Self) -> DFinite { DFinite::new_unchecked(self.0 - rhs.0) @@ -300,10 +302,8 @@ const impl core::ops::Mul for DPositive { /// Scales by a positive fraction. /// - /// The product of a positive value and a fraction in `(0, 1]` is positive, at most the - /// value, and never NaN, so overflow cannot occur. Underflow escapes to zero - a wrong - /// reading rather than a soundness break, since no unsafe code trusts the domain - and - /// asserts in debug builds through the constructor. + /// The rounded product must remain positive. For in-domain operands it cannot exceed the + /// positive value or become NaN, but underflow can round it to zero. #[inline] fn mul(self, rhs: PositiveUnitFraction) -> Self { Self::new_unchecked(self.0 * rhs.get()) @@ -315,10 +315,8 @@ const impl core::ops::Mul for OpenUnitFraction { /// Scales a positive value toward zero. /// - /// The product of a positive value and a fraction in `(0, 1)` is less than the value and - /// never NaN, so overflow cannot occur. Underflow escapes to zero - a wrong reading rather - /// than a soundness break, since no unsafe code trusts the domain - and asserts in debug - /// builds through the constructor. + /// The rounded product must remain positive. For in-domain operands it cannot exceed the + /// positive value or become NaN. Rounding can leave the value unchanged or underflow to zero. #[inline] fn mul(self, rhs: DPositive) -> DPositive { DPositive::new_unchecked(self.get() * rhs.0) @@ -335,7 +333,6 @@ const impl core::ops::Mul for DPositive { } const impl From for f64 { - /// Reads the value, exactly. #[inline] fn from(value: DPositive) -> Self { value.0 @@ -345,10 +342,6 @@ const impl From for f64 { const impl core::ops::Add for DPositive { type Output = f64; - /// Adds a raw offset. - /// - /// The raw operand is arbitrary, so the sum can leave any bounded domain and returns a raw - /// float. #[inline] fn add(self, rhs: f64) -> f64 { self.0 + rhs @@ -356,7 +349,6 @@ const impl core::ops::Add for DPositive { } const impl PartialEq for DPositive { - /// Compares across the scalar family, in one precision with no widening. #[inline] fn eq(&self, other: &DNonNegative) -> bool { self.0 == other.get() @@ -364,7 +356,6 @@ const impl PartialEq for DPositive { } const impl PartialOrd for DPositive { - /// Orders across the scalar family, in one precision with no widening. #[inline] fn partial_cmp(&self, other: &DNonNegative) -> Option { self.0.partial_cmp(&other.get()) @@ -372,7 +363,6 @@ const impl PartialOrd for DPositive { } const impl PartialEq for DPositive { - /// Compares across the scalar family, in one precision with no widening. #[inline] fn eq(&self, other: &OpenUnitFraction) -> bool { self.0 == other.get() @@ -380,7 +370,6 @@ const impl PartialEq for DPositive { } const impl PartialOrd for DPositive { - /// Orders across the scalar family, in one precision with no widening. #[inline] fn partial_cmp(&self, other: &OpenUnitFraction) -> Option { self.0.partial_cmp(&other.get()) @@ -390,7 +379,6 @@ const impl PartialOrd for DPositive { const impl TryFrom for DPositive { type Error = NotPositive; - /// Narrows from the enclosing domain, refusing exactly zero. #[inline] fn try_from(value: DNonNegative) -> Result { let value = value.get(); @@ -399,9 +387,6 @@ const impl TryFrom for DPositive { } const impl From for DPositive { - /// Widens into double precision, exactly. - /// - /// Every positive `f32` denotes a positive, finite `f64`, so no re-validation happens. #[inline] fn from(value: Positive) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. @@ -414,7 +399,6 @@ impl proptest::arbitrary::Arbitrary for DPositive { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole domain, subnormals included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -425,14 +409,12 @@ impl proptest::arbitrary::Arbitrary for DPositive { } impl serde::Serialize for DPositive { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f64(self.0) } } impl<'de> serde::Deserialize<'de> for DPositive { - /// Deserializes a plain number, refusing values outside the finite strictly positive range. fn deserialize>(deserializer: D) -> Result { let value = f64::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { diff --git a/libs/@local/graph/atlas/src/math/scalar/finite.rs b/libs/@local/graph/atlas/src/math/scalar/finite.rs index d47ae6ef059..4ebda25cc74 100644 --- a/libs/@local/graph/atlas/src/math/scalar/finite.rs +++ b/libs/@local/graph/atlas/src/math/scalar/finite.rs @@ -1,24 +1,20 @@ //! Finiteness-only guards that carry their domain in the type. //! //! [`Finite`] holds a finite `f32` and [`DFinite`] a finite `f64`, for quantities whose contract -//! is that they denote a real number and nothing further. Validation happens once, at -//! construction - refusal ([`new`](DFinite::new)), a caller's proof -//! ([`new_unchecked`](DFinite::new_unchecked)), or a widening conversion from a narrower domain - -//! so a value that exists is finite and consuming code trusts the domain instead of re-checking -//! it. +//! is that they denote a real number and nothing further. Checked constructors such as +//! [`DFinite::new`] validate raw inputs. Unchecked constructors such as +//! [`DFinite::new_unchecked`] require a caller's proof, while widening conversions preserve a +//! narrower source domain's finiteness. //! -//! Comparing, sorting and hashing a [`DFinite`] need no NaN case: [`Eq`], [`Ord`] and [`Hash`] -//! are total, agree with one another, and follow the IEEE total order restricted to the finite -//! values. Both zeros are admitted and keep their sign bit, so under that order `-0.0` and -//! `+0.0` are distinct readings with `-0.0 < +0.0`, and a reading round-trips bit for bit. +//! Comparing, sorting and hashing a [`DFinite`] need no NaN case: [`Eq`], [`Ord`] and [`Hash`] are +//! total, agree with one another, and follow the IEEE total order restricted to finite values. Both +//! zeros are admitted with their sign bits intact. Under this order, `-0.0` and `+0.0` are distinct +//! readings with `-0.0 < +0.0`. A reading round-trips bit for bit. //! -//! Arithmetic whose result provably stays finite stays in the type - negation, integer -//! conversion - with no run-time re-check. An operation that can leave the domain returns a -//! raw float instead, and the caller re-enters through a constructor at whichever boundary -//! proves the bound. The exception is the in-family sum, which stays in the type as the -//! family's folds do: overflow escapes to `±∞` and asserts in debug builds. Serialization writes -//! plain numbers and deserialization re-validates, so a persisted value is as trustworthy as a -//! constructed one. +//! Negation and integer conversion preserve finiteness. Other operations may return a raw +//! float or [`Derivation`] for later validation. The `DFinite` sum and difference return the +//! domain type and require a finite rounded result. Serialization writes plain numbers and +//! deserialization validates them again. use core::{ cmp::Ordering, @@ -34,8 +30,8 @@ use crate::math::Derivation; /// Validates a finite literal at compile time. /// -/// The expansion is a `const` block over [`Finite::new`], so a literal outside the domain fails the -/// build instead of a test run. Runtime values keep the checked constructor. +/// A `const` block validates the literal with [`Finite::new`] during compilation. A literal outside +/// the domain fails the build. Runtime values use the checked constructor. #[cfg(test)] macro_rules! finite { ($value:expr) => { @@ -47,8 +43,8 @@ pub(crate) use finite; /// Validates a finite double-precision literal at compile time. /// -/// The expansion is a `const` block over [`DFinite::new`], so a literal outside the domain fails -/// the build instead of a test run. Runtime values keep the checked constructor. +/// A `const` block validates the literal with [`DFinite::new`] during compilation. A literal +/// outside the domain fails the build. Runtime values use the checked constructor. macro_rules! d_finite { ($value:expr) => { const { $crate::math::DFinite::new($value).expect("the literal is finite") } @@ -58,18 +54,21 @@ pub(crate) use d_finite; /// A finite `f32`, valid by construction. /// -/// A value that exists is finite, so the domain check lives at the constructor and nowhere -/// else. Quantities with a sign or interval bound on top of finiteness take the narrower -/// [`Positive`], [`NonNegative`], or [`UnitFraction`](super::UnitFraction) instead, which -/// states that bound in the same place. +/// Finiteness is the construction invariant. Quantities with an additional sign or interval bound +/// take the narrower [`Positive`], [`NonNegative`], or [`UnitFraction`](super::UnitFraction), which +/// states that bound too. /// -/// Both zeros are admitted and keep their sign bit. The value serializes as a plain number, so -/// a format whose number grammar covers exactly the finite values represents every inhabitant -/// of this type and reads it back through the same validation. +/// Both zeros are admitted with their sign bits intact. Serialization writes plain numbers. A +/// format whose number grammar covers exactly the finite values represents every inhabitant of this +/// type and reads it back through the same validation. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{Finite}; +/// /// assert_eq!(Finite::new(-2.5).expect("-2.5 is finite").get(), -2.5); /// assert_eq!( /// Finite::new(f32::MIN).expect("the minimum is finite").get(), @@ -106,15 +105,13 @@ impl Finite { /// Returns whether `value`'s exact bits are a stored finite value. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: accepted - /// values store bit for bit, both zeros included, so the bits are valid exactly when - /// [`new`](Self::new) accepts the value. + /// Accepted values retain their bits, including both zeros. Persisted bits are valid exactly + /// when [`new`](Self::new) accepts the corresponding value. #[inline] #[must_use] pub(crate) const fn is_canonical(value: f32) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. + // compare with the constructed value to account for normalization Some(accepted) => accepted.0.to_bits() == value.to_bits(), None => false, } @@ -165,7 +162,6 @@ impl fmt::Display for Finite { } const impl From for Finite { - /// Widens into the enclosing domain: every positive value is finite. #[inline] fn from(value: Positive) -> Self { Self(value.get()) @@ -173,7 +169,6 @@ const impl From for Finite { } const impl From for Finite { - /// Widens into the enclosing domain: every non-negative value is finite. #[inline] fn from(value: NonNegative) -> Self { Self(value.get()) @@ -181,7 +176,6 @@ const impl From for Finite { } const impl From for f64 { - /// Widens into double precision, exactly. #[inline] fn from(value: Finite) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. @@ -192,12 +186,7 @@ const impl From for f64 { const impl core::ops::Div for Finite { type Output = f32; - /// Divides by a positive divisor, which is never zero. - /// - /// The result is a raw float: the quotient of a finite value by a small positive one can - /// overflow to either infinity, following the numerator's sign. It is never NaN: that - /// would take a zero or infinite operand, and both domains exclude them. The caller - /// re-enters a domain at whichever boundary proves its bound. + /// Divides by a nonzero divisor with deferred validation of overflow. #[inline] fn div(self, rhs: Positive) -> f32 { self.0 / rhs.get() @@ -209,7 +198,6 @@ impl proptest::arbitrary::Arbitrary for Finite { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole domain, both signs and subnormals included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -220,14 +208,12 @@ impl proptest::arbitrary::Arbitrary for Finite { } impl serde::Serialize for Finite { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f32(self.0) } } impl<'de> serde::Deserialize<'de> for Finite { - /// Deserializes a plain number, refusing NaN and the infinities. fn deserialize>(deserializer: D) -> Result { let value = f32::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { @@ -245,18 +231,22 @@ impl<'de> serde::Deserialize<'de> for Finite { /// real number and nothing further. A quantity that also has a sign or interval bound carries /// the narrower [`DPositive`], [`DNonNegative`], or [`UnitFraction`](super::UnitFraction). /// -/// Both zeros are admitted and keep their sign bit. The value serializes as a plain number, so -/// a format whose number grammar covers exactly the finite values represents every inhabitant -/// of this type and reads it back through the same validation. +/// Both zeros are admitted with their sign bits intact. Serialization writes plain numbers. A +/// format whose number grammar covers exactly the finite values represents every inhabitant of this +/// type and reads it back through the same validation. /// /// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow the IEEE total /// order restricted to the finite values. Values sort and key ordered maps like the numbers /// they hold, with one caveat the sign bit brings: `-0.0` and `+0.0` are distinct, and /// `-0.0 < +0.0`. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{DFinite}; +/// /// assert_eq!( /// DFinite::new(-1.0e-300) /// .expect("a tiny negative is finite") @@ -294,15 +284,13 @@ impl DFinite { /// Returns whether `value`'s exact bits are a stored finite value. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: accepted - /// values store bit for bit, both zeros included, so the bits are valid exactly when - /// [`new`](Self::new) accepts the value. + /// Accepted values retain their bits, including both zeros. Persisted bits are valid exactly + /// when [`new`](Self::new) accepts the corresponding value. #[inline] #[must_use] pub(crate) const fn is_canonical(value: f64) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. + // compare with the constructed value to account for normalization Some(accepted) => accepted.0.to_bits() == value.to_bits(), None => false, } @@ -313,9 +301,13 @@ impl DFinite { /// The sign bit of a promised zero is kept. Where the proof is not immediate, /// [`new`](Self::new) checks instead. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{DFinite}; + /// /// let span = DFinite::new(3.0).expect("3.0 is finite"); /// // A mean of finite values bounded far inside the exponent range cannot overflow. /// let mean = DFinite::new_unchecked((span.get() + span.get()) / 2.0); @@ -350,8 +342,7 @@ impl DFinite { /// Narrows to the strictly positive domain. /// - /// Returns [`None`] at zero and below, so an `if let` on the result is the sign guard and - /// the positivity witness in one move. + /// Returns [`None`] at zero and below. #[inline] #[must_use] pub(crate) const fn positive(self) -> Option { @@ -373,9 +364,9 @@ impl DFinite { /// Returns the total-order key: a bit pattern monotone in the value. /// - /// Flipping a negative value's bits and setting a non-negative value's sign bit maps the - /// IEEE ordering onto unsigned integer order, so one integer compare decides every pair - /// with no NaN branch. + /// Flipping a negative value's bits and setting a nonnegative value's sign bit maps IEEE + /// ordering onto unsigned integer order for one integer comparison of every pair, without a NaN + /// branch. #[inline] const fn order_key(self) -> u64 { let bits = self.0.to_bits(); @@ -407,7 +398,6 @@ const impl From for f64 { } const impl From for DFinite { - /// Widens into the enclosing domain: every positive value is finite. #[inline] fn from(value: DPositive) -> Self { Self(value.get()) @@ -415,7 +405,6 @@ const impl From for DFinite { } const impl From for DFinite { - /// Widens into the enclosing domain: every non-negative value is finite. #[inline] fn from(value: DNonNegative) -> Self { Self(value.get()) @@ -441,7 +430,7 @@ const impl From for DFinite { const impl PartialEq for DFinite { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per finite value, so bit equality is total-order equality + // unique finite total-order encodings make bit equality equivalent to total-order equality self.0.to_bits() == other.0.to_bits() } } @@ -465,7 +454,7 @@ const impl Ord for DFinite { impl Hash for DFinite { #[inline] fn hash(&self, state: &mut H) { - // one bit pattern per finite value, so `Hash` agrees with `Eq` + // hashing each finite total-order value's unique bit pattern preserves agreement with `Eq` state.write_u64(self.0.to_bits()); } } @@ -475,8 +464,7 @@ const impl core::ops::Add for DFinite { /// Adds. /// - /// Overflow escapes to `±∞` - a wrong reading rather than a soundness break, since no - /// unsafe code trusts the domain - and asserts in debug builds through the constructor. + /// The rounded sum must remain finite. Same-sign operands can overflow to infinity. #[inline] fn add(self, rhs: DFinite) -> DFinite { DFinite::new_unchecked(self.0 + rhs.0) @@ -495,8 +483,7 @@ const impl core::ops::Sub for DFinite { /// Subtracts. /// - /// Overflow escapes to `±∞` - a wrong reading rather than a soundness break, since no - /// unsafe code trusts the domain - and asserts in debug builds through the constructor. + /// The rounded difference must remain finite. Opposite-sign operands can overflow to infinity. #[inline] fn sub(self, rhs: DFinite) -> DFinite { DFinite::new_unchecked(self.0 - rhs.0) @@ -520,10 +507,10 @@ const impl core::ops::Div for DFinite { /// Divides by a positive divisor, which is never zero. /// - /// The quotient's exponent is the numerator's minus the divisor's, so a finite value over - /// a small positive one can overflow to either infinity, following the numerator's sign. - /// The fat exit enters the derivation, which claims finiteness at its finish. It is never NaN: - /// that would take a zero or infinite operand, and both domains exclude them. + /// Dividing a finite value by a small positive one can overflow to either infinity, following + /// the numerator's sign. The derivation validates finiteness at its finish. The quotient is + /// never NaN for in-domain operands: both are finite and the divisor is nonzero. A zero + /// numerator is valid. #[inline] fn div(self, rhs: DPositive) -> Derivation { Derivation::raw(self.0 / rhs.get()) @@ -543,7 +530,6 @@ where } const impl PartialEq for DFinite { - /// Compares across the scalar family, in one precision with no widening. #[inline] fn eq(&self, other: &OpenUnitFraction) -> bool { self.0 == other.get() @@ -551,7 +537,6 @@ const impl PartialEq for DFinite { } const impl PartialOrd for DFinite { - /// Orders across the scalar family, in one precision with no widening. #[inline] fn partial_cmp(&self, other: &OpenUnitFraction) -> Option { self.0.partial_cmp(&other.get()) @@ -563,7 +548,6 @@ impl proptest::arbitrary::Arbitrary for DFinite { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole domain, both signs and subnormals included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -574,14 +558,12 @@ impl proptest::arbitrary::Arbitrary for DFinite { } impl serde::Serialize for DFinite { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f64(self.0) } } impl<'de> serde::Deserialize<'de> for DFinite { - /// Deserializes a plain number, refusing NaN and the infinities. fn deserialize>(deserializer: D) -> Result { let value = f64::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { diff --git a/libs/@local/graph/atlas/src/math/scalar/log2.rs b/libs/@local/graph/atlas/src/math/scalar/log2.rs index 937648e816c..871605d7471 100644 --- a/libs/@local/graph/atlas/src/math/scalar/log2.rs +++ b/libs/@local/graph/atlas/src/math/scalar/log2.rs @@ -4,13 +4,17 @@ use super::unsafe_impl_try_from_bytes; /// A power-of-two exponent below the `u64` shift width, valid by construction. /// -/// Configuration fields named `*_log2` carry this type instead of a raw `u8`. Shifting a `u64` by -/// 64 or more panics in debug builds and masks in release, so the constructor checks the bound and -/// the shifting site validates nothing. +/// Values lie in `0..64`, the valid shift counts for a `u64`. A power of two computed as +/// `1_u64 << exponent.get()` therefore fits in that type. A narrower integer needs its own +/// shift-width bound. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{Log2}; +/// /// let span = Log2::new(6).expect("6 lies below the shift width"); /// assert_eq!(span.get(), 6); /// assert_eq!(1_u64 << span.get(), 64); @@ -25,8 +29,7 @@ pub(crate) struct Log2(u8); impl Log2 { /// Validates a shift exponent. /// - /// Returns [`None`] unless the value lies below the `u64` shift width: 64 and above have no - /// in-range power of two, and shifting by them panics in debug builds and masks in release. + /// Returns [`None`] unless the value is below 64, the `u64` shift width. #[inline] #[must_use] pub(crate) const fn new(value: u8) -> Option { @@ -39,14 +42,13 @@ impl Log2 { /// Returns whether `value`'s exact bits are a stored exponent. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: integers store - /// bit for bit, so the bits are valid exactly when the exponent lies below the shift width. + /// The constructor [`new`](Self::new) preserves integer bits. Persisted values are valid + /// exactly when the exponent lies below the `u64` shift width. #[inline] #[must_use] pub(crate) const fn is_canonical(value: u8) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. + // compare with the constructed value to account for normalization Some(accepted) => accepted.0 == value, None => false, } diff --git a/libs/@local/graph/atlas/src/math/scalar/mod.rs b/libs/@local/graph/atlas/src/math/scalar/mod.rs index de0e87459b1..c9a31efe26b 100644 --- a/libs/@local/graph/atlas/src/math/scalar/mod.rs +++ b/libs/@local/graph/atlas/src/math/scalar/mod.rs @@ -1,68 +1,51 @@ -//! Numerically stable scalar special functions with validated domains. +//! Validated scalar domains, special functions and checked precision conversions. //! -//! Checked float narrowing rides beside them. -//! -//! These formalize numeric patterns that recur across the crate: a stable softplus and the Huber -//! penalty for layout losses, checked `f64` to `f32` narrowing for persisted coordinates, the -//! unit-interval fractions [`UnitFraction`], [`PositiveUnitFraction`] and [`OpenUnitFraction`], -//! and the finiteness-only -//! guards [`Finite`] and [`DFinite`] for signed quantities, all validated once at construction and -//! carried by configuration knobs and measured shares alike. Vector reductions such as softmax and -//! log-sum-exp live on [`DVecN`](super::DVecN). +//! The scalar types express finiteness, sign and interval requirements. Their checked +//! constructors reject out-of-domain values. Unchecked constructors require the same facts from +//! you. [`softplus`], the [`Huber penalty`](NonNegative::huber) and the +//! [`logistic function`](NonNegative::sigmoid) supply stable forms for scalar losses and +//! probabilities. Vector reductions such as softmax and log-sum-exp live on +//! [`DVecN`](super::DVecN). //! //! # Precision ownership //! -//! The family carries every domain in two widths, and which width a value takes is decided by the -//! stage that owns it rather than field by field. -//! -//! The tensor stage computes in `f32`, so every reading born there is `f32` at birth: frozen -//! model state, band radii, activations, coefficients. No value derived from such a reading can -//! claim more accuracy than its birth allows. Derivation owns `f64`: a computation over readings -//! widens exactly at its first touch of an `f32`-born value and accumulates in double width for -//! headroom. It never narrows mid-expression. +//! [`Finite`], [`Positive`] and [`NonNegative`] use `f32`. Their `D`-prefixed counterparts use +//! `f64` for calculations that need more exponent range or significand precision. Widening an +//! `f32` value is exact. It cannot recover information already lost, but it avoids further +//! single-precision rounding in an accumulation. Keep the wider result until a storage or +//! computation boundary requires narrowing. //! -//! Storage width follows the value's real accuracy. A summary a derivation computes once, such as -//! a receipt, a certificate or a report statistic, keeps double width in the `D`-prefixed types, -//! because narrowing a singleton loses bits with no volume to pay for them. A bulk per-item -//! reading, such as an estimand vector or a step contribution, travels as `f32`, because its -//! accuracy is already `f32`-bounded at birth and the item count pays for the width. A mean over n -//! readings gains accuracy like √n, so a per-item average earns double width where a per-item -//! reading does not. +//! Fractions use `f64` to retain distinctions near the interval endpoints. The nearest `f32` +//! below one is 2⁻²⁴ away, and `1 − 10⁻⁹` rounds to one in that precision. Count ratios use +//! [`UnitFraction::ratio`], which documents both the exact-count range and the additional +//! rounding above 2⁵³. A wider representation does not itself establish the statistical +//! accuracy of a derived quantity. //! -//! Fractions always store double width. The unit interval needs no range headroom, so the extra -//! width spends entirely on precision, and fraction meaning concentrates at the endpoints, where -//! the nearest `f32` below `1.0` sits `6e-8` away and a fraction such as `1 - 1e-9` rounds to -//! exactly `1.0`. A ratio of exact counts carries full double-width accuracy besides. The family -//! therefore has no `f32` fraction type. +//! # Representation and conversion //! -//! Narrowing is lawful at the tensor entry and at the writing of a bulk record, and nowhere else. -//! `narrow_lossy` names those sites in code. A narrow followed by a widen loses bits that no -//! destination asked for, and is a defect wherever it appears. +//! Scalar [`Serialize`](serde::Serialize) implementations write the underlying primitive number. +//! [`Deserialize`](serde::Deserialize) applies the checked constructor after reading that number, +//! rejecting out-of-domain values and performing the constructor's zero normalization. Infallible +//! conversions preserve the numeric value except for integer-to-float rounding where documented. //! //! # Operator edge contracts //! -//! An operator impl on a domain scalar states an output type and edge behaviour derivable from -//! its operand types alone, never from a consumer: a contract that needs a caller to justify it -//! is not a contract. The discriminant is the measure of the exit set in exponent space, under -//! the maximally pessimistic log-uniform model over `f64`'s 2046 exponents. A sum's result -//! exponent is at most one above its larger operand, so its exit is one shell at the top of the -//! range - about 0.1% of input pairs, unreachable for physically-born readings without an -//! upstream defect. A product's exponents add, so its exit is two corner triangles of the -//! exponent plane, roughly 25% together, and `1e200 · 1e200` overflows with neither operand -//! extreme. A same-sign difference never exceeds its larger operand, exit measure zero, and a -//! contraction by a unit fraction exits only through the bottom shell. Real distributions are -//! tighter than log-uniform, and the shape of the asymmetry - one shell against a fat region - -//! is distribution-independent. +//! Some operations preserve their result domain for every valid operand. A difference of two +//! finite non-negative values has magnitude at most the larger operand and remains finite. +//! A sum can overflow, and a product or quotient can also underflow. Multiplication by a +//! positive unit fraction prevents overflow but can still round a positive result to zero. +//! These range facts do not quantify probabilities of failure for a workload. //! -//! | Operands | Exit measure | Form | -//! |---|---|---| -//! | A landing a theorem names totally | zero | the landing's type, no check | -//! | A sum's overflow, a contraction's underflow | one shell, ~0.1% | the typed escape, whose debug assert is a true structural claim | -//! | A product or quotient of two unbounded operands | fat, ~25% | no operator: `checked_*` at a one-shot site, a [`Derivation`](super::Derivation) for a fold, `new_unchecked` beside a site's own theorem | -//! | A raw operand or a sign crossing | - | the raw carrier, which claims nothing and cannot lie | +//! | Result form | Validation boundary | +//! |---|---| +//! | A domain scalar | The operation is closed over its operands, or its documented result-range requirement must hold. | +//! | `Option` from `checked_*` | The operation rejects a rounded result outside the domain. | +//! | [`Derivation`](super::Derivation) | Arithmetic keeps the raw value until [`finish`](super::Derivation::finish) validates it. | +//! | A raw float | The operation makes no domain claim. | //! -//! A derivation exits once, at its destination boundary. An exit followed immediately by more -//! arithmetic means the vocabulary one step earlier was wrong. +//! Choose a checked operation or a derivation when the rounded result's membership is not +//! established. A derivation may finish at an intermediate domain boundary before a new +//! calculation begins. mod d_non_negative; mod d_positive; mod finite; @@ -97,9 +80,9 @@ pub(crate) use unit_fraction::{UnitFraction, unit_fraction}; /// Validates a nonzero literal at compile time. /// -/// The expansion is a `const` block over [`NonZero::new`](core::num::NonZero::new), so a zero -/// literal fails the build instead of a test run. The expansion names the type by its full path, -/// so the calling scope imports nothing. +/// A `const` block validates the literal with [`NonZero::new`](core::num::NonZero::new) during +/// compilation. A zero literal fails the build. The expansion's fully qualified type name needs no +/// import in the calling scope. macro_rules! nz { ($value:expr) => { const { ::core::num::NonZero::new($value).expect("the literal is nonzero") } @@ -109,25 +92,21 @@ pub(crate) use nz; /// Implements [`zerocopy::TryFromBytes`] for `repr(transparent)` scalar newtypes. /// -/// A derive states a type's bit validity as its fields' bit validity, and a raw primitive field -/// admits every bit pattern, so a domain-validated newtype writes the impl by hand. The shape is -/// zerocopy's own `NonZero` impl: reinterpret the candidate as an unaligned primitive and copy -/// the value out, accepting exactly the bit patterns the type's `is_canonical` accepts. This is -/// what lets a mapped byte region serve typed rows, the domain validated as the bytes are read. +/// Accepts a comma-separated list of `Type[Primitive]` pairs, with an optional trailing comma. Each +/// generated validity check copies the stored primitive and accepts it exactly when +/// `Type::is_canonical` returns `true`. /// -/// Soundness rides on two facts per listed type. The layout claim, `repr(transparent)` over the -/// named primitive, is checked at compile time: the exact-size cast refuses a size mismatch. The -/// domain claim is `is_canonical`'s contract - `true` exactly for the bit patterns the validating -/// constructors can store - which its construct-and-compare body takes from the constructor -/// itself. +/// # Safety +/// +/// Each type must be `repr(transparent)` over the named primitive, whose initialized bit patterns +/// are all valid. Its `is_canonical` method must accept only representations satisfying all of the +/// type's invariants. The exact-size cast checks size equality, not these representation and +/// validity obligations. macro_rules! unsafe_impl_try_from_bytes { ($($ty:ident[$prim:ty]),* $(,)?) => { $( - // SAFETY: `is_bit_valid` returns `true` exactly when the candidate's bytes hold a - // valid value. The type is `repr(transparent)` over a primitive that every - // initialized byte pattern inhabits, so the read below yields the stored value, and - // `is_canonical` accepts exactly the bit patterns the validating constructors can - // store. + // zerocopy reserves TryFromBytes for its derive and excludes hidden validation APIs from compatibility guarantees. Recheck this implementation when updating the dependency. + // SAFETY: TryFromBytes requires every accepted candidate to be a valid value. The macro's invocation contract establishes transparent primitive storage and a predicate that accepts only valid representations. The exact-size unaligned read preserves those bits, and the predicate validates them. Every accepted candidate therefore satisfies the type's invariants. unsafe impl zerocopy::TryFromBytes for $ty { fn only_derive_is_allowed_to_implement_this_trait() {} @@ -155,10 +134,12 @@ pub(crate) use unsafe_impl_try_from_bytes; /// Implements comparisons and scaling against a type's raw primitive. /// -/// The generated comparisons are numeric rather than bitwise: the raw side may be `-0.0` or NaN, -/// and the comparison follows IEEE semantics for both, so a typed value never equals NaN. The -/// product with an arbitrary raw operand can leave the domain, so multiplication in either order -/// returns the raw primitive. A raw accumulator takes `*=` by the typed factor. +/// Accepts a comma-separated list of `Type[Primitive]` pairs, with an optional trailing comma. Each +/// type must have its primitive value in field `0`. +/// +/// Under IEEE semantics, a typed value never equals NaN, and signed zeros compare equal. +/// Multiplication by a raw operand returns the raw primitive in either order because its result can +/// leave the type's domain. A raw accumulator also supports `*=` by the typed factor. macro_rules! raw_interop { ($($ty:ident[$prim:ty]),* $(,)?) => { $( @@ -236,10 +217,6 @@ macro_rules! raw_interop { const impl core::ops::Mul<$prim> for $ty { type Output = $prim; - /// Scales a raw float. - /// - /// The raw operand is arbitrary, so the product can leave the domain and returns - /// a raw float. #[inline] fn mul(self, rhs: $prim) -> $prim { self.0 * rhs @@ -256,7 +233,6 @@ macro_rules! raw_interop { } const impl core::ops::MulAssign<$ty> for $prim { - /// Scales a raw float in place. #[inline] fn mul_assign(&mut self, rhs: $ty) { *self *= rhs.0; @@ -274,10 +250,14 @@ pub(crate) use raw_interop; /// for large positive inputs and `0` for large negative inputs. The output is non-negative and /// satisfies `softplus(value) - softplus(-value) == value` up to rounding. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore -/// // A naive `exp(50.0)` loses the asymptote. The stable form is exact. +/// use crate::math::{softplus}; +/// +/// // At 50 the correction is smaller than one f32 rounding step. /// assert_eq!(softplus(50.0), 50.0); /// assert!(softplus(-50.0) < 1e-20); /// assert!((softplus(0.0) - core::f32::consts::LN_2).abs() < 1e-7); @@ -290,13 +270,17 @@ pub(crate) fn softplus(value: f32) -> f32 { /// Narrows an `f64` to `f32`, permitting rounding. /// -/// This converts with round-to-nearest and returns the result whenever it is finite. Inputs whose -/// magnitude exceeds the `f32` range, the infinities, and NaN all yield [`None`]. Negative zero -/// narrows to negative zero, preserving the sign bit. +/// Converts with round-to-nearest and returns the result whenever it is finite. Returns [`None`] +/// for NaN and for inputs that round to either infinity. An input slightly beyond [`f32::MAX`] +/// can still round to that finite endpoint. Negative zero keeps its sign bit. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{narrow_f32}; +/// /// // Rounds to the nearest `f32`. /// assert_eq!(narrow_f32(0.1), Some(0.1_f32)); /// // Beyond the `f32` range. diff --git a/libs/@local/graph/atlas/src/math/scalar/negative.rs b/libs/@local/graph/atlas/src/math/scalar/negative.rs index 47c8ab70faa..baaa5c80dbd 100644 --- a/libs/@local/graph/atlas/src/math/scalar/negative.rs +++ b/libs/@local/graph/atlas/src/math/scalar/negative.rs @@ -10,22 +10,23 @@ use super::raw_interop; /// A finite, strictly negative `f32`, valid by construction. /// -/// The sign mirror of [`Positive`](super::Positive), for readings whose sign is part of the -/// contract: a slope that rewards rather than corrects carries its direction in the type, and the -/// consuming site validates nothing. +/// Use [`Positive`](super::Positive) for the positive domain. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore -/// assert_eq!(Negative::new(-2.5).expect("-2.5 is negative").get(), -2.5); +/// use crate::math::{Negative}; +/// +/// assert_eq!(f64::from(Negative::new(-2.5).expect("-2.5 is negative")), -2.5); /// assert_eq!(Negative::new(0.0), None); /// assert_eq!(Negative::new(-0.0), None); /// assert_eq!(Negative::new(f32::NAN), None); /// ``` /// /// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value. -/// The domain excludes NaN and both zeros, so every value owns one bit pattern with no -/// canonicalization step. +/// Excluding NaN and both zeros gives every value one bit pattern without canonicalization. #[derive(Copy, Clone)] #[repr(transparent)] pub(crate) struct Negative(f32); @@ -63,7 +64,7 @@ impl Negative { const impl PartialEq for Negative { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -80,8 +81,8 @@ const impl PartialOrd for Negative { const impl Ord for Negative { #[inline] fn cmp(&self, other: &Self) -> Ordering { - // For negative floats the bit pattern is monotone in the magnitude, so the value order - // is the bit order reversed: still a GPR compare with no NaN branch and no panic path. + // Negative finite floats have unsigned bit patterns increasing with magnitude. Reversing + // the bit comparison gives numeric order. other.0.to_bits().cmp(&self.0.to_bits()) } } @@ -89,7 +90,7 @@ const impl Ord for Negative { impl Hash for Negative { #[inline] fn hash(&self, state: &mut H) { - // one bit pattern per value, so `Hash` agrees with `Eq` + // hashing the unique representation agrees with numeric equality state.write_u32(self.0.to_bits()); } } @@ -107,7 +108,6 @@ impl fmt::Display for Negative { } const impl From for f64 { - /// Widens into double precision, exactly. #[inline] fn from(value: Negative) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. diff --git a/libs/@local/graph/atlas/src/math/scalar/non_negative.rs b/libs/@local/graph/atlas/src/math/scalar/non_negative.rs index 7278a28f213..aa5bc4d83ce 100644 --- a/libs/@local/graph/atlas/src/math/scalar/non_negative.rs +++ b/libs/@local/graph/atlas/src/math/scalar/non_negative.rs @@ -10,8 +10,7 @@ use super::{DNonNegative, Finite, Positive, raw_interop, unsafe_impl_try_from_by /// Validates a non-negative literal at compile time. /// -/// The expansion is a `const` block over [`NonNegative::new`], so a literal outside the domain -/// fails the build instead of a test run. Runtime values keep the checked constructor. +/// A literal outside the domain fails the build. Use [`NonNegative::new`] to check runtime values. macro_rules! non_negative { ($value:expr) => { const { @@ -23,17 +22,20 @@ pub(crate) use non_negative; /// A finite, non-negative `f32`, valid by construction. /// -/// The shared definition of the finite-and-non-negative check. Zero passes, so the type carries -/// magnitudes and weights that may legitimately switch a term off. +/// Admitting zero allows magnitudes and weights that switch a term off. /// /// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value, /// with `-0.0` and `+0.0` the same value: construction canonicalizes the sign of zero. Values /// sort and key ordered maps like the numbers they hold, with no NaN case, and /// [`to_bits`](Self::to_bits) is an identity: one bit pattern per value. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{NonNegative}; +/// /// assert_eq!(NonNegative::new(0.0).expect("zero is admitted").get(), 0.0); /// assert_eq!(NonNegative::new(-0.5), None); /// assert_eq!(NonNegative::new(f32::INFINITY), None); @@ -87,8 +89,8 @@ impl NonNegative { /// Returns the square of a raw scalar. /// - /// A square is never negative and never NaN for non-NaN arguments. Overflow escapes to `+∞` - /// and asserts in debug builds, mirroring integer `+`. + /// The rounded square must be finite. For a finite argument it is non-negative, but a + /// sufficiently large magnitude can overflow. #[inline] #[must_use] pub(crate) const fn square(value: f32) -> Self { @@ -98,6 +100,11 @@ impl NonNegative { Self(squared) } + /// Widens to double precision, exactly. + /// + /// Every finite `f32` is exactly representable as `f64`. Widening preserves non-negativity and + /// maps the canonical `+0.0` to `+0.0`. Therefore the widened value is the same real number + /// with no rounding and no re-validation. #[inline] #[must_use] pub(crate) const fn widen(self) -> DNonNegative { @@ -106,9 +113,10 @@ impl NonNegative { /// Squares into double precision, exactly and totally. /// - /// A 24-bit significand squares within 53 bits, so the widened square carries no rounding. - /// A doubled `f32` exponent sits far inside the `f64` range, so the square never leaves - /// the domain. The square of zero is zero. + /// Squaring a 24-bit significand needs at most 48 bits, within `f64`'s 53-bit precision. Every + /// nonzero finite `f32` square lies in [2⁻²⁹⁸, 2²⁵⁶), inside the normal `f64` range. Widening + /// before multiplication therefore gives an exact square that never leaves the domain. The + /// square of zero is zero. #[inline] #[must_use] pub(crate) const fn square_wide(self) -> DNonNegative { @@ -117,10 +125,9 @@ impl NonNegative { /// Divides into double precision, totally. /// - /// The quotient of an `f32`-born non-negative by an `f32`-born positive is never NaN and - /// never negative, and its exponent stays hundreds of shells inside the `f64` range in - /// both directions, so the quotient never leaves the domain. Zero divides to exactly zero. - /// One rounding. + /// Positive finite `f32` values lie in [2⁻¹⁴⁹, 2¹²⁸). Their positive quotients lie between + /// 2⁻²⁷⁷ and 2²⁷⁷, inside the normal `f64` range. Widening before division therefore gives + /// a finite non-negative result with one rounding. A zero numerator gives exactly zero. #[inline] #[must_use] pub(crate) const fn div_wide(self, rhs: Positive) -> DNonNegative { @@ -129,8 +136,7 @@ impl NonNegative { /// Narrows to the strictly positive domain. /// - /// Returns [`None`] exactly at zero, so an `if let` on the result is the zero guard and the - /// positivity witness in one move. + /// Returns [`None`] exactly at zero. #[inline] #[must_use] pub(crate) const fn positive(self) -> Option { @@ -146,8 +152,8 @@ impl NonNegative { /// Returns the canonical bit pattern. /// - /// Construction canonicalizes the sign of zero, so equal values share one bit pattern and - /// the bits identify the value exactly: fit for reproducibility records and bit-exact pins. + /// Equal values share one bit pattern, including the canonical `+0.0`. These bits identify the + /// value exactly. #[inline] #[must_use] pub(crate) const fn to_bits(self) -> u32 { @@ -161,14 +167,9 @@ impl NonNegative { self.0.is_normal() } - /// Returns whether the value stayed in domain. + /// Returns whether the stored reading is finite. /// - /// Construction admits only finite values and arithmetic escapes to `+∞` on overflow, so a - /// non-finite reading is exactly an escaped one. - /// - /// The one caller shape is a validation point that rejects escaped readings before acting on - /// a computed value. Anywhere else the query re-checks what construction already proved, and - /// the check itself is the defect. + /// Detects a non-finite result after arithmetic whose range requirements were not met. #[inline] #[must_use] pub(crate) const fn is_finite(self) -> bool { @@ -200,8 +201,7 @@ impl NonNegative { /// Clamps from below by a positive floor. /// - /// The larger of a non-negative value and a positive floor is at least the floor, so the - /// result carries the stricter domain with no re-validation. + /// The result is at least the positive finite floor and remains finite, with no re-validation. #[inline] #[must_use] pub(crate) const fn at_least(self, floor: Positive) -> Positive { @@ -210,11 +210,10 @@ impl NonNegative { /// Subtracts, saturating at zero. /// - /// The truncated difference `max(self - rhs, 0)`, mirroring the integer `saturating_sub`: a - /// difference below the domain floor returns zero. The magnitude of a difference of two - /// finite values of one sign never exceeds the larger operand, so the subtraction cannot - /// overflow and the result needs no re-validation. For the signed difference, `-` outputs - /// [`Finite`]. + /// Returns the truncated difference `max(self - rhs, 0)`. The magnitude of a difference of two + /// finite values of one sign never exceeds the larger operand. The subtraction cannot overflow, + /// and clamping negative differences to zero keeps the result in the domain without + /// re-validation. For the signed difference, `-` outputs [`Finite`]. #[inline] #[must_use] pub(crate) fn saturating_sub(self, rhs: Self) -> Self { @@ -240,13 +239,15 @@ impl NonNegative { /// argument, the infinities included. Once `exp` underflows, the asymptotes are exact /// (`sigmoid(200.0)` is `1.0` and `sigmoid(-200.0)` is `0.0`), and /// `sigmoid(-value) == 1 - sigmoid(value)` holds up to rounding. The logistic function is - /// the first derivative of [`softplus`](super::softplus). A NaN argument asserts in debug - /// builds and passes through in release, the way the arithmetic operators treat their - /// invalid inhabitants. + /// the first derivative of [`softplus`](super::softplus). The argument must not be NaN. + /// + /// # Example /// - /// # Examples + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{NonNegative}; + /// /// // At zero the two branches agree exactly: 1 / (1 + 1). /// assert_eq!(NonNegative::sigmoid(0.0), 0.5); /// // A naive `exp(200.0)` overflows. The stable form saturates. @@ -271,13 +272,16 @@ impl NonNegative { /// /// The penalty is `value²/2` up to the threshold and continues along the tangent line /// `threshold · (value - threshold/2)` above it. Both branches meet at `threshold²/2` with - /// matching first derivative `threshold`, so the penalty is continuous with a continuous - /// first derivative. An evaluation that overflows the `f32` range saturates at - /// [`f32::MAX`], the same resolution [`sigmoid`](Self::sigmoid) applies at its asymptote. + /// matching first derivative `threshold` in exact arithmetic. The floating-point evaluation + /// rounds these expressions and saturates an overflow at [`f32::MAX`]. + /// + /// # Example /// - /// # Examples + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{NonNegative, Positive}; + /// /// let threshold = Positive::new(1.0).expect("1.0 is positive"); /// /// // Quadratic regime: 0.5 · 0.5 · 0.5. @@ -298,11 +302,15 @@ impl NonNegative { threshold.mul_add(-0.5, self.0) * threshold }; - // Finite operands overflow only to +∞ and produce no NaN, so the clamp re-enters the - // domain. + // Both branches produce a nonnegative penalty with no NaN. Overflow can only produce +∞, + // which the clamp maps to `f32::MAX`. The clamped penalty is in the domain. Self::new_unchecked(penalty.min(f32::MAX)) } + /// Returns the square root. + /// + /// The root of a non-negative value is non-negative, with no re-validation. The root of + /// zero is zero. #[inline] #[must_use] pub(crate) fn sqrt(self) -> Self { @@ -311,11 +319,10 @@ impl NonNegative { Self(self.0.sqrt()) } - /// Raises to a real power. + /// Raises to a real power with deferred validation. /// - /// Never NaN over the domain: a negative base is unrepresentable, and `0⁰` is one. Overflow, - /// and a zero base under a negative exponent, escape to `+∞` and assert in debug builds, - /// mirroring integer `+`. + /// Overflow and zero raised to a negative exponent produce infinity in the [`Derivation`]. Zero + /// raised to zero is one. Underflow to zero remains nonnegative. #[inline] #[must_use] pub(crate) fn powf(self, exponent: f32) -> Self { @@ -327,8 +334,8 @@ impl NonNegative { /// Returns the reciprocal. /// - /// Never NaN and never negative over the domain. A zero reading escapes to `+∞` and asserts - /// in debug builds, mirroring integer `+`. An escaped `+∞` collapses back to `+0.0`. + /// The rounded reciprocal must be finite. Zero and sufficiently small positive operands + /// produce positive infinity. For other in-domain values the result is positive. #[inline] #[must_use] pub(crate) const fn inverse(self) -> Self { @@ -374,7 +381,7 @@ const impl Default for NonNegative { const impl PartialEq for NonNegative { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -400,7 +407,7 @@ const impl Ord for NonNegative { impl Hash for NonNegative { #[inline] fn hash(&self, state: &mut H) { - // canonical bits: equal values share one bit pattern, so `Hash` agrees with `Eq` + // hashing the canonical representation agrees with numeric equality state.write_u32(self.0.to_bits()); } } @@ -418,7 +425,6 @@ impl fmt::Display for NonNegative { } const impl From for NonNegative { - /// Widens into the enclosing domain: every positive value is non-negative. #[inline] fn from(value: Positive) -> Self { Self(value.get()) @@ -430,10 +436,8 @@ const impl core::ops::Add for NonNegative { /// Adds. /// - /// A sum of non-negatives is never NaN and never `-0.0`. Overflow escapes to `+∞` - a - /// wrong reading rather than a soundness break, since no unsafe code trusts the domain and - /// a persisted value re-validates at construction - and asserts in debug builds, mirroring - /// integer `+`. + /// The rounded sum must remain finite. A sum of in-domain values is never NaN or `-0.0`, + /// but it can overflow to positive infinity. #[inline] fn add(self, rhs: Self) -> Self { let sum = self.0 + rhs.0; @@ -455,9 +459,8 @@ const impl core::ops::Add for NonNegative { /// Adds a positive step into the positive domain. /// - /// The sum is positive - rounding is monotone, so it never rounds below the positive - /// operand - and never NaN. Overflow escapes to `+∞` and asserts in debug builds, - /// mirroring integer `+`. + /// The rounded sum must remain finite. Monotone rounding keeps it at least as large as the + /// positive operand, but does not prevent overflow. #[inline] fn add(self, rhs: Positive) -> Positive { let sum = self.0 + rhs.get(); @@ -472,10 +475,9 @@ const impl core::ops::Sub for NonNegative { /// Subtracts, into the finite domain. /// - /// The difference of two non-negative finite values is finite, with no re-validation: its - /// magnitude never exceeds the larger operand, so the subtraction cannot overflow. Equal - /// operands give `+0.0`. For the difference clamped back into this domain, - /// [`saturating_sub`](Self::saturating_sub) subtracts without leaving it. + /// The magnitude of a difference of two non-negative finite values never exceeds the larger + /// operand. The subtraction cannot overflow and needs no re-validation. Equal operands give + /// `+0.0`. Use [`saturating_sub`](Self::saturating_sub) to clamp negative differences to zero. #[inline] fn sub(self, rhs: Self) -> Finite { Finite::new_unchecked(self.0 - rhs.0) @@ -487,9 +489,8 @@ const impl core::ops::Mul for NonNegative { /// Multiplies by a finite signed value, leaving the domain. /// - /// The product follows the finite operand's sign and can overflow, so the result is a raw - /// float and the caller re-enters a domain at whichever boundary proves the bound. A NaN - /// cannot arise: both operands are finite. + /// The raw product follows the finite operand's sign and can overflow. A NaN cannot arise: both + /// operands are finite. #[inline] fn mul(self, rhs: Finite) -> f32 { self.0 * rhs.get() @@ -497,7 +498,6 @@ const impl core::ops::Mul for NonNegative { } const impl From for f64 { - /// Widens into double precision, exactly. #[inline] fn from(value: NonNegative) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. @@ -508,10 +508,6 @@ const impl From for f64 { const impl core::ops::Add for f32 { type Output = f32; - /// Adds a non-negative offset to a raw `f32`. - /// - /// The raw operand is arbitrary, so the sum can leave any bounded domain and returns a raw - /// float. #[inline] fn add(self, rhs: NonNegative) -> f32 { self + rhs.0 @@ -523,7 +519,6 @@ impl proptest::arbitrary::Arbitrary for NonNegative { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole domain, zero and subnormals included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -534,14 +529,12 @@ impl proptest::arbitrary::Arbitrary for NonNegative { } impl serde::Serialize for NonNegative { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f32(self.0) } } impl<'de> serde::Deserialize<'de> for NonNegative { - /// Deserializes a plain number, refusing values outside the finite non-negative range. fn deserialize>(deserializer: D) -> Result { let value = f32::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { diff --git a/libs/@local/graph/atlas/src/math/scalar/open_unit_fraction.rs b/libs/@local/graph/atlas/src/math/scalar/open_unit_fraction.rs index 59154629112..3d6ea22252a 100644 --- a/libs/@local/graph/atlas/src/math/scalar/open_unit_fraction.rs +++ b/libs/@local/graph/atlas/src/math/scalar/open_unit_fraction.rs @@ -11,8 +11,8 @@ use super::{DPositive, UnitFraction, raw_interop, unsafe_impl_try_from_bytes}; /// Validates an open-unit-fraction literal at compile time. /// -/// The expansion is a `const` block over [`OpenUnitFraction::new`], so a literal outside the domain -/// fails the build instead of a test run. Runtime values keep the checked constructor. +/// A literal outside the domain fails the build. Use [`OpenUnitFraction::new`] to check runtime +/// values. macro_rules! open_unit_fraction { ($value:expr) => { const { $crate::math::OpenUnitFraction::new($value).expect("the literal lies inside (0, 1)") } @@ -38,19 +38,21 @@ impl Error for NotInOpenUnitInterval {} /// A fraction strictly between zero and one, valid by construction. /// -/// The open-interval sibling of [`UnitFraction`], for parameters whose semantics degenerate at an -/// endpoint. A contraction factor of zero collapses whatever it scales, and a factor of one -/// never contracts, so an acceptance threshold at either end stops being a threshold. The -/// exclusion rides in the type: division by a fraction needs no zero check, and its logarithm -/// needs no domain check. A strict comparison against either endpoint still divides the domain -/// in two. +/// The open-interval sibling of [`UnitFraction`], for parameters that exclude both endpoints. +/// Division by a fraction needs no zero-divisor check, and its logarithm has a finite input in +/// its domain. As a threshold within `[0, 1]`, the fraction leaves values on either side of a +/// strict comparison. As a contraction factor, it is strictly below one in exact arithmetic, +/// although a floating-point product can round back to its operand or underflow to zero. /// -/// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value, so -/// fractions sort and key ordered maps with no NaN case. +/// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{OpenUnitFraction}; +/// /// let shrink = OpenUnitFraction::new(0.25).expect("a quarter contracts"); /// assert_eq!(shrink.get(), 0.25); /// @@ -79,15 +81,13 @@ impl OpenUnitFraction { /// Returns whether `value`'s exact bits are a stored fraction. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: the domain - /// holds no zero of either sign and accepted values store bit for bit, so the bits are - /// valid exactly when [`new`](Self::new) accepts the value. + /// For validating persisted bytes, this accepts exactly the values accepted by + /// [`new`](Self::new). The domain excludes both zeros and preserves accepted values bit for + /// bit. #[inline] #[must_use] pub const fn is_canonical(value: f64) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. Some(accepted) => accepted.0.to_bits() == value.to_bits(), None => false, } @@ -105,8 +105,7 @@ impl OpenUnitFraction { value > 0.0 && value < 1.0, "the caller promised a value strictly inside (0, 1)", ); - // No normalization: the promised domain contains no zero of either sign, so a kept - // promise is already canonical. + // excluding both zeros makes in-domain values canonical without normalization Self(value) } @@ -119,13 +118,17 @@ impl OpenUnitFraction { /// Returns the complement `1 − self`, widened to [`UnitFraction`]. /// - /// The result can be exactly one, because `1 − x` rounds to `1.0` for every `x ≤ 2⁻⁵⁴`, so - /// the complement of an open fraction lives in the closed type. It is never zero, bottoming - /// out at `2⁻⁵³`, the complement of the largest fraction below one. + /// The closed return type admits exactly one: `1 − x` rounds to `1.0` for every in-domain `x ≤ + /// 2⁻⁵⁴`. The complement is never zero. Its minimum is `2⁻⁵³`, the complement of the largest + /// fraction below one. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{OpenUnitFraction, UnitFraction}; + /// /// let kept = OpenUnitFraction::new(0.75).expect("0.75 lies inside (0, 1)"); /// assert_eq!( /// kept.complement(), @@ -139,23 +142,32 @@ impl OpenUnitFraction { #[inline] #[must_use] pub(crate) const fn complement(self) -> UnitFraction { - // In range with no check: the real result lies in (0, 1) and rounding cannot escape - // [0, 1]. Near one the subtraction is exact by Sterbenz, so the result is at least 2⁻⁵³; - // a positive result needs no sign normalization. + // Rounding cannot leave [0, 1], whose endpoints are representable. Here 1 − x lies in (0, + // 1), and Sterbenz's lemma makes subtraction exact for x ≥ 0.5. The minimum result is 2⁻⁵³, + // at x = 1 − 2⁻⁵³. The result is in-domain and strictly positive, requiring no sign + // normalization. UnitFraction::new_unchecked(1.0 - self.0) } /// Computes `ln(1 - self)` without forming the rounded difference. /// - /// Evaluates as `ln_1p(-self)`, which keeps relative precision where the fraction lies - /// close to one and `1.0 - self` would round away everything the logarithm reads. The open - /// interval excludes both endpoints, so the result is strictly negative and finite. + /// Evaluates as `ln_1p(-self)` to preserve a small fraction's contribution near zero, where + /// `1.0 - self` can round to exactly one. Near one, that subtraction is exact by Sterbenz's + /// lemma. The open interval keeps the logarithm finite and strictly negative. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{OpenUnitFraction}; + /// /// let confidence = OpenUnitFraction::new(0.999).expect("0.999 lies inside (0, 1)"); /// assert!((confidence.ln_complement() - 0.001_f64.ln()).abs() < 1e-12); + /// + /// let tiny = OpenUnitFraction::new(1e-300).expect("1e-300 is inside (0, 1)"); + /// assert_eq!(1.0 - tiny.get(), 1.0); + /// assert!(tiny.ln_complement() < 0.0); /// ``` #[inline] #[must_use] @@ -167,7 +179,7 @@ impl OpenUnitFraction { const impl PartialEq for OpenUnitFraction { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -192,7 +204,7 @@ const impl Ord for OpenUnitFraction { impl Hash for OpenUnitFraction { #[inline] fn hash(&self, state: &mut H) { - // canonical bits: equal fractions share one bit pattern, so `Hash` agrees with `Eq` + // hashing the unique representation agrees with numeric equality state.write_u64(self.0.to_bits()); } } @@ -206,7 +218,6 @@ impl fmt::Display for OpenUnitFraction { const impl TryFrom for OpenUnitFraction { type Error = NotInOpenUnitInterval; - /// Narrows, rejecting both endpoints. #[inline] fn try_from(value: UnitFraction) -> Result { let raw = value.get(); @@ -225,7 +236,6 @@ const impl From for f64 { const impl TryFrom for OpenUnitFraction { type Error = NotInOpenUnitInterval; - /// Validates as [`OpenUnitFraction::new`] does, carrying the rejected value in the error. #[inline] fn try_from(value: f64) -> Result { Self::new(value).ok_or(NotInOpenUnitInterval(value)) @@ -233,7 +243,6 @@ const impl TryFrom for OpenUnitFraction { } const impl PartialEq for OpenUnitFraction { - /// Compares across the scalar family, in one precision with no widening. #[inline] fn eq(&self, other: &DPositive) -> bool { self.0 == other.get() @@ -241,7 +250,6 @@ const impl PartialEq for OpenUnitFraction { } const impl PartialOrd for OpenUnitFraction { - /// Orders across the scalar family, in one precision with no widening. #[inline] fn partial_cmp(&self, other: &DPositive) -> Option { self.0.partial_cmp(&other.get()) @@ -253,9 +261,9 @@ const impl core::ops::Div for f64 { /// Divides a raw `f64` by the fraction, staying in `f64`. /// - /// The divisor lies strictly inside the unit interval, so it is never zero and never NaN: a - /// NaN quotient arrives only through the numerator. The quotient's magnitude exceeds the - /// numerator's and escapes to `±∞` on overflow against a small divisor. + /// A divisor in `(0, 1)` is never zero and never NaN. A NaN quotient can arise only through the + /// numerator. For a finite numerator, the rounded quotient's magnitude is at least the + /// numerator's and can overflow to infinity. #[inline] fn div(self, rhs: OpenUnitFraction) -> f64 { self / rhs.0 @@ -271,8 +279,6 @@ impl proptest::arbitrary::Arbitrary for OpenUnitFraction { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole open interval: the range starts at the smallest positive value and - /// excludes one. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -283,14 +289,12 @@ impl proptest::arbitrary::Arbitrary for OpenUnitFraction { } impl serde::Serialize for OpenUnitFraction { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f64(self.0) } } impl<'de> serde::Deserialize<'de> for OpenUnitFraction { - /// Deserializes a plain number, refusing values outside the open unit interval. fn deserialize>(deserializer: D) -> Result { let value = f64::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { diff --git a/libs/@local/graph/atlas/src/math/scalar/positive.rs b/libs/@local/graph/atlas/src/math/scalar/positive.rs index 5865dff94c0..402d9e0234c 100644 --- a/libs/@local/graph/atlas/src/math/scalar/positive.rs +++ b/libs/@local/graph/atlas/src/math/scalar/positive.rs @@ -10,8 +10,7 @@ use super::{DPositive, Finite, Negative, raw_interop, unsafe_impl_try_from_bytes /// Validates a positive literal at compile time. /// -/// The expansion is a `const` block over [`Positive::new`], so a literal outside the domain fails -/// the build instead of a test run. Runtime values keep the checked constructor. +/// A literal outside the domain fails the build. Use [`Positive::new`] to check runtime values. macro_rules! positive { ($value:expr) => { const { $crate::math::Positive::new($value).expect("the literal is finite and positive") } @@ -21,20 +20,23 @@ pub(crate) use positive; /// A finite, strictly positive `f32`, valid by construction. /// -/// The shared definition of the finite-and-positive check that recurs across configuration and -/// weight fields: a value that exists is valid, and the consuming site validates nothing. +/// Use [`DPositive`] when the finite domain needs double precision. Products and quotients return a +/// [`Derivation`] because rounding can produce infinity or zero. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{Positive}; +/// /// assert_eq!(Positive::new(2.5).expect("2.5 is positive").get(), 2.5); /// assert_eq!(Positive::new(0.0), None); /// assert_eq!(Positive::new(f32::NAN), None); /// ``` /// /// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value. -/// The domain excludes NaN and both zeros, so every value owns one bit pattern with no -/// canonicalization step. +/// Excluding NaN and both zeros gives every value one bit pattern without canonicalization. #[derive(Copy, Clone, zerocopy::Immutable, zerocopy::IntoBytes, zerocopy::KnownLayout)] #[repr(transparent)] pub(crate) struct Positive(f32); @@ -46,9 +48,6 @@ impl Positive { /// value overflows to `+∞` and leaves the domain. pub(crate) const MAX: Self = Self(f32::MAX); /// The domain's smallest value, the smallest positive subnormal `2⁻¹⁴⁹`. - /// - /// The domain admits subnormals, so this is the exact floor a representation check - /// compares against. pub(crate) const MIN: Self = Self(f32::from_bits(1)); /// The value one. pub(crate) const ONE: Self = Self(1.0); @@ -84,40 +83,42 @@ impl Positive { /// Returns whether `value`'s exact bits are a stored positive value. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: the domain - /// holds no zero of either sign and accepted values store bit for bit, so the bits are - /// valid exactly when [`new`](Self::new) accepts the value. + /// For validating persisted bytes, this accepts exactly the values accepted by + /// [`new`](Self::new). The domain excludes both zeros and preserves accepted values bit for + /// bit. #[inline] #[must_use] pub(crate) const fn is_canonical(value: f32) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. Some(accepted) => accepted.0.to_bits() == value.to_bits(), None => false, } } + /// Returns whether the value is a normal `f32`, at or above `2⁻¹²⁶`. + /// + /// The domain admits subnormals, and a subnormal carries fewer than 24 significand bits. A + /// caller whose relative-error argument assumes the full significand checks this before + /// relying on it. #[inline] #[must_use] pub(crate) const fn is_normal(self) -> bool { self.0.is_normal() } - /// Returns whether the value stayed in domain. - /// - /// Construction admits only finite values and arithmetic escapes to `+∞` on overflow, so a - /// non-finite reading is exactly an escaped one. + /// Returns whether the stored reading is finite. /// - /// The one caller shape is a validation point that rejects escaped readings before acting on - /// a computed value. Anywhere else the query re-checks what construction already proved, and - /// the check itself is the defect. + /// Detects a non-finite result after arithmetic whose range requirements were not met. #[inline] #[must_use] pub(crate) const fn is_finite(self) -> bool { self.0.is_finite() } + /// Widens to double precision, exactly. + /// + /// Every finite `f32` is exactly representable as `f64`. Widening preserves strict positivity. + /// Therefore the widened value is the same real number with no rounding and no re-validation. #[inline] #[must_use] pub(crate) const fn widen(self) -> DPositive { @@ -126,9 +127,10 @@ impl Positive { /// Multiplies into double precision, exactly and totally. /// - /// Two 24-bit significands multiply within 53 bits, so the widened product is the exact - /// real product with no rounding, and two `f32` exponents sum hundreds of shells inside - /// the `f64` range in both directions, so the product never leaves the positive domain. + /// The product of two 24-bit significands fits within 53 bits, and the product of two + /// positive finite `f32` values lies between `2⁻²⁹⁸` and `2²⁵⁶`. Widening both operands + /// before multiplication represents that product exactly in `f64`. Therefore the product + /// never leaves the positive domain and has no rounding error. #[inline] #[must_use] pub(crate) const fn mul_wide(self, rhs: Self) -> DPositive { @@ -137,9 +139,9 @@ impl Positive { /// Divides into double precision, totally. /// - /// The quotient of two `f32`-born positives is never NaN, and its exponent - one `f32` - /// exponent less another - stays hundreds of shells inside the `f64` range in both - /// directions, so the quotient never leaves the positive domain. One rounding. + /// Positive finite `f32` values lie in [2⁻¹⁴⁹, 2¹²⁸). Their quotient lies between + /// 2⁻²⁷⁷ and 2²⁷⁷, inside the normal `f64` range. Widening before division therefore gives + /// a finite positive result with one rounding. #[inline] #[must_use] pub(crate) const fn div_wide(self, rhs: Self) -> DPositive { @@ -148,9 +150,9 @@ impl Positive { /// Squares into double precision, exactly and totally. /// - /// A 24-bit significand squares within 53 bits, so the widened square carries no rounding. - /// A doubled `f32` exponent sits far inside the `f64` range on both sides, so the square - /// never leaves the positive domain. + /// Squaring a 24-bit significand needs at most 48 bits, within `f64`'s 53-bit precision. Every + /// positive finite `f32` square lies in [2⁻²⁹⁸, 2²⁵⁶), inside the normal `f64` range. Widening + /// before multiplication therefore gives an exact square that never leaves the positive domain. #[inline] #[must_use] pub(crate) const fn square_wide(self) -> DPositive { @@ -159,9 +161,8 @@ impl Positive { /// Multiplies, refusing an escape from the domain. /// - /// A product of positives is never NaN and never negative, so [`None`] is exactly an - /// overflow to `+∞` or an underflow to zero, and the caller owns the refusal that escape - /// deserves. + /// A product of positives is never NaN and never negative. Returns [`None`] exactly on overflow + /// to `+∞` or underflow to zero. #[inline] #[must_use] pub(crate) const fn checked_mul(self, other: Self) -> Option { @@ -187,18 +188,18 @@ impl Positive { #[inline] #[must_use] pub(crate) fn sqrt(self) -> Self { - // In domain with no check: sqrt is monotone from (0, MAX] into (0, ~1.8e19], never NaN - // for a positive operand, and never zero - the root halves the exponent, so the - // smallest input roots far above the underflow threshold. + // The square root is monotone and never NaN for a positive finite operand. In-domain inputs + // lie in [2⁻¹⁴⁹, 2¹²⁸), with real roots in [√(2⁻¹⁴⁹), 2⁶⁴). These bounds fit inside the + // normal `f32` range. The rounded root remains positive and finite, never zero, without + // re-validation. Self::new_unchecked(self.0.sqrt()) } /// Returns the geometric mean `√(self · rhs)`, total. /// - /// The widened product is exact, its root is at most the larger operand and at least the - /// smaller, and one `f64` rounding cannot carry a value bounded by [`MAX`](Self::MAX) past - /// the narrowing's rounding boundary, so the mean of two representable positives is - /// representable: the narrowing needs no check. + /// The exact geometric mean lies between the operands. The widened product is exact, and + /// both root rounding and narrowing are monotone. Therefore the result remains between + /// the representable positive operands and needs no range check. #[inline] #[must_use] pub(crate) fn geometric_mean(self, rhs: Self) -> Self { @@ -212,10 +213,9 @@ impl Positive { /// Returns the reciprocal. /// - /// The reciprocal of a positive value is positive and never rounds to zero, since even the - /// largest finite input's reciprocal stays above the smallest subnormal. An operand below - /// `1/MAX` overflows to `+∞` - a wrong reading rather than a soundness break, since no - /// unsafe code trusts the domain - and asserts in debug builds. + /// The rounded reciprocal must be finite. A sufficiently small operand produces positive + /// infinity. A valid operand's reciprocal never rounds to zero: even the reciprocal of + /// [`Self::MAX`] exceeds the smallest positive subnormal. #[inline] #[must_use] pub(crate) const fn recip(self) -> Self { @@ -232,7 +232,7 @@ impl Positive { const impl PartialEq for Positive { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -258,7 +258,7 @@ const impl Ord for Positive { impl Hash for Positive { #[inline] fn hash(&self, state: &mut H) { - // one bit pattern per value, so `Hash` agrees with `Eq` + // hashing the unique representation agrees with numeric equality state.write_u32(self.0.to_bits()); } } @@ -286,7 +286,6 @@ const impl core::ops::Neg for Positive { } const impl From for f64 { - /// Widens into double precision, exactly. #[inline] fn from(value: Positive) -> Self { // `f64::from` is not const-callable. The widening cast is lossless. @@ -299,9 +298,9 @@ const impl core::ops::Sub for Positive { /// Subtracts, into the finite domain. /// - /// The difference of two positive finite values is finite, with no re-validation: its - /// magnitude never exceeds the larger operand, so the subtraction cannot overflow. Equal - /// operands give `+0.0`. + /// The magnitude of a difference of two positive finite values never exceeds the larger + /// operand. The subtraction cannot overflow and needs no re-validation. Equal operands give + /// `+0.0`. #[inline] fn sub(self, rhs: Self) -> Finite { Finite::new_unchecked(self.0 - rhs.0) @@ -313,8 +312,8 @@ const impl core::ops::Div for f32 { /// Divides a raw `f32` by a positive divisor, which is never zero. /// - /// The result is a raw float: the numerator is arbitrary, so the quotient can leave any - /// bounded domain. + /// An arbitrary numerator can produce a quotient outside any bounded domain. The result remains + /// a raw float. #[inline] fn div(self, rhs: Positive) -> f32 { self / rhs.0 @@ -326,7 +325,6 @@ impl proptest::arbitrary::Arbitrary for Positive { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole domain, subnormals included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -337,14 +335,12 @@ impl proptest::arbitrary::Arbitrary for Positive { } impl serde::Serialize for Positive { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f32(self.0) } } impl<'de> serde::Deserialize<'de> for Positive { - /// Deserializes a plain number, refusing values outside the finite positive range. fn deserialize>(deserializer: D) -> Result { let value = f32::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { diff --git a/libs/@local/graph/atlas/src/math/scalar/positive_unit_fraction.rs b/libs/@local/graph/atlas/src/math/scalar/positive_unit_fraction.rs index e2a5b4f2fee..35f303de895 100644 --- a/libs/@local/graph/atlas/src/math/scalar/positive_unit_fraction.rs +++ b/libs/@local/graph/atlas/src/math/scalar/positive_unit_fraction.rs @@ -11,8 +11,8 @@ use super::{UnitFraction, raw_interop, unsafe_impl_try_from_bytes}; /// Validates a positive-unit-fraction literal at compile time. /// -/// The expansion is a `const` block over [`PositiveUnitFraction::new`], so a literal outside the -/// domain fails the build instead of a test run. Runtime values keep the checked constructor. +/// A literal outside the domain fails the build. Use [`PositiveUnitFraction::new`] to check runtime +/// values. macro_rules! positive_unit_fraction { ($value:expr) => { const { @@ -25,18 +25,20 @@ pub(crate) use positive_unit_fraction; /// A finite fraction in `(0, 1]`, valid by construction. /// -/// The half-open sibling of [`UnitFraction`], for factors whose lower endpoint alone degenerates: -/// a factor of zero silences whatever mass it scales, while a factor of one leaves it whole and -/// keeps its effect. The exclusion rides in the type, so dividing by the fraction needs no zero -/// check at the use site, and a product with the fraction vanishes only when the other operand -/// does. +/// The half-open sibling of [`UnitFraction`], for factors that may preserve a magnitude but +/// must be greater than zero. Dividing by the fraction needs no zero-divisor check. A product +/// of positive operands is positive in exact arithmetic, but it can underflow to zero in +/// floating-point arithmetic. /// -/// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value, so -/// fractions sort and key ordered maps with no NaN case. +/// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{PositiveUnitFraction}; +/// /// let share = PositiveUnitFraction::new(0.25).expect("0.25 lies inside (0, 1]"); /// assert_eq!(share.get(), 0.25); /// @@ -68,15 +70,13 @@ impl PositiveUnitFraction { /// Returns whether `value`'s exact bits are a stored fraction. /// - /// The bit-level twin of [`new`](Self::new), for validating persisted bytes: the domain - /// holds no zero of either sign and accepted values store bit for bit, so the bits are - /// valid exactly when [`new`](Self::new) accepts the value. + /// For validating persisted bytes, this accepts exactly the values accepted by + /// [`new`](Self::new). The domain excludes both zeros and preserves accepted values bit for + /// bit. #[inline] #[must_use] pub const fn is_canonical(value: f64) -> bool { match Self::new(value) { - // Compare against what construction stored, so the check follows any future - // normalization. Some(accepted) => accepted.0.to_bits() == value.to_bits(), None => false, } @@ -94,8 +94,7 @@ impl PositiveUnitFraction { value > 0.0 && value <= 1.0, "the caller promised a value inside (0, 1]", ); - // No normalization: the promised domain contains no zero of either sign, so a kept - // promise is already canonical. + // excluding both zeros makes in-domain values canonical without normalization Self(value) } @@ -110,7 +109,7 @@ impl PositiveUnitFraction { const impl PartialEq for PositiveUnitFraction { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -135,7 +134,7 @@ const impl Ord for PositiveUnitFraction { const impl PartialEq for PositiveUnitFraction { #[inline] fn eq(&self, other: &UnitFraction) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // both domains use the same unique representation for each shared value self.get().to_bits() == other.get().to_bits() } } @@ -151,7 +150,7 @@ const impl PartialOrd for PositiveUnitFraction { impl Hash for PositiveUnitFraction { #[inline] fn hash(&self, state: &mut H) { - // canonical bits: equal fractions share one bit pattern, so `Hash` agrees with `Eq` + // hashing the unique representation agrees with numeric equality state.write_u64(self.0.to_bits()); } } @@ -163,14 +162,12 @@ impl fmt::Display for PositiveUnitFraction { } impl serde::Serialize for PositiveUnitFraction { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f64(self.0) } } impl<'de> serde::Deserialize<'de> for PositiveUnitFraction { - /// Deserializes a plain number, refusing values outside the half-open unit interval. fn deserialize>(deserializer: D) -> Result { let value = f64::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { diff --git a/libs/@local/graph/atlas/src/math/scalar/tests.rs b/libs/@local/graph/atlas/src/math/scalar/tests.rs index 338a95ec6ac..37c47f0d369 100644 --- a/libs/@local/graph/atlas/src/math/scalar/tests.rs +++ b/libs/@local/graph/atlas/src/math/scalar/tests.rs @@ -17,6 +17,8 @@ use crate::math::{ }, }; +/// `UnitFraction::new` accepts `0`, `1` and interior values and refuses negatives, values above +/// one, NaN and both infinities. #[test] fn unit_fraction_accepts_exactly_the_closed_interval() { assert_eq!(UnitFraction::new(0.0), Some(UnitFraction::ZERO)); @@ -263,6 +265,8 @@ fn unit_fraction_hashes_follow_numeric_value() { assert_ne!(hash_of(quarter), hash_of(half)); } +/// `softplus(50)` is exactly `50`, `softplus(-50)` is positive but below `1e-20`, and +/// `softplus(-200)` is exactly zero once `exp` underflows `f32`. #[test] fn softplus_approaches_asymptotes() { // ln_1p(exp(-50)) is far below f32 ε at 50, so the positive @@ -275,6 +279,7 @@ fn softplus_approaches_asymptotes() { assert_eq!(softplus(-200.0), 0.0); } +/// `softplus(x) - softplus(-x) = x` within `1e-5` across negative, zero and positive inputs. #[test] fn softplus_satisfies_shift_identity() { // softplus(x) - softplus(-x) == x: the ln_1p terms share |x| and cancel. @@ -284,6 +289,8 @@ fn softplus_satisfies_shift_identity() { } } +/// The stable `softplus` agrees with the textbook `ln(1 + exp(x))` within `1e-5` on moderate +/// inputs. #[test] #[expect( clippy::imprecise_flops, @@ -297,6 +304,7 @@ fn softplus_matches_naive_on_small_values() { } } +/// `sigmoid(0)` is exactly `0.5`, and at `±200` the asymptotes `1` and `0` are exact. #[test] fn sigmoid_matches_hand_computed_values() { // At zero the two branches agree exactly: 1 / (1 + 1). @@ -306,6 +314,8 @@ fn sigmoid_matches_hand_computed_values() { assert_eq!(NonNegative::sigmoid(-200.0), 0.0); } +/// `sigmoid(-20)` stays positive and within a relative `1e-6` of `exp(-20)`, where the complement +/// form would round to zero. #[test] fn sigmoid_keeps_relative_precision_on_the_negative_tail() { // The complement form `1 - 1/(1 + exp(-|x|))` rounds to zero once exp(-|x|) drops below f32 ε. @@ -316,6 +326,8 @@ fn sigmoid_keeps_relative_precision_on_the_negative_tail() { assert!((tail - expected).abs() <= 1e-6 * expected, "tail {tail}"); } +/// `huber` is `0.5 · v²` below the threshold, `0.5 · t²` at it and `t · (v - 0.5 t)` above it, on +/// exactly representable inputs. #[test] fn huber_matches_hand_computed_regimes() { // Quadratic regime: 0.5 · value^2, over exactly-representable inputs. @@ -352,6 +364,8 @@ fn huber_is_continuous_at_the_threshold() { assert!((above.get() - below.get()) < 1e-3); } +/// `huber` at `1e20` against a `1e20` threshold clamps to `f32::MAX` instead of overflowing to +/// infinity. #[test] fn huber_saturates_instead_of_overflowing() { // In the quadratic regime the square of 10²⁰ overflows the `f32` range. The reading clamps @@ -362,12 +376,14 @@ fn huber_saturates_instead_of_overflowing() { ); } +/// `narrow_f32` returns powers of two unchanged. #[test] fn narrowing_round_trips_powers_of_two() { assert_eq!(narrow_f32(0.25), Some(0.25_f32)); assert_eq!(narrow_f32(-1024.0), Some(-1024.0_f32)); } +/// `narrow_f32(0.1)` rounds to the nearest `f32`, the value the `0.1_f32` literal denotes. #[test] fn narrow_f32_rounds_where_exact_rejects() { // 0.1 has no exact binary representation at either width; narrowing @@ -375,6 +391,7 @@ fn narrow_f32_rounds_where_exact_rejects() { assert_eq!(narrow_f32(0.1), Some(0.1_f32)); } +/// `narrow_f32` returns `None` for values beyond the `f32` range, infinity and NaN. #[test] fn narrowing_rejects_overflow_and_nan() { assert_eq!(narrow_f32(1e300), None); @@ -382,6 +399,7 @@ fn narrowing_rejects_overflow_and_nan() { assert!(narrow_f32(f64::NAN).is_none()); } +/// `narrow_f32(-0.0)` keeps the sign bit. #[test] fn narrowing_preserves_negative_zero() { let rounded = narrow_f32(-0.0).expect("negative zero is finite"); @@ -825,6 +843,9 @@ fn greater_than_one_requires_actual_growth() { assert_eq!(GreaterThanOne::new(f64::NAN), None); } +/// Deserialising refuses exactly what the constructors refuse (`0` for the positive unit fraction, +/// negatives, values above one) and admits the closed endpoints, and serializing writes the plain +/// number. #[test] fn serde_doors_validate_the_domain() { // A published record's wire form reads through `Deserialize`, so the door refuses @@ -866,6 +887,7 @@ mod miri { OpenUnitFraction, Positive, PositiveUnitFraction, UnitFraction, }; + /// `NonNegative` reads canonical positive bytes and refuses `-0.0`, negatives and NaN. #[test] fn non_negative_try_from_bytes() { assert_eq!( @@ -881,6 +903,7 @@ mod miri { NonNegative::try_read_from_bytes(&f32::NAN.to_ne_bytes()).expect_err("NaN is refused"); } + /// `DNonNegative` reads canonical positive bytes and refuses `-0.0` and NaN. #[test] fn d_non_negative_try_from_bytes() { assert_eq!( @@ -894,6 +917,7 @@ mod miri { DNonNegative::try_read_from_bytes(&f64::NAN.to_ne_bytes()).expect_err("NaN is refused"); } + /// `Positive` reads canonical positive bytes and refuses zero and NaN. #[test] fn positive_try_from_bytes() { assert_eq!( @@ -906,6 +930,7 @@ mod miri { Positive::try_read_from_bytes(&f32::NAN.to_ne_bytes()).expect_err("NaN is refused"); } + /// `DPositive` reads canonical positive bytes and refuses zero and NaN. #[test] fn d_positive_try_from_bytes() { assert_eq!( @@ -918,6 +943,7 @@ mod miri { DPositive::try_read_from_bytes(&f64::NAN.to_ne_bytes()).expect_err("NaN is refused"); } + /// `Finite` reads finite bytes of either sign and refuses NaN and infinity. #[test] fn finite_try_from_bytes() { assert_eq!( @@ -930,6 +956,7 @@ mod miri { Finite::try_read_from_bytes(&f32::INFINITY.to_ne_bytes()).expect_err("infinity is refused"); } + /// `DFinite` reads finite bytes of either sign and refuses NaN and infinity. #[test] fn d_finite_try_from_bytes() { assert_eq!( @@ -943,6 +970,7 @@ mod miri { .expect_err("infinity is refused"); } + /// `GreaterThanOne` reads `2.0` and refuses exactly one and NaN. #[test] fn greater_than_one_try_from_bytes() { assert_eq!( @@ -956,6 +984,7 @@ mod miri { GreaterThanOne::try_read_from_bytes(&f64::NAN.to_ne_bytes()).expect_err("NaN is refused"); } + /// `Log2` reads a byte below the shift width and refuses `64` and `255`. #[test] fn log2_try_from_bytes() { assert_eq!( @@ -968,6 +997,7 @@ mod miri { Log2::try_read_from_bytes(&[255_u8]).expect_err("255 is far past the shift width"); } + /// `UnitFraction` reads canonical interior bytes and refuses `-0.0`, values above one and NaN. #[test] fn unit_fraction_try_from_bytes() { assert_eq!( @@ -982,6 +1012,8 @@ mod miri { UnitFraction::try_read_from_bytes(&f64::NAN.to_ne_bytes()).expect_err("NaN is refused"); } + /// `PositiveUnitFraction` reads the closed endpoint `1.0` and refuses zero, values above one + /// and NaN. #[test] fn positive_unit_fraction_try_from_bytes() { assert_eq!( diff --git a/libs/@local/graph/atlas/src/math/scalar/unit_fraction.rs b/libs/@local/graph/atlas/src/math/scalar/unit_fraction.rs index cf26eb17bd2..f8310ee7563 100644 --- a/libs/@local/graph/atlas/src/math/scalar/unit_fraction.rs +++ b/libs/@local/graph/atlas/src/math/scalar/unit_fraction.rs @@ -14,8 +14,7 @@ use super::{ /// Validates a unit-fraction literal at compile time. /// -/// The expansion is a `const` block over [`UnitFraction::new`], so a literal outside the domain -/// fails the build instead of a test run. Runtime values keep the checked constructor. +/// A literal outside the domain fails the build. Use [`UnitFraction::new`] to check runtime values. macro_rules! unit_fraction { ($value:expr) => { const { $crate::math::UnitFraction::new($value).expect("the literal lies in [0, 1]") } @@ -23,6 +22,7 @@ macro_rules! unit_fraction { } pub(crate) use unit_fraction; +/// The rejected value of a failed [`UnitFraction`] conversion. /// /// [`TryFrom`] returns this error where [`UnitFraction::new`] returns [`None`] - the value lies /// outside `[0, 1]` or is NaN. The error carries the rejected value and displays it together with @@ -40,23 +40,28 @@ impl Error for NotInUnitInterval {} /// A finite fraction in `[0, 1]`, valid by construction. /// -/// A fraction that exists is valid, so the domain check lives at the constructor and nowhere else: -/// configuration knobs (thresholds, retained shares, rate fractions) and measured quantities -/// (recalls, admitted shares) both travel as this type, and the consuming site trusts the domain -/// instead of re-checking it. +/// Use [`Self::new`] to validate a raw value or [`Self::ratio`] to compute a fraction from counts. +/// Arithmetic whose rounded result remains in the interval returns a fraction without repeating +/// validation. +/// +/// Serializes as a number. Deserialization rejects NaN and values outside `[0, 1]`, and [`TryFrom`] +/// returns [`NotInUnitInterval`] for the same rejected values. /// /// [`Eq`], [`Ord`] and [`Hash`] are total, agree with one another, and follow numeric value, with /// `-0.0` and `+0.0` the same fraction. Fractions sort and key ordered maps with no NaN case. /// -/// Arithmetic that stays in `[0, 1]` stays in the type: [`complement`](Self::complement), -/// fraction-by-fraction `*` (with [`Product`](core::iter::Product) over iterators), and -/// [`ratio`](Self::ratio) construct valid fractions with no run-time re-check. Multiplying by a raw -/// `f64` returns a raw `f64`, and comparisons against raw floats follow IEEE semantics, so a -/// fraction never equals NaN. +/// [`complement`](Self::complement), fraction-by-fraction `*` (with +/// [`Product`](core::iter::Product) over iterators), and [`ratio`](Self::ratio) construct valid +/// fractions with no run-time re-check. Multiplying by a raw `f64` returns a raw `f64`. Comparisons +/// against raw floats follow IEEE semantics: a fraction never equals NaN. +/// +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{UnitFraction}; +/// /// let quarter = UnitFraction::new(0.25).expect("0.25 lies inside [0, 1]"); /// assert_eq!(quarter.get(), 0.25); /// @@ -72,8 +77,8 @@ impl Error for NotInUnitInterval {} /// UnitFraction::new(0.1875).expect("the product is exact") /// ); /// ``` -// No `FromBytes` and no `FromZeros`: byte-level construction could produce NaN or a value -// outside the interval in safe code, bypassing the validating constructors. +// `FromBytes` would admit NaN and out-of-interval values without validation. An all-zero +// representation is valid, although this type does not implement `FromZeros`. #[derive( Debug, Copy, @@ -99,7 +104,7 @@ impl UnitFraction { /// /// Returns [`None`] unless the value lies in `[0, 1]`. NaN fails both bounds. For a computed /// value whose rounding may drift just past an endpoint, use - /// [`new_clamped`](Self::new_clamped); for a quotient of integer counts, use + /// [`new_clamped`](Self::new_clamped). For a quotient of integer counts, use /// [`ratio`](Self::ratio). #[inline] #[must_use] @@ -116,13 +121,16 @@ impl UnitFraction { /// A promised `-0.0` is stored as `+0.0`. Where the proof is not immediate, [`new`](Self::new) /// checks instead. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{UnitFraction, unit_fraction}; + /// /// let low = unit_fraction!(0.5); /// let high = UnitFraction::new(0.75).expect("0.75 lies inside [0, 1]"); - /// // A midpoint of two fractions cannot leave [0, 1]: the sum rounds within - /// // [0, 2] because both endpoints are representable, and halving is exact. + /// // Rounding a sum within [0, 2] and halving keeps the midpoint in [0, 1]. /// let mid = UnitFraction::new_unchecked((low.get() + high.get()) / 2.0); /// assert_eq!(mid.get(), 0.625); /// ``` @@ -152,9 +160,13 @@ impl UnitFraction { /// landing at `1.0 + 2ε`. A value that is supposed to already be in range keeps /// [`new`](Self::new), which turns the drift into a visible refusal instead of absorbing it. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{UnitFraction}; + /// /// // Rounding drift saturates instead of failing. /// let similarity = UnitFraction::new_clamped(1.0 + f64::EPSILON).expect("only NaN is refused"); /// assert_eq!(similarity, UnitFraction::ONE); @@ -186,9 +198,13 @@ impl UnitFraction { /// The result is the correctly rounded quotient for counts up to 2⁵³. For larger counts it is /// approximate, within a relative error of `2⁻⁵¹` of the exact ratio. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{UnitFraction}; + /// /// let admitted = UnitFraction::ratio(34_317, 34_400).expect("the part is within its total"); /// assert!(admitted > 0.99); /// @@ -208,11 +224,14 @@ impl UnitFraction { return None; } - // Monotone casts keep the converted part at or below the converted total, a quotient of - // non-negatives carries a positive sign, and a real quotient ≤ 1 rounds to at most the - // representable 1.0: in range with no check, canonical with no normalization. Counts up to - // 2⁵³ cast exactly, so one rounding remains and the quotient is correctly rounded; above, - // three roundings compose to below 2⁻⁵¹ relative error. + // Monotone conversion preserves 0 ≤ part ≤ total and a positive total. Division of these + // finite values gives a real quotient in [0, 1], and rounding cannot leave this interval, + // whose endpoints are representable. A zero part gives canonical +0.0. The result is + // in-domain without re-validation or normalization. + // + // Counts up to 2⁵³ convert exactly, leaving only the division's rounding. For larger + // counts, the two conversions and division compose to less than 2⁻⁵¹ relative error for a + // nonzero part. Some(Self(part as f64 / total as f64)) } @@ -243,9 +262,8 @@ impl UnitFraction { /// Returns the canonical bit pattern. /// - /// Construction canonicalizes the sign of zero, so equal fractions share one bit pattern - /// and the bits identify the fraction exactly: fit for reproducibility records and - /// bit-exact pins. + /// Equal fractions share one bit pattern, including the canonical `+0.0`. These bits identify + /// the fraction exactly. #[inline] #[must_use] pub(crate) const fn to_bits(self) -> u64 { @@ -273,7 +291,7 @@ impl UnitFraction { /// on `[0.5, 1]`, and the only zero result is the complement of one. /// /// Complementing twice reproduces fractions in `[0.5, 1]` exactly and elsewhere returns to - /// within `2⁻⁵⁴` of the start, so a fraction below `2⁻⁵⁴` can come back as zero. + /// within `2⁻⁵⁴` of the start. A fraction below `2⁻⁵⁴` can return as zero. #[inline] #[must_use] pub(crate) const fn complement(self) -> Self { @@ -285,9 +303,8 @@ impl UnitFraction { /// Returns the square root. /// - /// The root of a fraction is a fraction, with no re-validation: the square root is monotone - /// on `[0, 1]` with `√0 = 0` and `√1 = 1` exact. A correctly rounded root of a value just - /// below one can round to exactly one, which the closed interval admits. + /// The square root is monotone on `[0, 1]`, with √0 = 0 and √1 = 1 exact. Rounding + /// cannot leave this interval, whose endpoints are representable. #[inline] #[must_use] pub(crate) fn sqrt(self) -> Self { @@ -300,7 +317,7 @@ impl UnitFraction { const impl PartialEq for UnitFraction { #[inline] fn eq(&self, other: &Self) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // a unique bit pattern per value makes bit equality agree with numeric equality self.0.to_bits() == other.0.to_bits() } } @@ -326,7 +343,7 @@ const impl Ord for UnitFraction { const impl PartialEq for UnitFraction { #[inline] fn eq(&self, other: &PositiveUnitFraction) -> bool { - // one bit pattern per value, so bit equality is numeric equality + // both domains use the same unique representation for each shared value self.get().to_bits() == other.get().to_bits() } } @@ -342,7 +359,7 @@ const impl PartialOrd for UnitFraction { impl Hash for UnitFraction { #[inline] fn hash(&self, state: &mut H) { - // canonical bits: equal fractions share one bit pattern, so `Hash` agrees with `Eq` + // hashing the canonical representation agrees with numeric equality state.write_u64(self.0.to_bits()); } } @@ -358,10 +375,8 @@ const impl Sub for UnitFraction { /// The difference of two unit fractions. /// - /// Both operands lie in [0, 1], so the difference lies in [−1, 1] and is always finite: - /// the landing is total and the unchecked constructor rides that theorem. The typed - /// carrier for the [−1, 1] landing itself does not exist yet, so the output claims - /// finiteness alone. + /// The rounded difference lies in [−1, 1] for every pair of unit fractions. It can be negative, + /// and [`DFinite`] preserves its finiteness without constraining its sign. #[inline] fn sub(self, rhs: Self) -> DFinite { DFinite::new_unchecked(self.0 - rhs.0) @@ -377,10 +392,9 @@ const impl Mul for UnitFraction { /// to [`UnitFraction::ZERO`]. #[inline] fn mul(self, rhs: Self) -> Self { - // In range with no check: the real product of values in [0, 1] stays in [0, 1] and - // rounding cannot escape an interval whose endpoints are representable. A product of - // non-negatives keeps the positive sign even at underflow, so zero arrives as the - // canonical +0.0. + // Rounding cannot leave an interval with representable endpoints. The real product lies in + // [0, 1], and both operands have sign bit zero, which multiplication preserves even on + // underflow to +0.0. The rounded product is in-domain and canonical without normalization. Self(self.0 * rhs.0) } } @@ -411,7 +425,6 @@ const impl From for f64 { const impl TryFrom for UnitFraction { type Error = NotInUnitInterval; - /// Validates as [`UnitFraction::new`] does, carrying the rejected value in the error. #[inline] fn try_from(value: f64) -> Result { Self::new(value).ok_or(NotInUnitInterval(value)) @@ -423,7 +436,6 @@ impl proptest::arbitrary::Arbitrary for UnitFraction { type Parameters = (); type Strategy = proptest::strategy::BoxedStrategy; - /// Draws from the whole closed interval, both endpoints included. fn arbitrary_with((): Self::Parameters) -> Self::Strategy { use proptest::strategy::Strategy as _; @@ -434,14 +446,12 @@ impl proptest::arbitrary::Arbitrary for UnitFraction { } impl serde::Serialize for UnitFraction { - /// Serializes as the plain number. fn serialize(&self, serializer: S) -> Result { serializer.serialize_f64(self.0) } } impl<'de> serde::Deserialize<'de> for UnitFraction { - /// Deserializes a plain number, refusing values outside the closed unit interval. fn deserialize>(deserializer: D) -> Result { let value = f64::deserialize(deserializer)?; Self::new(value).ok_or_else(|| { @@ -453,6 +463,10 @@ impl<'de> serde::Deserialize<'de> for UnitFraction { } } +/// Decodes a database float, clamping out-of-range values into the unit interval. +/// +/// Values below zero become zero, values above one become one, and NaN becomes zero. Each +/// out-of-domain value emits a warning. Decoding errors from the database float are preserved. impl<'row> tokio_postgres::types::FromSql<'row> for UnitFraction { fn from_sql( ty: &tokio_postgres::types::Type, @@ -479,20 +493,17 @@ impl<'row> tokio_postgres::types::FromSql<'row> for UnitFraction { } const impl From for UnitFraction { - /// Widens into the enclosing closed interval. #[inline] fn from(value: OpenUnitFraction) -> Self { - // No normalization: the open domain contains no -0.0, so the value is already canonical. + // excluding -0.0 makes the open-domain value canonical without normalization Self(value.get()) } } const impl From for UnitFraction { - /// Widens into the enclosing closed interval. #[inline] fn from(value: PositiveUnitFraction) -> Self { - // No normalization: the half-open domain contains no -0.0, so the value is already - // canonical. + // excluding -0.0 makes the half-open-domain value canonical without normalization Self(value.get()) } } @@ -500,10 +511,6 @@ const impl From for UnitFraction { const impl Sub for f64 { type Output = f64; - /// Subtracts a fraction from a raw `f64`. - /// - /// The raw operand is arbitrary, so the difference can leave any bounded domain and returns - /// a raw float. #[inline] fn sub(self, rhs: UnitFraction) -> f64 { self - rhs.0 @@ -520,10 +527,9 @@ const impl Mul for UnitFraction { /// [`UnitFraction::ZERO`], which is why the half-open type cannot hold the result. #[inline] fn mul(self, rhs: PositiveUnitFraction) -> Self { - // In range with no check: the real product of values in [0, 1] stays in [0, 1] and - // rounding cannot escape an interval whose endpoints are representable. A product of - // non-negatives keeps the positive sign even at underflow, so zero arrives as the - // canonical +0.0. + // Rounding cannot leave an interval with representable endpoints. The real product lies in + // [0, 1], and both operands have sign bit zero, which multiplication preserves even on + // underflow to +0.0. The rounded product is in-domain and canonical without normalization. Self(self.0 * rhs.get()) } } @@ -531,18 +537,16 @@ const impl Mul for UnitFraction { raw_interop!(UnitFraction[f64]); unsafe_impl_try_from_bytes!(UnitFraction[f64]); -// SAFETY: `repr(transparent)` over `f64` gives one stable layout - size 8, alignment 8, no -// padding - on every target, and the type has no interior mutability. The stored bits are the -// writer's native `f64`, so a reader on the other byte order computes a different value from -// the same bytes: every format that maps this type stamps its writer's byte order and refuses -// the other order at open, before any archived value is reached. This is also why `Archive` -// below is hand-written as the identity: the derive would route the field through the -// endian-tagged `Archived`, and native bits under a stamped manifest are the contract. +// SAFETY: repr(transparent) preserves the native f64 layout, and this type has no interior +// mutability. unsafe impl rkyv::Portable for UnitFraction {} -// SAFETY: `repr(transparent)` over `f64`, so every byte of a value is initialized. +// SAFETY: An f64 has no padding or uninitialized bytes. The transparent representation adds no +// bytes. Every byte of UnitFraction is therefore defined. unsafe impl rkyv::traits::NoUndef for UnitFraction {} +// identity archiving retains native f64 storage instead of converting to rkyv's endian-tagged +// Archived impl rkyv::Archive for UnitFraction { type Archived = Self; type Resolver = (); @@ -558,10 +562,10 @@ impl rkyv::Serialize for UnitFraction { } } -// SAFETY: a unit fraction imposes no bit-validity condition, since every `f64` bit pattern is -// constructible. Its domain is instead a value condition, checked here after construction for the -// same reason `unchecked` construction is safe. Running the check through `&self` is therefore -// sound, and `verify` returning `Ok` is exactly [`UnitFraction::is_canonical`]. +// SAFETY: Verify guarantees valid fields, but not the enclosing type's invariants. This check reads +// only the raw f64 field, which accepts every initialized bit pattern, and tests its range and +// canonical zero through is_canonical. Returning Ok therefore establishes the complete UnitFraction +// invariant without assuming it beforehand. unsafe impl rkyv::bytecheck::Verify for UnitFraction where C: rkyv::rancor::Fallible + ?Sized, diff --git a/libs/@local/graph/atlas/src/math/similarity/fit.rs b/libs/@local/graph/atlas/src/math/similarity/fit.rs index 8d1e571ea5f..6ba1f4e3720 100644 --- a/libs/@local/graph/atlas/src/math/similarity/fit.rs +++ b/libs/@local/graph/atlas/src/math/similarity/fit.rs @@ -1,7 +1,20 @@ -//! Weighted Procrustes fitting of a similarity to point correspondences. +//! Weighted Procrustes alignment from point-pair moments. //! -//! The closed-form solve consumes seven raw weighted moments that accumulate in one fused pass, -//! four pairs at a time, serially or across rayon workers. +//! For points pᵢ, qᵢ ∈ ℝ² and weights wᵢ ≥ 0, the model minimizes E(a, θ, t) = Σᵢ wᵢ‖aRθpᵢ + t − +//! qᵢ‖² over scale a > 0, rotation angle θ and translation t ∈ ℝ². Let W = Σᵢ wᵢ > 0, p̄ = Σᵢ wᵢpᵢ / +//! W and q̄ = Σᵢ wᵢqᵢ / W. Centring gives uᵢ = pᵢ − p̄ and vᵢ = qᵢ − q̄, with moments V = Σᵢ wᵢ‖uᵢ‖², +//! D = Σᵢ wᵢ⟨uᵢ, vᵢ⟩ and H = Σᵢ wᵢ(uᵢₓvᵢᵧ − uᵢᵧvᵢₓ). +//! +//! The best translation is t = q̄ − aRθp̄. After substitution, the scale-and-angle terms are a²V − +//! 2a(D cos θ + H sin θ). For V > 0 and C = √(D² + H²) > 0, the minimizing coefficients are a = +//! C/V, cos θ = D/C and sin θ = H/C. Raw moments recover the centred quantities in one pass, +//! avoiding a second read of the inputs. +//! +//! Accumulation and centring use `f64`, followed by narrowing to `f32` coefficients. Subtracting +//! raw moments can lose small variances or covariances, especially with large offsets or uneven +//! weights. The computed solution and its rejection tests are approximations to this +//! real-arithmetic model. Parallel reduction changes the grouping and can change both coefficients +//! and acceptance. use core::{ num::NonZero, @@ -30,20 +43,37 @@ impl Similarity { /// large enough that per-task overhead disappears against the fold. pub(crate) const PARALLEL_CHUNK: NonZero = NonZero::new(4096).expect("4096 is not zero"); - /// Fits the weighted orientation-preserving Procrustes alignment of paired points. + /// Estimates the weighted Procrustes alignment of paired points. + /// + /// The weighted covariance determines rotation and scale, and the translation maps the source + /// centroid to the target centroid, following the [Procrustes + /// model](crate::math::similarity::fit). The fold accumulates four pairs at a time in `f64` + /// SIMD lanes, then handles the trailing pairs individually. The fitted coefficients narrow to + /// `f32`. Use [`fit_par`](Self::fit_par) for parallel accumulation. + /// + /// A finite zero-weight pair contributes zero to the mathematical objective, but its + /// coordinates are still validated. Adding or removing such pairs can change the fold's + /// grouping and rounding. Raw-moment cancellation can also make a nondegenerate fit fail or + /// degrade its accuracy. + /// + /// Returns [`None`] for unequal slice lengths or fewer than two pairs. The accumulated data is + /// rejected if any coordinate or weight is non-finite or any weight is negative. The computed + /// total weight, source variance and covariance magnitude must be normal, with positive source + /// variance. Finally, all fitted coefficients must narrow to finite `f32` values and satisfy + /// [`new`](Self::new), including its rotation norm tolerance. These numerical tests do not + /// certify the exact rank or conditioning of the input. /// - /// The result is the similarity minimizing the weighted squared error `sum(weights[i] * - /// |apply(source[i]) - target[i]|^2)` in closed form. The weighted covariance between the centred point sets determines the rotation and scale, and the translation recovers the target centroid from the transformed source centroid. The fold takes four pairs at a time on SIMD lanes and the trailing `len % 4` pairs one at a time, accumulating every sum in double precision before the result narrows to the working `f32` coefficients. Zero-weight pairs leave the fit unchanged. For large inputs, [`fit_par`](Self::fit_par) runs the same accumulation across rayon workers. + /// # Complexity /// - /// Returns [`None`] when the slice lengths differ, the caller passes fewer than two pairs, any - /// coordinate or weight is not finite, any weight is negative, the total weight is not a normal - /// positive number, the weighted source points are coincident, the pairs do not determine an - /// orientation (the covariance cancels exactly), or the resulting coefficients leave the `f32` - /// range that [`new`](Self::new) accepts. + /// O(n) time and constant additional storage for n pairs. /// - /// # Examples + /// # Example + /// + /// This example is ignored because [`Similarity`] is crate-private. /// /// ```ignore + /// use crate::math::{Similarity, Rotation, Vec2, positive}; + /// /// let expected = /// Similarity::new(positive!(2.0), Rotation::from_cos_sin(0.0, 1.0), Vec2::new(1.0, -2.0)) /// .expect("scale 2.0 is normal and positive"); @@ -71,14 +101,13 @@ impl Similarity { /// Fits the weighted Procrustes alignment of large inputs in parallel. /// - /// The contract is identical to [`fit`](Self::fit), so the same inputs yield [`Some`] and - /// [`None`] in the same cases. This splits the slices into chunks whose moments accumulate on - /// rayon workers and combine at the end. Floating-point addition rounds per operation, so the - /// chunked reduction can differ from [`fit`](Self::fit)'s serial fold by a few units in the - /// last place. + /// This uses [`fit`](Self::fit)'s model, input checks and coefficient-range checks. Chunked + /// accumulation changes floating-point grouping and may change whether the computed moments + /// pass validation. No fixed ULP bound relates the parallel result to the serial fit, and + /// results are not promised to be bit-reproducible across parallel reductions. /// - /// The fold is memory-bound, so parallelism pays off from about a hundred thousand pairs. Below - /// that, [`fit`](Self::fit) is faster. + /// Parallel work is O(n) for n pairs. Benchmark the serial and parallel forms on the intended + /// input sizes and hardware before choosing a crossover. /// /// Work splits into chunks of [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK) pairs. Use /// [`fit_par_with`](Self::fit_par_with) to choose the pairs per chunk. @@ -90,10 +119,10 @@ impl Similarity { /// Fits the weighted Procrustes alignment in parallel with a caller-chosen chunk size. /// - /// The contract is identical to [`fit`](Self::fit). Each rayon work item accumulates the - /// moments of `chunk` pairs. Smaller chunks balance better across uneven core loads, larger - /// chunks amortize task overhead. [`fit_par`](Self::fit_par) uses - /// [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK). + /// This has [`fit_par`](Self::fit_par)'s model and numerical limits. Each work item accumulates + /// at most `chunk` pairs. Smaller chunks offer more scheduling units, while larger chunks + /// reduce the number of moment merges. Chunk size can affect both rounding and acceptance. + /// [`fit_par`](Self::fit_par) uses [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK). #[must_use] pub(crate) fn fit_par_with( source: &[Vec2], @@ -122,18 +151,23 @@ impl Similarity { /// Fits the unweighted Procrustes alignment of paired fields. /// - /// Equivalent to [`fit`](Self::fit) with every weight `1.0`, without materializing a weight - /// slice. The fields carry the finiteness proof, so the uniform moments accumulate with no - /// validity scan, and aligning corpus-scale fields costs no allocation. + /// This uses [`fit`](Self::fit)'s unit-weight model without materializing a weight slice. + /// [`FinitePointField`] establishes coordinate finiteness. The fold takes O(n) time and + /// constant additional storage, with no heap allocation. /// - /// Returns [`None`] when the field lengths differ, the caller passes fewer than two pairs, - /// the source points are coincident, the pairs do not determine an orientation (the - /// covariance cancels exactly), or the resulting coefficients leave the `f32` range that - /// [`new`](Self::new) accepts. + /// Returns [`None`] for unequal field lengths or fewer than two pairs, or when the computed + /// moments or narrowed coefficients fail [`fit`](Self::fit)'s numerical checks. Arithmetic + /// grouping can differ from the weighted implementation even with unit weights. /// - /// # Examples + /// # Example + /// + /// This example is ignored because the fitting API is crate-private and test-only. /// /// ```ignore + /// use hashql_core::id::IdSlice; + /// use crate::math::{FinitePointField, Similarity, Rotation, Vec2, positive}; + /// # hashql_core::id::newtype! { struct RowId(u32) } + /// /// let expected = /// Similarity::new(positive!(0.5), Rotation::from_cos_sin(1.0, 0.0), Vec2::new(3.0, 1.0)) /// .expect("scale 0.5 is normal and positive"); @@ -166,9 +200,11 @@ impl Similarity { /// Fits the unweighted Procrustes alignment of large fields in parallel. /// - /// The contract is [`fit_par`](Self::fit_par)'s with unit weights: the chunked reduction - /// carries the same units-in-the-last-place caveat and the same break-even near a hundred - /// thousand pairs. Work splits into chunks of [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK) pairs. + /// This uses [`fit_par`](Self::fit_par)'s model and numerical checks with unit weights. + /// [`FinitePointField`] establishes coordinate finiteness, and no weight slice is allocated. + /// Returns [`None`] for unequal lengths, fewer than two pairs or failed moment/coefficient + /// checks. Work splits into chunks of [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK) pairs, with the + /// same grouping-dependent rounding and acceptance as the weighted parallel form. #[inline] #[must_use] pub(crate) fn fit_uniform_par( @@ -189,39 +225,37 @@ impl Similarity { } } -/// Validity and weighted raw moments of a run of point pairs, accumulated in double precision. +/// Input validity and `f64` raw moments for a Procrustes solve. /// -/// One pass over the pairs gathers everything the closed-form Procrustes solve needs; -/// [`combine`](Self::combine) merges the moments of two runs, which makes the accumulation -/// chunkable across SIMD lanes and rayon workers. Both [`Similarity::fit`] and -/// [`Similarity::fit_par`] feed the same [`solve`](Self::solve). +/// One pass gathers the weighted point sums, source norm and source-target products. +/// [`combine`](Self::combine) adds partial moments for chunked accumulation. [`solve`](Self::solve) +/// centres them and computes the coefficients. #[derive(Debug, Copy, Clone)] struct FitSums { /// Whether every coordinate is finite and every weight finite and non-negative. /// - /// The weighted pass scans for it; the uniform pass holds it by construction over its - /// proven-finite fields. + /// Uniform accumulation requires finite points and uses unit weights. valid: bool, - /// The total weight `sum(w)`. + /// The total weight Σᵢ wᵢ. weight: f64, - /// The weighted source sum `sum(w · source)`. + /// The weighted source sum Σᵢ wᵢpᵢ. source: DVec2, - /// The weighted target sum `sum(w · target)`. + /// The weighted target sum Σᵢ wᵢqᵢ. target: DVec2, - /// The weighted product moment `sum(w · dot(source, target))`. + /// The weighted dot-product moment Σᵢ wᵢ⟨pᵢ, qᵢ⟩. dot: f64, - /// The weighted product moment `sum(w · perp_dot(source, target))`. + /// The weighted signed-area moment Σᵢ wᵢ(pᵢₓqᵢᵧ − pᵢᵧqᵢₓ). perp_dot: f64, - /// The weighted source moment `sum(w · |source|^2)`. + /// The weighted squared-source-norm moment Σᵢ wᵢ‖pᵢ‖². source_norm: f64, } impl FitSums { /// Accumulates the weighted raw moments of the paired slices. /// - /// The slices carry equal lengths, which the `fit` entry points check once before accumulating. - /// The fold takes four pairs at a time on double-precision lanes, and the trailing `len % 4` - /// pairs one at a time. + /// The slices must have equal lengths. The fold handles four pairs at a time on + /// double-precision lanes, then the trailing `len % 4` pairs individually. Invalid coordinates + /// or weights set [`valid`](Self::valid) to false. fn from_slices(source: &[Vec2], target: &[Vec2], weights: &[f32]) -> Self { let (source_batches, source_rest) = source.as_chunks::<4>(); let (target_batches, target_rest) = target.as_chunks::<4>(); @@ -247,11 +281,11 @@ impl FitSums { valid &= (source.to_simd().is_finite() & target.to_simd().is_finite()) & (weight.is_finite() & weight.simd_ge(Simd::splat(0.0))).resize(true); - // `f32` values widen exactly, and each product of two widened - // values fits in `f64`'s 53-bit significand, so the batch - // products and the fused axis accumulations below are exact; - // only the running additions round. That exactness is what - // keeps the centred-moment cancellation in `solve` accurate. + // Finite f32 values widen exactly. Products of two such values need at most 48 + // significand bits and fit f64's exponent range. Dot products and squared norms add two + // products and can round. Weighting those results introduces another product, and every + // running accumulation can round. Double precision reduces these errors but does not + // prevent cancellation during centring. let weight: Simd = weight.cast(); let source = DVec2x4T::from(Vec2x4T::from(source)); let target = DVec2x4T::from(Vec2x4T::from(target)); @@ -301,11 +335,9 @@ impl FitSums { /// Accumulates the raw moments of the paired slices under uniform unit weights. /// - /// The slices carry equal lengths; the `fit_uniform` entry points check this once before - /// accumulating. They also arrive from proven-finite fields, so the pass runs no validity - /// scan and `valid` holds by construction. The total weight is the exact pair count, and - /// every weighted moment degenerates to its plain sum, so the pass reads two slices instead - /// of three. + /// Both slices must have equal lengths and finite coordinates. Unit weights remove the weight + /// reads and multiplications. The total weight is the pair count converted to `f64`, which can + /// round above 2⁵³. #[expect( clippy::cast_precision_loss, reason = "pair counts remain exactly representable in f64 far beyond any corpus" @@ -325,8 +357,8 @@ impl FitSums { let mut perp_sum = Simd::splat(0.0_f64); let mut norm_sum = Simd::splat(0.0_f64); for (source, target) in source_batches.iter().zip(target_batches) { - // Widening is exact and each product of two widened values fits in `f64`'s 53-bit - // significand, exactly as in the weighted pass. Only the running additions round. + // Finite f32 products are exact in f64. The within-pair dot/norm sums and the running + // additions can still round. let source = DVec2x4T::from(Vec2x4T::from(Vec2x4::from(*source))); let target = DVec2x4T::from(Vec2x4T::from(Vec2x4::from(*target))); @@ -363,8 +395,8 @@ impl FitSums { /// Merges the moments of two runs of pairs. /// - /// Floating-point addition rounds per operation, so combining chunked sums can differ from one - /// serial fold over the concatenated runs by units in the last place. + /// The validity flags are conjoined and each moment is added in `f64`. Grouping changes + /// rounding, which the centring subtraction can amplify. const fn combine(self, other: Self) -> Self { Self { valid: self.valid && other.valid, @@ -379,10 +411,9 @@ impl FitSums { /// Solves the closed-form Procrustes alignment from the accumulated moments. /// - /// Returns [`None`] under exactly the data-dependent rejection cases documented on - /// [`Similarity::fit`]: an invalid coordinate or weight, a non-normal total weight, - /// weight-coincident source points, an exactly cancelling covariance, or coefficients leaving - /// the range [`Similarity::new`] accepts. + /// Returns [`None`] when the accumulated validity flag, computed moments or narrowed + /// coefficients fail the numerical checks described by [`Similarity::fit`]. Pair-count and + /// slice-length checks belong to the fitting entry points. fn solve(self) -> Option { if !self.valid || !self.weight.is_normal() { return None; @@ -391,27 +422,29 @@ impl FitSums { let source_centroid = self.source / self.weight; let target_centroid = self.target / self.weight; - // Centered moments follow from the raw ones by the parallel-axis - // identity. With `W = sum(w)`, `ms = sum(w s)`, `mt = sum(w t)`, - // and centroids `cs = ms / W`, `ct = mt / W`, expanding each - // centred product leaves cross terms that all collapse into one - // correction because `sum(w (s - cs)) = 0`: - // sum(w dot(s - cs, t - ct)) = sum(w dot(s, t)) - dot(ms, mt) / W - // sum(w perp(s - cs, t - ct)) = sum(w perp(s, t)) - perp(ms, mt) / W - // sum(w |s - cs|^2) = sum(w |s|^2) - |ms|^2 / W - // This fuses centring into the single accumulation pass shared - // by the serial and parallel fits. + // In real arithmetic, weighted centred deviations sum to zero. With mₚ = Σᵢ wᵢpᵢ and m_q = + // Σᵢ wᵢqᵢ, expanding each centred product leaves one correction: + // + // D = Σᵢ wᵢ⟨pᵢ, qᵢ⟩ − ⟨mₚ, m_q⟩ / W. + // + // H = Σᵢ wᵢ(pᵢₓqᵢᵧ − pᵢᵧqᵢₓ) − (mₚₓm_qᵧ − mₚᵧm_qₓ) / W. + // + // V = Σᵢ wᵢ‖pᵢ‖² − ‖mₚ‖² / W. + // + // Therefore raw moments suffice for the centred solve without a second input pass. The + // implemented subtraction uses rounded moments and can lose small differences. let dot = self.dot - self.source.dot(self.target).into_raw() / self.weight; let perp_dot = self.perp_dot - self.source.perp_dot(self.target).into_raw() / self.weight; let variance = self.source_norm - self.source.norm_squared().into_raw() / self.weight; - // Coincident (up to weight) source points give no scale. The identity's cancellation can - // round a mathematically zero variance below zero, which the sign check rejects together - // with the non-normal cases. + // Positive source variance determines the scale denominator. Cancellation can give a + // nonpositive computed value for distinct weighted points, or a positive value for a + // mathematically zero variance. This tests the computed denominator, not exact + // nondegeneracy. if !variance.is_normal() || variance <= 0.0 { return None; } - // An exactly cancelling covariance gives no orientation. + // a nonzero covariance magnitude determines orientation in the real-arithmetic model let covariance = dot.hypot(perp_dot); if !covariance.is_normal() { return None; diff --git a/libs/@local/graph/atlas/src/math/similarity/mod.rs b/libs/@local/graph/atlas/src/math/similarity/mod.rs index 871dbe8ddf0..3131d784798 100644 --- a/libs/@local/graph/atlas/src/math/similarity/mod.rs +++ b/libs/@local/graph/atlas/src/math/similarity/mod.rs @@ -1,15 +1,14 @@ -//! Orientation-preserving similarities: uniform scale, rotation, and translation. +//! Uniform scale, rotation and translation for shape-preserving alignment. //! -//! A similarity is the transform family produced by Procrustes alignment: it changes size, -//! orientation, and position while preserving every angle and every length ratio, so aligned -//! layouts keep their shape. [`Similarity`] restricts [`Transform`] to exactly this family, which -//! buys a guarantee the general type gives up, a total inverse. An affine map can collapse an -//! axis and lose its inverse. A similarity's scale is positive by construction, so every -//! similarity inverts. +//! In real arithmetic an orientation-preserving similarity preserves every angle and length ratio. +//! [`Similarity`] represents this model with `f32` coefficients and a checked positive scale. +//! Reciprocation preserves its scale range. Construction validates finite translation and an +//! approximately unit rotation, and composition and inversion reject invalid coefficients. +//! Transformed coordinates can still overflow. [`Similarity::fit`] estimates coefficients from +//! weighted point correspondences. //! -//! Use [`Similarity`] for a value that is a rigid motion plus uniform scaling (such as aligning one -//! generation's layout onto the previous one). Widen to [`Transform`] (via [`From`]) only when -//! composing with general affine maps. +//! Use [`Similarity`] for rigid motion with uniform scaling. Convert to [`Transform`] via [`From`] +//! when composing with general affine maps. use core::simd::Simd; @@ -28,29 +27,33 @@ mod residual; #[cfg(test)] mod tests; -/// An orientation-preserving similarity of 2D space: uniform scale, rotation, and translation. +/// An approximate orientation-preserving similarity of 2D space. /// -/// A similarity maps a vector `p` to `scale · R · p + translation`, where `R` is the rotation's -/// matrix. It scales lengths uniformly and preserves every angle, so shapes keep their proportions -/// and their winding direction. +/// The model maps p ∈ ℝ² to aRp + t, where a is the positive scale, R is the [`Rotation`] matrix +/// and t is the translation. With a unit rotation pair this scales all lengths by a and preserves +/// angles and winding direction in real arithmetic. The `f32` evaluation rounds and can overflow. /// -/// The scale of every value and its reciprocal are both strictly positive normal numbers: -/// [`new`](Self::new) and [`from_array`](Self::from_array) return [`None`] for anything else. -/// Reciprocals of accepted scales are themselves accepted, so every similarity's inverse is -/// again a lawful similarity. +/// [`new`](Self::new) and [`from_array`](Self::from_array) validate that the scale and its +/// reciprocal are both positive normal numbers. The rotation must satisfy |cos² + sin² − 1| ≤ 10⁻⁶, +/// evaluated in `f64`, and the translation must be finite. Construction enforces these conditions. +/// [`then`](Self::then) and [`inverse`](Self::inverse) reject results outside this domain. /// -/// Obtain one from weighted point correspondences with [`fit`](Self::fit), the closed-form -/// Procrustes alignment, or with [`fit_par`](Self::fit_par) when the pairs number in the hundreds -/// of thousands. Apply one to a single vector with [`apply`](Self::apply). A similarity widens -/// losslessly into a [`Transform`] via [`From`], so it composes with general affine transforms -/// through [`Transform::then`]. +/// Obtain a least-squares estimate from weighted point correspondences with [`fit`](Self::fit), or +/// use [`fit_par`](Self::fit_par) for parallel accumulation. Apply one to a single vector with +/// [`apply`](Self::apply). Conversion to [`Transform`] folds the scale into the rotation columns +/// with `f32` multiplication. It preserves the real-arithmetic model up to coefficient rounding, +/// and its application can differ from [`apply`](Self::apply). /// /// The coefficients persist in the order `[scale, cos, sin, x, y]`, the layout /// [`from_array`](Self::from_array) reads. /// -/// # Examples +/// # Example +/// +/// This example is ignored because [`Similarity`] is crate-private. /// /// ```ignore +/// use crate::math::{Similarity, Rotation, Vec2, positive}; +/// /// // Double the size, quarter-turn counterclockwise, then move right. /// let similarity = /// Similarity::new(positive!(2.0), Rotation::from_cos_sin(0.0, 1.0), Vec2::new(10.0, 0.0)) @@ -58,9 +61,7 @@ mod tests; /// /// assert_eq!(similarity.apply(Vec2::new(3.0, 4.0)), Vec2::new(2.0, 6.0)); /// ``` -// No `FromBytes` and no `FromZeros`: byte-level construction could mint a -// zero, negative, subnormal, or non-finite scale in safe code, bypassing -// the validating constructors that keep `inverse` total. +// byte construction would bypass validation of the scale and its reciprocal #[derive( Debug, Copy, @@ -78,7 +79,7 @@ pub(crate) struct Similarity { } impl Similarity { - /// The similarity that maps every vector to itself. + /// Unit scale, identity rotation and zero translation. pub(crate) const IDENTITY: Self = Self { scale: positive!(1.0), rotation: Rotation::IDENTITY, @@ -89,8 +90,9 @@ impl Similarity { /// /// Returns [`None`] unless `scale` and its reciprocal are both strictly positive normal /// numbers, which accepts magnitudes from [`f32::MIN_POSITIVE`] up to about `8.5e37`. The - /// reciprocal bound keeps inversion closed, because reciprocals of accepted scales are - /// themselves accepted. + /// reciprocal bound keeps the scale range closed under reciprocation. The rotation must satisfy + /// |cos² + sin² − 1| ≤ 10⁻⁶, evaluated in `f64`, and the translation must be finite. Construction + /// retains accepted coefficients without normalization. #[inline] #[must_use] pub(crate) const fn new( @@ -130,14 +132,14 @@ impl Similarity { self.translation } - /// Returns the similarity equivalent to applying `self` first, then `next`. + /// Composes `self` followed by `next`, validating the resulting coefficients. /// - /// This reads in application order. The scales multiply, the rotations compose via - /// [`Rotation::then`], and `next` transforms `self`'s translation. + /// In real arithmetic the parameters are a₂a₁, R₂R₁ and a₂R₂t₁ + t₂. [`Rotation::then`] + /// composes the rotation pair. Coefficient rounding can make the result differ from sequential + /// application. /// - /// Returns [`None`] when the product of the scales leaves the range [`new`](Self::new) accepts. - /// Two accepted f32 scales can overflow to infinity or underflow past the normal range, so - /// composition is not closed and the revalidation is what upholds the type's invariant. + /// Returns [`None`] when the rounded coefficients fail [`new`](Self::new), including scale + /// overflow or underflow, rotation drift beyond the norm tolerance, or non-finite translation. #[inline] #[must_use] pub(crate) const fn then(self, next: Self) -> Option { @@ -154,11 +156,14 @@ impl Similarity { ) } - /// Returns the similarity that undoes `self`. + /// Forms the inverse parameters using a reciprocal scale and conjugate rotation. + /// + /// For an exact unit rotation, the inverse model is a⁻¹Rᵀp − a⁻¹Rᵀt. The reciprocal scale + /// remains in the accepted range. The rotation retains any norm drift described by + /// [`Rotation::inverse`]. Inversion has no finite round-trip error guarantee. /// - /// [`new`](Self::new) admits only scales whose reciprocal is also normal, so the inverse's - /// scale satisfies the same invariant and inversion is total: applying a similarity and then - /// its inverse reproduces the input up to floating-point rounding. + /// Returns [`None`] when the computed coefficients fail [`new`](Self::new), including overflow + /// while rotating or scaling the inverse translation. #[inline] #[must_use] pub(crate) const fn inverse(self) -> Self { @@ -189,13 +194,13 @@ impl Similarity { ) } - /// Transforms four vectors at once, entirely in SIMD registers. + /// Applies the similarity to four vectors with SIMD arithmetic. /// - /// This folds the scale into the rotation coefficients, so each axis is two fused multiply-adds - /// over the batch's lane groups with no shuffles. On targets with native FMA those fused - /// operations round once where [`apply`](Self::apply) rounds after each multiply and each add. - /// Results differ by at most a few units in the last place of the intermediate terms, and by - /// many units in the last place of the result itself where the terms cancel. + /// This folds the scale into the rotation coefficients before using two fused multiply-adds per + /// axis. Each fused operation rounds its product and addition once, independently of native FMA + /// availability. The coefficient products, grouping and fusion differ from + /// [`apply`](Self::apply), with no uniform result-relative ULP bound between paths, especially + /// near cancellation or overflow. #[inline] #[must_use] pub(crate) fn apply_x4(self, batch: Vec2x4T) -> Vec2x4T { @@ -228,8 +233,8 @@ impl Similarity { /// Decomposes the similarity into its five coefficients. /// /// The order is `[scale, cos, sin, x, y]`: the uniform scale, the rotation's cosine and sine, - /// and the translation's components. The array round-trips through - /// [`from_array`](Self::from_array) bit for bit. + /// and the translation's components. [`from_array`](Self::from_array) accepts an array returned + /// from any similarity and preserves every component value. #[inline] #[must_use] pub(crate) const fn to_array(self) -> [f32; 5] { @@ -245,10 +250,10 @@ impl Similarity { /// Creates a similarity from its five coefficients. /// /// This reads the array as `[scale, cos, sin, x, y]`, the persisted coefficient layout. The - /// caller keeps the cosine and sine on the unit circle up to rounding, matching - /// the contract of [`Rotation::from_cos_sin`]. + /// cosine and sine must satisfy the squared-norm tolerance in [`new`](Self::new), and the + /// translation must be finite. /// - /// Returns [`None`] unless the scale lies in the range [`new`](Self::new) accepts. + /// Returns [`None`] unless every coefficient satisfies [`new`](Self::new). #[inline] #[must_use] pub(crate) const fn from_array( diff --git a/libs/@local/graph/atlas/src/math/similarity/residual.rs b/libs/@local/graph/atlas/src/math/similarity/residual.rs index 94e796fddd8..2d4c9612ffe 100644 --- a/libs/@local/graph/atlas/src/math/similarity/residual.rs +++ b/libs/@local/graph/atlas/src/math/similarity/residual.rs @@ -1,11 +1,15 @@ -//! RMS residual of a similarity over point correspondences. +//! RMS alignment error over point correspondences. //! -//! Paired with the Procrustes fit, the residual measures alignment quality. The similarity -//! transforms every source point, and the reduction turns the squared distances from the targets -//! into one root-mean-square. The fields carry the finiteness proof and the accumulation is -//! bounded far inside `f64`'s range, so the reading is total: a [`DNonNegative`] with no -//! rejection arm. Squares accumulate in double precision, four pairs at a time, serially or -//! across rayon workers. +//! For n > 0 paired points pᵢ and qᵢ, the model is RMS = √(Σᵢ‖aRpᵢ + t − qᵢ‖² / n), using the +//! similarity's scale a, rotation R and translation t. This measures the error left after +//! alignment. Coefficients widen to `f64` before application, and squared errors accumulate in +//! `f64`, serially or in parallel. +//! +//! Finite fields and finite similarity coefficients keep this computation within `f64` range on +//! 32-bit and 64-bit targets. The fields establish coordinate finiteness, but the similarity +//! constructors do not validate every coefficient. SIMD batches and the scalar remainder use +//! different fusion, and parallel reduction changes grouping. The result is an approximation with +//! no bitwise-equivalence promise between these evaluations. use core::simd::{Simd, num::SimdFloat as _}; @@ -25,29 +29,37 @@ use crate::math::{ impl Similarity { /// Returns the root-mean-square distance from transformed source points to their targets. /// - /// This is the movement a fitted alignment could not explain: after - /// [`fit_uniform_par`](Self::fit_uniform_par) it measures how far the two fields differ beyond - /// scale, rotation, and translation. This applies the transform with coefficients widened - /// to `f64`, and the squared distances accumulate in double precision, so corpus-scale sums - /// keep their accuracy. Pairs fold four at a time, and the trailing `len % 4` fold one at a - /// time. + /// For a fitted alignment, this measures the remaining movement beyond uniform scale, rotation + /// and translation. Coefficients widen to `f64` before application. This can differ from + /// measuring the `f32` outputs of [`apply`](Self::apply), even for targets generated by that + /// method. /// - /// The reading is total over the proven-finite fields: the accumulation is bounded far - /// inside `f64`'s range, so no rejection arm exists. The similarity's rotation and - /// translation coefficients must be finite, which every fit in this module produces and - /// [`new_unchecked`](DNonNegative::new_unchecked)'s debug assertion guards. + /// The similarity's rotation and translation coefficients must be finite. Successful fits + /// establish this condition, but arbitrary construction and inverse/composition operations may + /// not. With this condition and the fields' finite coordinates, the squared sum and RMS remain + /// finite on 32-bit and 64-bit targets. /// /// # Panics /// /// This panics when the field lengths differ or the fields are empty, because the residual /// is defined over matched pairs and an empty set has no mean. /// - /// # Examples + /// # Complexity + /// + /// O(n) time and constant additional storage for n pairs. + /// + /// # Example + /// + /// This example is ignored because [`Similarity`] and [`FinitePointField`] are crate-private. /// /// ```ignore + /// use hashql_core::id::IdSlice; + /// use crate::math::{FinitePointField, Similarity, Vec2}; + /// # hashql_core::id::newtype! { struct RowId(u32) } + /// /// let source = [Vec2::new(0.0, 0.0), Vec2::new(1.0, 0.0)]; - /// // Identity residual against offset targets: both points miss by - /// // (0.0, 3.0), so the RMS is exactly 3.0. + /// // Both targets are displaced by (0.0, 3.0) from the identity images. + /// // The RMS is exactly √((9 + 9) / 2) = 3. /// let target = [Vec2::new(0.0, 3.0), Vec2::new(1.0, 3.0)]; /// /// let source = FinitePointField::new(IdSlice::::from_raw(&source)) @@ -81,9 +93,11 @@ impl Similarity { /// Returns the root-mean-square residual of large fields in parallel. /// - /// The chunked reduction carries [`fit_par`](Self::fit_par)'s units-in-the-last-place - /// caveat and the same break-even near a hundred thousand pairs. Work splits into chunks of - /// [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK) pairs. + /// The coefficient-finiteness requirement and numerical model are those of + /// [`rms_residual`](Self::rms_residual). Work splits into chunks of + /// [`PARALLEL_CHUNK`](Self::PARALLEL_CHUNK) pairs. Different grouping changes rounding, and no + /// fixed ULP difference or bitwise reproducibility is promised. Benchmark the serial and + /// parallel forms for the intended input size and hardware. /// /// # Panics /// @@ -117,11 +131,11 @@ impl Similarity { /// Accumulates squared transformed-source-to-target distances in double precision. /// - /// The slices carry equal lengths, which the `rms_residual` entry points check once, and - /// arrive from proven-finite fields, so the sum is finite by [`finish_rms`]'s bound. + /// The slices must have equal lengths and finite coordinates, and the similarity's coefficients + /// must all be finite. The sum is then nonnegative and finite on 32-bit and 64-bit targets, by + /// the bound at [`finish_rms`]. fn squared_residuals(self, source: &[Vec2], target: &[Vec2]) -> f64 { - // `scale · R · p + t` with the scale folded into the rotation - // columns once, in double precision. + // fold the scale into the rotation columns in double precision let scale = f64::from(self.scale); let cos = scale * f64::from(self.rotation.cos()); let sin = scale * f64::from(self.rotation.sin()); @@ -162,19 +176,24 @@ impl Similarity { } } -/// Reduces an accumulated squared-distance sum to the RMS. +/// Divides a squared-error sum by its positive pair count and takes the square root. +/// +/// `squared` must be nonnegative and finite, and `pairs` must be nonzero. #[expect( clippy::cast_precision_loss, reason = "pair counts remain exactly representable in f64 far beyond any corpus" )] fn finish_rms(squared: f64, pairs: usize) -> DNonNegative { - // In domain with no check: every coordinate is field-proven finite and every coefficient - // is a finite f32, each below 2^128 in magnitude. A residual component is two - // scale-rotation-coordinate products (each below 2^128 cubed = 2^384) plus a translation - // and a target coordinate, so it stays below 2^386, its square below 2^772, a pair's - // squared distance below 2^773, and a sum of fewer than 2^60 pairs (a slice of 8-byte - // points cannot hold more) below 2^833 - finite in `f64` with room to spare, and - // non-negative as a sum of squares. The quotient by a positive pair count and the square - // root keep both properties. + // Finite f32 coordinates and coefficients have magnitude below 2¹²⁸. Each residual component + // has two scale-rotation-coordinate products below 2³⁸⁴, plus translation and target terms. + // Allowing for the fixed-operation rounding, a squared pair error is nonnegative and below B = + // 2⁷⁷⁴. On 32-bit and 64-bit targets, an 8-byte-point slice contains n < 2⁶⁰ pairs. Adding + // initialization or unused-lane zeros is exact for these nonnegative finite terms. Along a + // term's path, fewer than n additions combine contributions. With binary64 unit roundoff u = + // 2⁻⁵³, their amplification is at most (1 + u)ⁿ < exp(128) < 2¹⁸⁵. Thus the rounded sum is + // below nB · 2¹⁸⁵ < 2¹⁰¹⁹, within f64 range. Subnormal additions cannot threaten this upper + // bound. Dividing by the positive converted count and taking its square root preserve + // finiteness and nonnegativity. Therefore the result satisfies DNonNegative's numerical + // domain. DNonNegative::new_unchecked((squared / pairs as f64).sqrt()) } diff --git a/libs/@local/graph/atlas/src/math/similarity/tests.rs b/libs/@local/graph/atlas/src/math/similarity/tests.rs index 70e03fe84a5..6cb4acc5874 100644 --- a/libs/@local/graph/atlas/src/math/similarity/tests.rs +++ b/libs/@local/graph/atlas/src/math/similarity/tests.rs @@ -16,18 +16,24 @@ use crate::math::{ }; hashql_core::id::newtype! { - /// The fit tests' row domain. + /// Row identifiers for paired fitting fixtures. + /// #[id(const)] struct PairId(u32) } -/// Proves a fixture's points finite over the tests' row domain. +/// Validates fixture coordinates over the tests' row domain. +/// +/// # Panics +/// +/// Panics when a point is non-finite, including when locating it requires an unrepresentable row +/// ID. #[track_caller] fn field(points: &[Vec2]) -> &FinitePointField { FinitePointField::new(IdSlice::from_raw(points)).expect("the fixture points are finite") } -/// A similarity mixing all three components with inexact rotation angles. +/// Creates a scale-and-translation fixture with an inexact rotation angle. fn mixed_similarity() -> Similarity { Similarity::new( positive!(2.0), @@ -37,7 +43,7 @@ fn mixed_similarity() -> Similarity { .expect("scale 2.0 is normal and positive") } -/// Six well-spread, non-degenerate sample points for fitting. +/// Source points spanning both axes, with distinct points in every prefix of length two. const FIT_POINTS: [Vec2; 6] = [ Vec2::new(0.0, 0.0), Vec2::new(4.0, 1.0), @@ -47,7 +53,7 @@ const FIT_POINTS: [Vec2; 6] = [ Vec2::new(5.0, 5.0), ]; -/// Twelve spread-out, non-symmetric sample points for the fit certificates. +/// Asymmetric source points for noisy-fit comparisons. const CERT_POINTS: [Vec2; 12] = [ Vec2::new(0.0, 0.0), Vec2::new(4.0, 1.0), @@ -63,14 +69,14 @@ const CERT_POINTS: [Vec2; 12] = [ Vec2::new(-6.0, -0.5), ]; -/// Varied positive weights for the certificate points. +/// Varied positive weights for [`CERT_POINTS`]. const CERT_WEIGHTS: [f32; 12] = [ 1.0, 2.0, 0.5, 1.5, 3.0, 0.25, 1.25, 0.75, 2.5, 0.125, 1.75, 0.375, ]; /// Small asymmetric offsets keeping the fitted residual nonzero. /// -/// The error surface then has a strict minimum away from the exact-recovery case. +/// These perturb the known similarity's images to produce a noisy alignment fixture. const CERT_NOISE: [Vec2; 12] = [ Vec2::new(0.02, -0.03), Vec2::new(-0.04, 0.01), @@ -86,9 +92,7 @@ const CERT_NOISE: [Vec2; 12] = [ Vec2::new(0.03, 0.01), ]; -/// The certificate target. -/// -/// A known similarity image of [`CERT_POINTS`] plus the asymmetric [`CERT_NOISE`]. +/// Transforms [`CERT_POINTS`] and adds the asymmetric [`CERT_NOISE`]. fn noisy_certificate_target() -> [Vec2; 12] { let known = Similarity::new( positive!(1.75), @@ -100,10 +104,10 @@ fn noisy_certificate_target() -> [Vec2; 12] { core::array::from_fn(|index| known.apply(CERT_POINTS[index]) + CERT_NOISE[index]) } -/// Weighted squared alignment error of `similarity` over the pairs. +/// Computes weighted squared alignment error with separate `f64` operations. /// -/// Computed in plain double precision, independent of the fit's fused accumulation, so it can -/// referee the optimality certificate. +/// The comparison uses ordinary floating-point rounding over only the common prefix of the slices. +/// It evaluates the objective directly from the coefficients, independently of the fitting moments. #[expect( clippy::suboptimal_flops, reason = "the reference error deliberately uses plain arithmetic, independent of the FMA path \ @@ -136,8 +140,12 @@ fn weighted_error( /// Asserts two scalars agree up to a magnitude-scaled tolerance. /// -/// The fit narrows double-precision sums built from `f32`-rounded inputs, so its coefficients carry -/// a few ulps of working-precision error. +/// Requires |actual − expected| < 32 · EPSILON · max(|expected|, 1), using [`f32::EPSILON`]. The +/// absolute floor also covers coefficients near zero. +/// +/// # Panics +/// +/// Panics when the comparison fails, including for non-finite inputs. #[track_caller] fn assert_scalar_close(actual: f32, expected: f32) { let tolerance = 32.0 * f32::EPSILON * expected.abs().max(1.0); @@ -154,7 +162,6 @@ fn identity_maps_points_to_themselves() { assert_eq!(Similarity::IDENTITY.apply(point), point); } - // The identity coefficients match salt's persistence default. assert_eq!(Similarity::IDENTITY.to_array(), [1.0, 1.0, 0.0, 0.0, 0.0]); } @@ -257,7 +264,7 @@ fn composition_rejects_scales_leaving_the_range() { // 1e20 · 1e20 overflows to infinity. 1e-30 · 1e-30 underflows to zero. assert!(large.then(large).is_none()); assert!(small.then(small).is_none()); - // The same magnitudes compose once the scales cancel. + // the mixed product is about 10⁻¹⁰, within the accepted range assert!(large.then(small).is_some()); } @@ -332,13 +339,17 @@ fn fit_round_trips_an_exact_similarity_image() { let fitted = Similarity::fit(&FIT_POINTS, &target, &weights) .expect("exact correspondences determine the transform"); - // The target is an exact similarity image of the source, so applying - // the fit reproduces it point for point. + // targets are rounded f32 images, and the fitted application is compared with a tolerance for (point, reference) in FIT_POINTS.into_iter().zip(target) { assert_vec2_close(fitted.apply(point), reference); } } +/// Compares the fitted objective against coordinatewise perturbations. +/// +/// Scale changes by a relative ±10⁻³, while angle and each translation component change by an +/// absolute ±10⁻³. These comparisons sample the nearby error surface without certifying vanishing +/// partial derivatives or an exact minimizer. #[test] fn fit_is_optimal_against_a_perturbation_grid() { let target = noisy_certificate_target(); @@ -351,11 +362,10 @@ fn fit_is_optimal_against_a_perturbation_grid() { let angle = rotation.sin().atan2(rotation.cos()); let translation = fitted.translation(); - // The objective is smooth in each of the four parameters (quadratic - // in scale and translation, analytic in the angle), so a vanishing - // directional derivative along each coordinate axis is exactly a - // vanishing gradient: perturbing one parameter at a time certifies - // each partial at the returned minimizer. + // With the other parameters fixed, the real-arithmetic objective is quadratic in scale and + // translation, and sinusoidal in angle. A minimum cannot improve under either signed + // perturbation. Finite steps and rounded coefficients limit this check to the sampled + // candidates. for delta in [-1e-3_f32, 1e-3] { let perturbed = [ Similarity::new( @@ -388,8 +398,10 @@ fn fit_is_equivariant_under_target_transformation() { let base = Similarity::fit(&CERT_POINTS, &target, &CERT_WEIGHTS) .expect("twelve spread pairs determine the transform"); - // Post-transforming the target by a similarity scales every residual - // uniformly, so the minimizer moves to the composition with it. + // In real arithmetic, post-composing both the candidate and target with an invertible + // similarity of scale a multiplies every squared residual by a². The candidate family maps + // bijectively onto itself. Therefore its minimizer post-composes by the same similarity. The + // fixture comparisons allow for f32 rounding. let post = Similarity::new( positive!(0.5), Rotation::from_radians(-0.9), @@ -415,9 +427,9 @@ fn fit_is_invariant_under_uniform_weight_scaling() { let base = Similarity::fit(&CERT_POINTS, &target, &CERT_WEIGHTS) .expect("twelve spread pairs determine the transform"); - // Every moment scales by the common factor, which the total-weight - // divisions cancel. Multiplying by five rounds each accumulation - // differently, so agreement is ulp-level rather than bit-exact. + // Multiplying positive weights by five multiplies the real-arithmetic objective by five and + // preserves its minimizer. The absolute/relative tolerance accounts for differently rounded + // moment accumulation and centring. let scaled_weights = CERT_WEIGHTS.map(|weight| weight * 5.0); let scaled = Similarity::fit(&CERT_POINTS, &target, &scaled_weights) .expect("uniform weight scaling keeps the system well-determined"); @@ -438,8 +450,7 @@ fn fit_par_matches_fit_on_large_input() { ) .expect("scale 1.25 is normal and positive"); - // The logistic map below is deterministic, allocation-light, and chaotic enough to spread - // points, noise, and weights. + // the logistic recurrence supplies a reproducible, bounded sequence for the fixture let mut value = 0.37_f32; let mut pseudo = move || { value = 3.9 * value * (1.0 - value); @@ -468,9 +479,7 @@ fn fit_par_matches_fit_on_large_input() { let parallel = Similarity::fit_par(&source, &target, &weights) .expect("the parallel fit shares the serial contract"); - // The parallel reduction combines per-chunk sums in a different - // order than the serial fold, so agreement is magnitude-scaled ulps - // rather than bit-exact. + // parallel moment grouping can round differently from the serial fold for (actual, reference) in parallel.to_array().into_iter().zip(serial.to_array()) { assert_scalar_close(actual, reference); } @@ -489,8 +498,9 @@ fn fit_ignores_zero_weight_pairs() { let without_outlier = Similarity::fit(&FIT_POINTS, &target, &[1.0; 6]) .expect("exact correspondences determine the transform"); - // Append a pair far outside the correspondence with weight zero: every sum it touches gains an - // exact zero, so the fit is bit-identical. + // Appending pair seven keeps the first four pairs in the SIMD batch and the remaining pairs in + // the scalar tail. Its zero weight adds only zeros without regrouping the existing terms. + // Therefore the numerical coefficients remain equal in this fixture. let mut source = FIT_POINTS.to_vec(); let mut target = target.to_vec(); source.push(Vec2::new(1000.0, -1000.0)); @@ -513,9 +523,7 @@ fn fit_uniform_matches_fit_with_unit_weights() { let uniform = Similarity::fit_uniform(field(&CERT_POINTS), field(&target)) .expect("the uniform fit shares the weighted contract"); - // The uniform pass accumulates the same moments without the weight - // multiplications, so each sum rounds differently: agreement is - // magnitude-scaled ulps rather than bit-exact. + // removing unit-weight operations preserves the model, without requiring bitwise equality for (actual, reference) in uniform.to_array().into_iter().zip(weighted.to_array()) { assert_scalar_close(actual, reference); } @@ -587,8 +595,8 @@ fn rms_residual_vanishes_on_an_exact_image() { let residual = similarity.rms_residual(field(&FIT_POINTS), field(&target)); - // The `f32` application produced the targets while the residual applies widened `f64` - // coefficients, so the mismatch is the `f32` rounding of the application, not zero. + // f32 application produced the targets, while the residual applies widened coefficients with + // different grouping and fusion assert!(residual.get() < 1e-5, "exact image residual was {residual}"); } @@ -598,8 +606,8 @@ fn rms_residual_is_the_fit_objective_at_the_minimizer() { let fitted = Similarity::fit(&CERT_POINTS, &target, &CERT_WEIGHTS) .expect("twelve spread pairs determine the transform"); - // The unweighted residual of a nearby similarity must not fall below the unweighted optimum's, - // so this certifies against the uniform fit. + // The unweighted real-arithmetic optimum minimizes this residual. Compare its estimated fit + // with the weighted fit, allowing a 10⁻⁹ absolute tolerance. let uniform = Similarity::fit_uniform(field(&CERT_POINTS), field(&target)) .expect("twelve spread pairs determine the transform"); let best = uniform.rms_residual(field(&CERT_POINTS), field(&target)); @@ -674,9 +682,11 @@ fn rms_residual_par_panics_on_empty_fields() { let _: DNonNegative = Similarity::IDENTITY.rms_residual_par(field(&[]), field(&[])); } -/// Asserts both fit entry points reject the pairing. +/// Asserts that both weighted fit entry points reject the pairing. +/// +/// # Panics /// -/// Certifies their [`None`] agreement case by case. +/// Panics when either fit returns [`Some`]. #[track_caller] fn assert_fit_rejects(source: &[Vec2], target: &[Vec2], weights: &[f32]) { assert!(Similarity::fit(source, target, weights).is_none()); @@ -735,11 +745,8 @@ fn fit_recovers_exact_images_at_every_accepted_length() { let known = mixed_similarity(); let target = FIT_POINTS.map(|point| known.apply(point)); - // Lengths 2 through 6 cover the acceptance boundary from the accepting side (a rejection - // bound drifting to `<= 2` turns the shortest prefix into `None`) and every split between - // the SIMD batch and the trailing scalar loop: 2 and 3 fold entirely in the rest loop, - // 4 entirely in the batch, 5 and 6 in both. An accumulator defect in either path moves - // the recovered coefficients at the lengths that exercise it. + // Lengths 2 and 3 fold entirely in the scalar tail, 4 entirely in a SIMD batch, and 5 and 6 in + // both. The shortest prefix also exercises the minimum accepted pair count. let unit_weights = [1.0_f32; 6]; for pairs in 2..=FIT_POINTS.len() { let source = &FIT_POINTS[..pairs]; @@ -765,8 +772,6 @@ fn fit_uniform_rejects_mismatched_lengths() { let known = mixed_similarity(); let target = FIT_POINTS.map(|point| known.apply(point)); - // Both truncation directions: a rejection that degrades into a zip would silently fit the - // shorter prefix of these exact images and return `Some`. assert!(Similarity::fit_uniform(field(&FIT_POINTS[..5]), field(&target)).is_none()); assert!(Similarity::fit_uniform(field(&FIT_POINTS), field(&target[..5])).is_none()); assert!(Similarity::fit_uniform_par(field(&FIT_POINTS[..5]), field(&target)).is_none()); @@ -779,11 +784,7 @@ fn fit_rejects_a_negative_weight_at_every_index() { let source = &FIT_POINTS[..5]; let target: Vec = source.iter().map(|&point| known.apply(point)).collect(); - // The sweep places the negative weight in every SIMD batch lane and in the trailing - // scalar pair, since indices 0 through 3 fill the one full batch and index 4 rides the - // rest loop. The coordinates stay finite and the moments stay well conditioned under the - // mixed-sign weights, so the validity mask is the only rejection, and a fold that loses - // the weight-sign lane fits these exact images instead. + // indices 0 through 3 fill one SIMD batch, and index 4 uses the scalar tail for index in 0..source.len() { let mut weights = [1.0_f32; 5]; weights[index] = -0.5; @@ -805,11 +806,8 @@ fn fit_par_rejects_an_invalid_chunk_beside_a_valid_one() { let source = &CERT_POINTS[..8]; let target: Vec = source.iter().map(|&point| known.apply(point)).collect(); - // The chunk size of four splits the eight pairs so that the first chunk is entirely - // valid and the second carries the negative weight, leaving `FitSums::combine`'s - // validity conjunction as the only rejection of the merged moments. The pairs are exact - // images with finite coordinates, so a merge that keeps the invalid side's moments while - // losing its flag fits them exactly. + // a chunk size of four puts the negative weight in the second partial sum and exercises + // validity propagation through the merge let mut weights = [1.0_f32; 8]; weights[6] = -0.5; let chunk = NonZero::new(4).expect("four is not zero"); @@ -832,8 +830,8 @@ fn invalid_scales_are_rejected() { ]; for scale in invalid_scales { - // Negative, NaN, and infinite scales are unrepresentable as `NonNegative`; the ones the - // domain admits must still fail `new`'s reciprocal validation. + // Positive::new rejects nonpositive and non-finite values. The remaining invalid scales + // exercise new's normality and reciprocal checks. if let Some(scale) = Positive::new(scale) { assert!( Similarity::new(scale, Rotation::IDENTITY, Vec2::ZERO).is_none(), @@ -847,9 +845,10 @@ fn invalid_scales_are_rejected() { } } -/// An arbitrary well-conditioned similarity. +/// Generates similarities with bounded scale, angle and translation. /// -/// Scale in `0.1..10`, an arbitrary rotation angle, and a translation bounded to `-1e2..1e2`. +/// Scale lies in `0.1..10`, the angle in `-16..16` radians and each translation component in +/// `-1e2..1e2`. fn similarity_strategy() -> impl Strategy { (0.1_f32..10.0, -16.0_f32..16.0, -1e2_f32..1e2, -1e2_f32..1e2).prop_map( |(scale, radians, translate_x, translate_y)| { @@ -863,11 +862,10 @@ fn similarity_strategy() -> impl Strategy { ) } -/// A similarity scales all distances uniformly. +/// Compares distance ratios on separated points under a bounded similarity. /// -/// For any two points separated by at least one unit, the distance ratio equals the scale up to a -/// relative tolerance. The strategy bounds coordinates to `-1e3..1e3`, and the separation floor -/// keeps the subtraction's cancellation error small relative to the distance. +/// Coordinates lie in `-1e3..1e3`. Requiring separation of at least one unit limits cancellation +/// relative to the reference distance. The assertion allows a relative error of 10⁻³. #[property_test] fn apply_scales_distances_uniformly( #[strategy = similarity_strategy()] similarity: Similarity, @@ -894,19 +892,17 @@ fn apply_scales_distances_uniformly( ); } -/// Fitting an exact similarity image of non-collinear points recovers the coefficients. +/// Fits rounded similarity images of distinct jittered source points. /// -/// The sources are four well-spread base points jittered by at most `0.5`, far less than the base -/// triangle's extent, so the points can never become collinear. +/// Every prefix of length two or more contains the first two points, whose x coordinates differ by +/// more than seven. A similarity needs distinct source points, not noncollinearity. #[property_test] fn fit_recovers_a_random_similarity( #[strategy = similarity_strategy()] similarity: Similarity, #[strategy = proptest::array::uniform8(-0.5_f32..0.5)] jitter: [f32; 8], #[strategy = 2_usize..=8] pairs: usize, ) { - // The pool spreads eight points so every prefix of two or more is well conditioned, and - // the varying prefix length exercises every split between the SIMD batch and the trailing - // scalar loop across the random input space. + // varying the prefix length exercises full SIMD batches and every scalar-tail length let pool = [ Vec2::new(jitter[0], jitter[1]), Vec2::new(8.0 + jitter[2], jitter[3]), @@ -926,9 +922,8 @@ fn fit_recovers_a_random_similarity( let fitted = Similarity::fit(source, &target, &[1.0_f32; 8][..pairs]) .expect("well-spread points with an exact image are well-conditioned"); - // The target coordinates are f32-rounded images, so the recovered - // coefficients carry working-precision error scaled by their - // magnitude. + // f32 rounding of the target images introduces working-precision error into the recovered + // coefficients. The comparison allows for this error relative to coefficient magnitude. let expected = similarity.to_array(); for (index, (actual, expected)) in fitted.to_array().into_iter().zip(expected).enumerate() { prop_assert!( @@ -941,11 +936,10 @@ fn fit_recovers_a_random_similarity( } } -/// The inverse of any similarity satisfies the constructor's own invariant. +/// Checks reciprocal-scale closure across the accepted exponent range. /// -/// The scale spans the accepted range's full exponent spread with a non-dyadic mantissa, so the -/// reciprocal rounds rather than inverting exactly, and the property exercises both boundaries, -/// where a rounded reciprocal has the least room before the normal range ends. +/// Mantissas in [1, 2) and exponents from −126 through 125 give positive normal scales below 2¹²⁶. +/// Reciprocals may round. Zero translation isolates scale closure from translation overflow. #[property_test] fn inverse_stays_inside_the_constructed_range( #[strategy = (1.0_f32..2.0, -126_i32..=125, -16.0_f32..16.0)] (mantissa, exponent, radians): ( diff --git a/libs/@local/graph/atlas/src/math/test_alloc.rs b/libs/@local/graph/atlas/src/math/test_alloc.rs index e61004339ab..e26101e6a21 100644 --- a/libs/@local/graph/atlas/src/math/test_alloc.rs +++ b/libs/@local/graph/atlas/src/math/test_alloc.rs @@ -1,5 +1,3 @@ -//! A forwarding allocator that counts deallocations, for `Drop` contract tests. - use alloc::alloc::Global; use core::{ alloc::{AllocError, Allocator, Layout}, @@ -7,24 +5,29 @@ use core::{ ptr::NonNull, }; -/// Forwards to [`Global`] and counts deallocations. +/// A [`Global`] allocator with a deallocation counter for ownership tests. pub(crate) struct CountingAllocator { deallocations: Cell, } impl CountingAllocator { + /// Creates an allocator with a zero deallocation count. pub(crate) fn new() -> Self { Self { deallocations: Cell::new(0), } } + /// Returns how many deallocations have passed through this allocator. pub(crate) fn deallocations(&self) -> usize { self.deallocations.get() } } -// SAFETY: allocation and deallocation forward to `Global` unchanged. The count is bookkeeping. +// SAFETY: Allocator requires allocations to remain valid until released through that allocator. All +// instances delegate storage management to Global, and moving or dropping the counter leaves those +// allocations unchanged. Therefore this implementation preserves Global's allocation lifetime and +// layout contracts. unsafe impl Allocator for CountingAllocator { fn allocate(&self, layout: Layout) -> Result, AllocError> { Global.allocate(layout) @@ -32,7 +35,10 @@ unsafe impl Allocator for CountingAllocator { unsafe fn deallocate(&self, ptr: NonNull, layout: Layout) { self.deallocations.set(self.deallocations.get() + 1); - // SAFETY: `ptr` came from `allocate` above, which forwarded to `Global` with this layout. + // SAFETY: the caller must provide a currently allocated pointer and a fitting layout. All + // buffers this allocator obtains come from Global, with their allocation layouts preserved + // unchanged through this call. Therefore Global may deallocate this pointer with the + // supplied layout under the caller's obligations. unsafe { Global.deallocate(ptr, layout) } } } diff --git a/libs/@local/graph/atlas/src/math/tests.rs b/libs/@local/graph/atlas/src/math/tests.rs index 0a54668477e..aa9fb12b99e 100644 --- a/libs/@local/graph/atlas/src/math/tests.rs +++ b/libs/@local/graph/atlas/src/math/tests.rs @@ -10,7 +10,7 @@ pub(crate) const POINTS: [Vec2; 4] = [ Vec2::new(4.0, 8.0), ]; -/// Points for deterministic apply/`apply_x4` agreement sweeps. +/// Points for scalar and batch transform comparisons. /// /// Spans magnitudes well below and above 1.0, negative coordinates, mixed signs, and zero. pub(crate) const SWEEP_POINTS: [Vec2; 12] = [ @@ -28,7 +28,7 @@ pub(crate) const SWEEP_POINTS: [Vec2; 12] = [ Vec2::new(-99999.0, 0.0), ]; -/// Translation offsets for deterministic apply/`apply_x4` agreement sweeps. +/// Translation offsets for scalar and batch transform comparisons. /// /// Spans zero, small fractional offsets, mixed-sign offsets, and offsets large enough to move a /// result across a magnitude decade. @@ -43,8 +43,12 @@ pub(crate) const SWEEP_TRANSLATIONS: [Vec2; 6] = [ /// Asserts two vectors agree up to a magnitude-scaled tolerance. /// -/// The tolerance is a few dozen ulps of the expected value, which absorbs the rounding of -/// trigonometry, FMA contraction, and inverse round trips without accepting real errors. +/// Each component permits an absolute error below 32 · `f32::EPSILON` · max(|expected|, 1). The +/// floor allows an absolute error near zero rather than an ULP bound at the result's magnitude. +/// +/// # Panics +/// +/// Panics if either component's error is outside the tolerance or is NaN. #[track_caller] pub(crate) fn assert_vec2_close(actual: Vec2, expected: Vec2) { let tolerance = |reference: f32| 32.0 * f32::EPSILON * reference.abs().max(1.0); diff --git a/libs/@local/graph/atlas/src/math/transform/fit.rs b/libs/@local/graph/atlas/src/math/transform/fit.rs index 03514134991..66f04dc2fe9 100644 --- a/libs/@local/graph/atlas/src/math/transform/fit.rs +++ b/libs/@local/graph/atlas/src/math/transform/fit.rs @@ -1,10 +1,19 @@ -//! Least-squares affine fitting of point correspondences. +//! Least-squares affine alignment, including anisotropic deformation. //! -//! The closed-form solve consumes eleven raw moments that accumulate in one serial -//! double-precision pass. The Procrustes fit constrains its linear part to a rotation under one -//! scale. This fit releases both axes and therefore absorbs the anisotropic deformation a -//! similarity leaves in its residual, which is what makes the pair of fits a decomposition for -//! evidence. +//! For paired points pᵢ, qᵢ ∈ ℝ², the model minimizes E(A, t) = Σᵢ‖Apᵢ + t − qᵢ‖² over a real 2x2 +//! matrix A and translation t. Let p̄ and q̄ be the point means, S = Σᵢ(pᵢ − p̄)(pᵢ − p̄)ᵀ the source +//! scatter and C = Σᵢ(qᵢ − q̄)(pᵢ − p̄)ᵀ the target-source cross-scatter. The centred normal equation +//! is AS = C. When S is nonsingular, A = CS⁻¹ and t = q̄ − Ap̄ give the unique minimizer. The fitted +//! A may itself be singular. +//! +//! A serial pass accumulates eleven raw scalar moments in `f64`, then centres them and solves the +//! 2x2 system before narrowing coefficients to `f32`. Raw-moment subtraction and a near-singular +//! source scatter can amplify rounding. The determinant check tests the computed system, without +//! certifying exact source rank. +//! +//! A general affine fit can absorb shear and anisotropic scale that a +//! [`Similarity`](crate::math::Similarity) cannot represent. Comparing their residuals measures the +//! additional error explained by that broader family, subject to the fits' numerical errors. use hashql_core::id::Id; @@ -12,31 +21,33 @@ use super::Transform; use crate::math::{DNonNegative, Derivation, FinitePointField, dvec2::DVec2}; impl Transform { - /// Fits the unweighted least-squares affine map of paired fields. + /// Estimates the unweighted least-squares affine map of paired fields. + /// + /// The centred normal equations determine the linear part and translation as described by the + /// [affine fitting model](crate::math::transform::fit). Moments accumulate serially in `f64`, + /// then the coefficients narrow to `f32`. Large offsets relative to point spread can make + /// raw-moment centring inaccurate. Near-collinearity also makes the solve sensitive to + /// perturbations. + /// + /// Returns [`None`] for unequal field lengths or fewer than three pairs. Three noncollinear + /// source points are needed to determine all six affine coefficients. The computed + /// source-scatter determinant must be positive and normal, and every fitted coefficient must + /// narrow to finite `f32`. These checks can reject an exactly full-rank input or accept an + /// exactly singular one because the scatter and determinant have already rounded. /// - /// The result is the transform minimizing `sum(|apply(source[i]) - target[i]|^2)` over all - /// affine maps, in closed form: the centred normal equations give the linear part as the - /// target-source cross-scatter times the inverse source scatter, and the translation - /// recovers the target centroid from the mapped source centroid. Every moment accumulates - /// serially in double precision. The fit reads gauge-population constellations, whose size - /// sits far below the parallel Procrustes fold's break-even, so no parallel form exists - /// until a corpus-scale consumer does. + /// # Complexity /// - /// Returns [`None`] when the field lengths differ, the caller passes fewer than three pairs - /// (six coefficients need three correspondences, and two points are always collinear), the - /// source scatter's determinant is not a normal positive number (coincident or collinear - /// source points collapse an axis, leaving no invertible linear part), - /// or a fitted coefficient leaves the finite `f32` range. A nearly collinear source - /// constellation conditions the solve poorly and the coefficients grow accordingly, exactly - /// as a near-singular matrix inflates its inverse. + /// O(n) time and constant additional storage for n pairs. /// - /// # Examples + /// # Example /// - /// The example is `ignore`d because a doctest compiles as an external consumer of the crate, - /// which cannot name this crate-internal function. The same fixture runs compiled in the - /// module's test suite. + /// This example is ignored because [`Transform`] and [`FinitePointField`] are crate-private. /// /// ```ignore + /// use hashql_core::id::IdSlice; + /// use crate::math::{FinitePointField, Transform, Vec2}; + /// # hashql_core::id::newtype! { struct RowId(u32) } + /// /// let expected = Transform::from_cols( /// Vec2::new(2.0, 0.0), /// Vec2::new(0.0, 0.5), @@ -85,9 +96,9 @@ impl Transform { let mut cross_yy = 0.0_f64; for (&source, &target) in source.iter().zip(target.iter()) { - // `f32` values widen exactly and each product of two widened values fits in `f64`'s - // 53-bit significand, so only the running additions round - the same exactness the - // Procrustes accumulation relies on for its centred-moment cancellation. + // Finite f32 values widen exactly. Their products need at most 48 significand bits and + // fit f64's exponent range. Each moment update therefore rounds only when adding to the + // accumulator. Subsequent centring can still cancel most of the significand bits. let source = DVec2::from(source); let target = DVec2::from(target); @@ -106,9 +117,10 @@ impl Transform { let source_centroid = source_sum / count; let target_centroid = target_sum / count; - // Centred moments follow from the raw ones by the parallel-axis identity, exactly as in - // the Procrustes solve: expanding each centred product leaves cross terms that collapse - // into one correction because the centred source sums to zero. + // In real arithmetic, centred deviations sum to zero. Writing mₚ = Σᵢ pᵢ and m_q = Σᵢ qᵢ + // gives S = Σᵢ pᵢpᵢᵀ − mₚmₚᵀ/n and C = Σᵢ qᵢpᵢᵀ − m_qmₚᵀ/n. Therefore the raw moments + // determine both centred matrices without another input pass. Subtraction uses rounded + // sums, and converting n to f64 can round above 2⁵³. let scatter_xx = source_xx - source_sum.x() * source_sum.x() / count; let scatter_xy = source_xy - source_sum.x() * source_sum.y() / count; let scatter_yy = source_yy - source_sum.y() * source_sum.y() / count; @@ -117,9 +129,11 @@ impl Transform { let centred_yx = cross_yx - target_sum.y() * source_sum.x() / count; let centred_yy = cross_yy - target_sum.y() * source_sum.y() / count; - // The scatter matrix is positive semidefinite, so a mathematically singular determinant - // can only round to a small value of either sign; the sign check rejects it together - // with the non-normal cases. + // The exact source scatter is positive semidefinite and is invertible precisely for + // noncollinear points. Rounded moments need not retain that property. A singular scatter + // can leave a positive normal determinant, including through the residual of the fused + // product minus the separately rounded square. This check rejects nonpositive and + // non-normal computed values only. let determinant = scatter_xx.mul_add(scatter_yy, -(scatter_xy * scatter_xy)); if !determinant.is_normal() || determinant <= 0.0 { return None; @@ -146,19 +160,24 @@ impl Transform { /// Returns the root-mean-square distance from transformed source points to their targets. /// - /// Paired with [`fit_uniform`](Self::fit_uniform), the residual measures the movement no - /// affine map explains. This applies the transform with coefficients widened to `f64`, and - /// the squared distances accumulate serially in double precision. + /// For n > 0 pairs, the model is RMS = √(Σᵢ‖Apᵢ + t − qᵢ‖² / n). At the least-squares fit it + /// measures the error left after affine alignment. Coefficients widen to `f64`, and squared + /// distances accumulate serially in `f64`. This can differ from measuring the `f32` outputs of + /// [`apply`](Self::apply). /// - /// The reading is total over the proven-finite fields: the accumulation is bounded far - /// inside `f64`'s range, so no rejection arm exists. The transform's six coefficients must - /// be finite, which the fit produces and - /// [`new_unchecked`](DNonNegative::new_unchecked)'s debug assertion guards. + /// Every transform coefficient must be finite. A successful [`fit_uniform`](Self::fit_uniform) + /// establishes this condition, but arbitrary construction and inverse/composition operations + /// may not. Together with finite field coordinates, this keeps the squared sum and RMS finite + /// on 32-bit and 64-bit targets. /// /// # Panics /// /// This panics when the field lengths differ or the fields are empty, because the residual /// is defined over matched pairs and an empty set has no mean. + /// + /// # Complexity + /// + /// O(n) time and constant additional storage for n pairs. #[must_use] pub(crate) fn rms_residual( self, @@ -198,14 +217,17 @@ impl Transform { squared += residual.norm_squared(); } - // In domain with no check: every coordinate is field-proven finite and every - // coefficient is a finite f32, each below 2^128 in magnitude. A residual component is - // two coefficient-coordinate products (each below 2^128 squared = 2^256) plus a - // translation and a target coordinate, so it stays below 2^258, its square below - // 2^516, a pair's squared distance below 2^517, and a sum of fewer than 2^60 pairs (a - // slice of 8-byte points cannot hold more) below 2^577 - finite in `f64` with room to - // spare, and non-negative as a sum of squares. The quotient by a positive pair count - // and the square root keep both properties. + // Finite f32 coefficients and coordinates have magnitude below 2¹²⁸. Each residual + // component has two products below 2²⁵⁶, plus translation and target terms. Allowing for + // fixed-operation rounding, the nonnegative squared error of one pair is below B = 2⁵¹⁸. + // On 32-bit and 64-bit targets, an 8-byte-point slice contains n < 2⁶⁰ pairs. Adding the + // initial zero is exact for these nonnegative finite terms. Along any term's path, fewer + // than n additions combine contributions. With binary64 unit roundoff u = 2⁻⁵³, those + // additions amplify the term by at most (1 + u)ⁿ < exp(128) < 2¹⁸⁵. Thus the rounded sum is + // below nB · 2¹⁸⁵ < 2⁷⁶³. Subnormal additions cannot threaten this upper bound. + // Division by the positive converted count and the square root preserve finiteness + // and nonnegativity. Therefore the final value satisfies DNonNegative's numerical + // domain. (squared / DNonNegative::from_usize(source.len())) .sqrt() .finish_unchecked() diff --git a/libs/@local/graph/atlas/src/math/transform/mod.rs b/libs/@local/graph/atlas/src/math/transform/mod.rs index 0451dc7a56b..32cab89b2c0 100644 --- a/libs/@local/graph/atlas/src/math/transform/mod.rs +++ b/libs/@local/graph/atlas/src/math/transform/mod.rs @@ -1,4 +1,4 @@ -//! Affine transformations of 2D vectors and batches of them. +//! General affine maps for composition, application and least-squares alignment. use core::simd::Simd; @@ -13,24 +13,28 @@ mod fit; #[cfg(test)] mod tests; -/// An affine transformation of 2D space: scale, rotation, and translation. +/// An affine map of 2D space, including shear, reflection and axis collapse. /// /// A transform maps a vector `p` to `x_axis · p.x + y_axis · p.y + translation`, where `x_axis` and -/// `y_axis` are the columns of a 2x2 linear part. This is the top of the usual 3x3 homogeneous -/// matrix with its constant `[0 0 1]` bottom row omitted, so a transform stores six coefficients -/// rather than nine. Perspective is intentionally out of scope; every representable transform keeps -/// parallel lines parallel. +/// `y_axis` are the columns of a 2x2 linear part. In the usual 3x3 homogeneous matrix, the six +/// stored coefficients form the first two rows and the constant bottom row is `[0 0 1]`. This model +/// includes anisotropic scale, shear and reflection. A singular linear part can collapse lines or +/// the whole plane. Coefficients accept arbitrary `f32` values, and application rounds and can +/// overflow. /// /// Build transforms from the constructors ([`from_scale`](Self::from_scale), /// [`from_rotation`](Self::from_rotation), [`from_translation`](Self::from_translation), or /// [`from_cols`](Self::from_cols) for the general case) and combine them with [`then`](Self::then), /// which reads in application order. Apply a transform to a single vector with -/// [`apply`](Self::apply) or to a whole [`Vec2x4T`] batch with [`apply_x4`](Self::apply_x4), which -/// stays entirely in SIMD registers. +/// [`apply`](Self::apply) or to a whole [`Vec2x4T`] batch with [`apply_x4`](Self::apply_x4). /// -/// # Examples +/// # Example: scaling before translation +/// +/// This example is ignored because [`Transform`] is crate-private. /// /// ```ignore +/// use crate::math::{Transform, Vec2}; +/// /// // Scale by 2 around the origin, then move 10 to the right. /// let transform = Transform::from_scale(Vec2::new(2.0, 2.0)) /// .then(Transform::from_translation(Vec2::new(10.0, 0.0))); @@ -38,9 +42,14 @@ mod tests; /// assert_eq!(transform.apply(Vec2::new(3.0, 4.0)), Vec2::new(16.0, 8.0)); /// ``` /// -/// Rotations are exact only where sine and cosine are, so compare with a tolerance: +/// # Example: applying a rotation +/// +/// Trigonometric coefficients and application can round. Compare the result with a tolerance. This +/// example is ignored because [`Transform`] is crate-private. /// /// ```ignore +/// use crate::math::{Rotation, Transform, Vec2}; +/// /// let quarter_turn = /// Transform::from_rotation(Rotation::from_radians(core::f32::consts::FRAC_PI_2)); /// let rotated = quarter_turn.apply(Vec2::new(1.0, 0.0)); @@ -66,7 +75,7 @@ pub(crate) struct Transform { } impl Transform { - /// The transform that maps every vector to itself. + /// The identity linear map with zero translation. pub(crate) const IDENTITY: Self = Self::from_cols( Vec2::new(1.0, 0.0), Vec2::new(0.0, 1.0), @@ -108,24 +117,28 @@ impl Transform { ) } - /// Creates a transform that moves every vector by `translation`. + /// Creates an identity linear map with the given translation. #[inline] #[must_use] pub(crate) const fn from_translation(translation: Vec2) -> Self { Self::from_cols(Vec2::new(1.0, 0.0), Vec2::new(0.0, 1.0), translation) } - /// Returns the transform equivalent to applying `self` first, then `next`. + /// Composes `self` followed by `next`. /// /// This reads in application order: `scale.then(translate)` scales before it translates. In - /// matrix notation the result is `next · self`. + /// homogeneous matrix notation the model is next · self. Coefficient rounding can make the + /// result differ from sequential application. + /// + /// You can pass [`Rotation`] and [`Translation`] values directly as `next`. /// - /// `next` is anything convertible into a transform, so [`Rotation`] and [`Translation`] values - /// compose directly without widening at the call site. + /// # Example /// - /// # Examples + /// This example is ignored because [`Transform`] is crate-private. /// /// ```ignore + /// use crate::math::{Rotation, Transform, Vec2, translation::Translation}; + /// /// let transform = Transform::from_scale(Vec2::new(2.0, 2.0)) /// .then(Translation::new(10.0, 0.0)) /// .then(Rotation::from_radians(core::f32::consts::PI)); @@ -146,7 +159,7 @@ impl Transform { ) } - /// Transforms a single vector. + /// Applies the linear part and translation with separate `f32` products and sums. #[inline] #[must_use] pub(crate) const fn apply(self, vec: Vec2) -> Vec2 { @@ -158,20 +171,20 @@ impl Transform { ) } - /// Transforms four vectors at once, entirely in SIMD registers. + /// Applies the affine map to four vectors with SIMD arithmetic. /// - /// Each coefficient is splat across a [`Simd`](Simd) lane group and combined with the - /// batch's axis groups, so the whole transformation is two fused multiply-adds per axis with no - /// shuffles. Transform batches in the [`Vec2x4T`] layout inside hot loops. + /// Each axis uses two fused multiply-adds. Each fused operation rounds its product and addition + /// once, independently of native FMA availability. Grouping and fusion differ from + /// [`apply`](Self::apply), with no uniform result-relative ULP bound between paths, especially + /// near cancellation or overflow. /// - /// On targets with native FMA each per-axis result takes a single rounding per multiply-add, so - /// it can differ from [`apply`](Self::apply) by a few units in the last place of the - /// intermediate terms, and by many units in the last place of the result itself where the terms - /// cancel. + /// # Example /// - /// # Examples + /// This example is ignored because [`Transform`] is crate-private. /// /// ```ignore + /// use crate::math::{Transform, Vec2, Vec2x4T}; + /// /// let batch = Vec2x4T::from([ /// Vec2::new(1.0, 1.0), /// Vec2::new(2.0, 1.0), @@ -212,20 +225,25 @@ impl Transform { ) } - /// Returns the transform that undoes `self`, when one exists. + /// Forms an approximate inverse using the computed determinant. /// - /// The result maps every output of [`apply`](Self::apply) back to its input, up to - /// floating-point rounding. The rounding grows with the condition of the linear part: a - /// transform close to collapsing an axis inverts with proportionally amplified error. + /// For a nonsingular linear part A and translation t, the inverse model is A⁻¹p − A⁻¹t. This + /// computes A⁻¹ from its adjugate and a reciprocal determinant. Near-singular A amplifies + /// errors, and application can already have lost information that inversion cannot recover. /// - /// Returns [`None`] when the determinant of the linear part is zero, subnormal, or not finite, - /// in which case no usable inverse exists. Note that [`Rotation::inverse`] and - /// [`Translation::inverse`] are infallible and exact; prefer them when you know the transform - /// kind. + /// Returns [`None`] when the computed `f32` determinant is zero, subnormal or non-finite. This + /// check does not establish exact invertibility: product rounding can leave a normal + /// determinant for a singular matrix or reject an invertible matrix. Even [`Some`] can contain + /// non-finite coefficients or translation after overflow. Use [`Rotation::inverse`] or + /// [`Translation::inverse`] when the narrower model applies. /// - /// # Examples + /// # Example + /// + /// This example is ignored because [`Transform`] is crate-private. /// /// ```ignore + /// use crate::math::{Transform, Vec2}; + /// /// let transform = Transform::from_scale(Vec2::new(2.0, 4.0)) /// .then(Transform::from_translation(Vec2::new(10.0, -2.0))); /// let inverse = transform.inverse().expect("scale is non-zero"); diff --git a/libs/@local/graph/atlas/src/math/transform/tests.rs b/libs/@local/graph/atlas/src/math/transform/tests.rs index 41f3aaca052..bddfb883453 100644 --- a/libs/@local/graph/atlas/src/math/transform/tests.rs +++ b/libs/@local/graph/atlas/src/math/transform/tests.rs @@ -14,12 +14,18 @@ use crate::math::{ }; hashql_core::id::newtype! { - /// The fit-comparison tests' row domain. + /// Row identifiers for paired affine-fitting fixtures. + /// #[id(const)] struct PairId(u32) } -/// Proves a fixture's points finite over the tests' row domain. +/// Validates fixture coordinates over the tests' row domain. +/// +/// # Panics +/// +/// Panics when a point is non-finite, including when locating it requires an unrepresentable row +/// ID. #[track_caller] fn field(points: &[Vec2]) -> &FinitePointField { FinitePointField::new(IdSlice::from_raw(points)).expect("the fixture points are finite") @@ -85,9 +91,7 @@ fn apply_x4_matches_scalar_apply() { let batch = transform.apply_x4(Vec2x4T::from(POINTS)); - // FMA fuses the rounding of multiply and add, so the SIMD path may - // differ from the scalar path by a few units in the last place of - // the intermediate terms. + // SIMD fusion and grouping differ from the scalar expression for (index, point) in POINTS.into_iter().enumerate() { assert_vec2_close(batch.get(index), transform.apply(point)); } @@ -157,10 +161,11 @@ fn then_widens_rotation_and_translation() { assert_vec2_close(transform.apply(Vec2::new(3.0, 4.0)), Vec2::new(-8.0, 7.0)); } -/// A well-conditioned transform. +/// Generates bounded transforms with invertible linear parts. /// -/// Per-axis scale magnitudes in `0.1..10` (condition number at most 100), an arbitrary rotation, -/// and a translation bounded to `-1e3..1e3`. +/// Per-axis scale magnitudes lie in `0.1..10`, the angle in `-16..16` radians and each translation +/// component in `-1e3..1e3`. Before coefficient rounding, the scale ratio bounds the linear part's +/// condition number by 100. fn transform_strategy() -> impl Strategy { ( 0.1_f32..10.0, @@ -185,18 +190,20 @@ fn transform_strategy() -> impl Strategy { ) } -/// A point with coordinates bounded to the well-conditioned `-1e3..1e3` range. +/// Generates points with coordinates in `-1e3..1e3`. fn point_strategy() -> impl Strategy { (-1e3_f32..1e3, -1e3_f32..1e3).prop_map(|(x, y)| Vec2::new(x, y)) } /// Asserts two points agree up to a magnitude-scaled tolerance. /// -/// The tolerance scales with the magnitude of the values flowing through the transforms under test. +/// The absolute allowance is 128 · EPSILON · max(magnitude, 1), using [`f32::EPSILON`]. +/// Cancellation can leave a result much smaller than its intermediate terms. The supplied scale +/// estimates those terms rather than the final result alone. /// -/// Intermediate coordinates reach the order of `magnitude`, and cancellation can leave a result far -/// smaller than the values that produced it, so the tolerance scales with the inputs' magnitude -/// rather than the result's. +/// # Panics +/// +/// Panics when either coordinate comparison fails. #[track_caller] fn assert_close_at_magnitude(actual: Vec2, expected: Vec2, magnitude: f32) { let tolerance = 128.0 * f32::EPSILON * magnitude.max(1.0); @@ -208,10 +215,6 @@ fn assert_close_at_magnitude(actual: Vec2, expected: Vec2, magnitude: f32) { ); } -/// A well-conditioned transform's inverse round-trips points. -/// -/// `inverse().apply(apply(p)) == p` up to rounding amplified by the bounded (at most 100) condition -/// of the linear part. #[property_test] fn inverse_round_trips_arbitrary_points( #[strategy = transform_strategy()] transform: Transform, @@ -221,16 +224,13 @@ fn inverse_round_trips_arbitrary_points( .inverse() .expect("scales bounded away from zero keep the determinant normal"); - // The forward image reaches |p| · 10 + 1e3; the inverse multiplies - // the rounding by up to another factor of 10. + // each translation component is bounded by 10³, and forward/inverse linear scale magnitudes are + // bounded near 10. The coordinatewise tolerance scale allows a factor of 100 on ‖p‖ and 10⁴ for + // translation. let magnitude = point.length().get().mul_add(100.0, 1e4); assert_close_at_magnitude(inverse.apply(transform.apply(point)), point, magnitude); } -/// Composition distributes over application. -/// -/// `a.then(b).apply(p) == b.apply(a.apply(p))` up to rounding scaled by the intermediate -/// coordinates' magnitude. #[property_test] fn then_matches_sequential_application_on_arbitrary_transforms( #[strategy = transform_strategy()] first: Transform, @@ -240,14 +240,16 @@ fn then_matches_sequential_application_on_arbitrary_transforms( let composed = first.then(second).apply(point); let sequential = second.apply(first.apply(point)); - // The first image reaches |p| · 10 + 1e3, the second another - // factor of 10 plus 1e3. + // the tolerance scale grows with the two linear scale factors and both translations let magnitude = point.length().get().mul_add(100.0, 1.1e4); assert_close_at_magnitude(composed, sequential, magnitude); } -/// The dyadic anisotropic fixture lands every fitted coefficient on an exactly representable -/// value, so the recovery asserts an exact contract. +/// Fits an anisotropic map on symmetric axis points. +/// +/// The source centroid is zero and its scatter is 2I. The target-source cross-scatter is diag(4, +/// 1). Dividing by the source scatter recovers diag(2, 1/2), and the target centroid is the +/// translation (1, −2). These small dyadic operations and their determinant are exact. #[test] fn fit_recovers_an_exact_anisotropic_map() { let expected = Transform::from_cols( @@ -273,8 +275,6 @@ fn fit_recovers_an_exact_anisotropic_map() { ); } -/// The affine fit subsumes the similarity family: on exactly similar data it recovers the -/// similarity's own transform, coefficient for coefficient. #[test] fn fit_recovers_an_exact_similarity() { let source = [ @@ -303,9 +303,11 @@ fn fit_recovers_an_exact_similarity() { ); } -/// A trace-free deformation - one axis contracted, the other expanded - leaves the similarity -/// fit a residual while the affine fit absorbs it whole. The readings are exact: the -/// similarity's best scale on this square is `1.25` and every point misses it by `0.75`. +/// Separates uniform scale from an anisotropic deformation. +/// +/// The target map is diag(2, 1/2) = (5/4)I + diag(3/4, −3/4). The source scatter is 2I, giving the +/// best similarity scale 5/4 and identity rotation. The trace-free remainder moves every unit-axis +/// point by exactly 3/4. The affine fit absorbs both parts and has zero residual. #[test] fn fit_absorbs_the_deformation_a_similarity_cannot() { let source = [ @@ -339,9 +341,11 @@ fn fit_absorbs_the_deformation_a_similarity_cannot() { ); } -/// Collinear source points collapse an axis of the scatter, and the fit refuses them the way -/// the inverse refuses a collapsed transform. The diagonal fixture's determinant cancels -/// exactly in dyadic arithmetic. +/// Rejects a dyadic collinear fixture and invalid pair counts. +/// +/// Both coordinates follow the same sequence 0, 1, 2, 3. The centred scatter has every entry equal +/// to 5, and its determinant is exactly 25 − 25 = 0. This fixture avoids the rounding residuals +/// that can affect other singular inputs. #[test] fn fit_refuses_degenerate_sources() { let collinear = [ diff --git a/libs/@local/graph/atlas/src/math/translation/mod.rs b/libs/@local/graph/atlas/src/math/translation/mod.rs index 85d5dd4909e..defe0ea62b3 100644 --- a/libs/@local/graph/atlas/src/math/translation/mod.rs +++ b/libs/@local/graph/atlas/src/math/translation/mod.rs @@ -1,4 +1,4 @@ -//! Translations of 2D space. +//! Fixed offsets for moving points and composing translations. use core::simd::Simd; @@ -9,13 +9,18 @@ mod tests; /// A translation of 2D space by a fixed offset. /// -/// A `translation: Translation` in a signature promises that the value moves points and composes by -/// adding offsets. Composition via [`then`](Self::then) adds the offsets, and -/// [`inverse`](Self::inverse) negates them, which is exact: no rounding occurs at all. +/// With offset t, application computes p + t. [`then`](Self::then) adds offsets in application +/// order. Both operations round in `f32` and can overflow. [`inverse`](Self::inverse) negates +/// finite offsets without rounding, but cannot recover information lost during application. +/// Construction accepts arbitrary components, including non-finite ones. /// -/// # Examples +/// # Example +/// +/// This example is ignored because the math module is crate-private. /// /// ```ignore +/// use crate::math::{Vec2, translation::Translation}; +/// /// let right = Translation::new(10.0, 0.0); /// let up = Translation::new(0.0, 2.0); /// @@ -43,7 +48,7 @@ mod tests; pub struct Translation(Vec2); impl Translation { - /// The translation that moves nothing. + /// The zero offset. pub const IDENTITY: Self = Self(Vec2::new(0.0, 0.0)); /// Creates a translation from its `x` and `y` offsets. @@ -60,10 +65,11 @@ impl Translation { self.0 } - /// Returns the translation equivalent to applying `self` first, then `next`. + /// Adds the offsets to compose `self` followed by `next`. /// - /// Translations commute, so the order only matters for consistency with the other transform - /// types. This adds the two offsets. + /// Real-arithmetic translations commute. The composed offset rounds once per component, while + /// sequential application rounds after each offset. Composed and sequential evaluations can + /// differ. #[inline] #[must_use] pub const fn then(self, next: Self) -> Self { @@ -72,8 +78,9 @@ impl Translation { /// Returns the translation by the negated offset. /// - /// Negation is exact, so a translation followed by its inverse reproduces the input bit for bit - /// whenever the intermediate sum is exactly representable. + /// Negation of finite offsets is exact. For finite input and offset, an exactly representable + /// intermediate sum allows the reverse addition to recover the numerical input value. This does + /// not promise preservation of a zero's sign. #[inline] #[must_use] pub const fn inverse(self) -> Self { @@ -87,7 +94,7 @@ impl Translation { Vec2::new(vec.x() + self.0.x(), vec.y() + self.0.y()) } - /// Moves four vectors at once, entirely in SIMD registers. + /// Adds the offset to four vectors with lane-wise SIMD arithmetic. #[inline] #[must_use] pub fn apply_x4(self, batch: Vec2x4T) -> Vec2x4T { diff --git a/libs/@local/graph/atlas/src/math/translation/tests.rs b/libs/@local/graph/atlas/src/math/translation/tests.rs index 0d74943c8af..7bf58411fd3 100644 --- a/libs/@local/graph/atlas/src/math/translation/tests.rs +++ b/libs/@local/graph/atlas/src/math/translation/tests.rs @@ -3,6 +3,10 @@ use crate::math::{ tests::{POINTS, SWEEP_POINTS, SWEEP_TRANSLATIONS}, }; +/// Composes and reverses translations on a dyadic fixture. +/// +/// Offsets and points are small multiples of 1/2. Every sum fits exactly in `f32`, including the +/// inverse round trip. #[test] fn translation_composes_and_inverts_exactly() { let translation = Translation::new(10.0, -2.5).then(Translation::new(0.5, 4.0)); @@ -31,12 +35,11 @@ fn translation_apply_x4_matches_apply() { } } -/// `apply_x4` is exact against `apply` over a deterministic sweep of offsets and points. +/// Compares scalar and SIMD translation over the finite fixture sweep. /// -/// The SIMD path is a plain lane-wise `f32` addition with no fused operation to round differently -/// from the scalar path's addition, so every lane must match bit for bit; measured across -/// [`SWEEP_TRANSLATIONS`] and [`SWEEP_POINTS`] (spanning zero, sub-unit and super-unit magnitudes, -/// and mixed signs), the maximum observed distance is 0 ULP in both components. +/// Both paths use one `f32` addition per component, without different fusion or grouping. The +/// assertions compare numerical values. They do not distinguish signed zeros or establish NaN-bit +/// behavior. #[test] fn translation_apply_x4_matches_apply_exactly_over_a_sweep() { for &offset in &SWEEP_TRANSLATIONS { diff --git a/libs/@local/graph/atlas/src/math/vec2/interleaved.rs b/libs/@local/graph/atlas/src/math/vec2/interleaved.rs index 6c49f095847..70356d3706a 100644 --- a/libs/@local/graph/atlas/src/math/vec2/interleaved.rs +++ b/libs/@local/graph/atlas/src/math/vec2/interleaved.rs @@ -1,8 +1,8 @@ //! The natural (array-of-structures) batch of four 2D vectors. //! -//! This layout exists for work that treats vectors as whole units. It matches the memory order -//! of `[Vec2; 4]`, so packing from a borrowed point slice needs no shuffle and an individual -//! vector reads out directly. +//! Matching the memory order of `[Vec2; 4]` permits packing borrowed points without a shuffle and +//! direct access to individual vectors. Use this layout for work that treats vectors as whole +//! units. use core::{ ops::{Add, Index, Mul, Neg, Sub}, @@ -15,14 +15,21 @@ use super::{Vec2, Vec2x4T}; /// Four 2D vectors packed in natural (array-of-structures) order. /// /// This layout keeps each vector whole and interleaves the components as `x0 y0 x1 y1 x2 y2 x3 y3`, -/// the memory order of a four-element `Vec2` array. Packing from `[Vec2; 4]` therefore needs no +/// the memory order of a four-element [`Vec2`] array. Packing from `[Vec2; 4]` therefore needs no /// shuffle, [`get`](Self::get) reads an individual vector directly, and the type's alignment /// satisfies [`Simd`](Simd). Use this layout when operations treat vectors as whole /// units. For axis-independent arithmetic, convert to [`Vec2x4T`]. /// -/// # Examples +/// Arithmetic acts component-wise, with scalar multiplication scaling every component. Indexing +/// selects one of the four vectors and panics at indices of four or more. +/// +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::{Vec2, vec2::Vec2x4}; +/// /// let batch = Vec2x4::from([ /// Vec2::new(1.0, 5.0), /// Vec2::new(2.0, 6.0), @@ -58,15 +65,19 @@ impl Vec2x4 { /// /// The middle is a run of whole batches placed where the slice meets this type's alignment. The /// prefix and suffix hold the points before and after it. Concatenating the three parts in - /// order yields the input exactly, so a bulk pass processes the middle four vectors at a time - /// and the edges one by one. + /// order yields the input exactly. Process the middle four vectors at a time and the edges one + /// by one. /// /// The slice's address and length decide where the split falls. Any part may be empty. The /// middle's size affects performance only, never correctness. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::{Vec2, vec2::Vec2x4}; + /// /// let points: Vec = (0..11_u8).map(|i| Vec2::splat(f32::from(i))).collect(); /// let (prefix, batches, suffix) = Vec2x4::from_slice(&points); /// @@ -85,9 +96,11 @@ impl Vec2x4 { /// ``` #[must_use] pub fn from_slice(slice: &[Vec2]) -> (&[Vec2], &[Self], &[Vec2]) { - // SAFETY: `Self` is `repr(C)` over `[Vec2; 4]` with no padding (const-asserted below to - // match `Simd` in size), every bit pattern of four vectors is a valid batch, and - // `align_to` places the middle only at addresses meeting the raised 32-byte alignment. + // SAFETY: `align_to` requires the middle's bytes to be valid instances of its target type. + // `Self` is repr(C, align(32)) over four transparent two-f32 arrays, filling 32 bytes + // without padding or additional validity constraints. `align_to` supplies the aligned + // partition and retains the input borrow. Therefore every complete middle batch is valid + // for the shared view. unsafe { slice.align_to::() } } @@ -100,9 +113,10 @@ impl Vec2x4 { pub fn from_lanes(xs: Simd, ys: Simd) -> Self { // `[x0, x1, x2, x3]` + `[y0, y1, y2, y3]` -> `[x0, y0, x1, y1, x2, y2, x3, y3]` let this = simd_swizzle!(xs, ys, [0, 4, 1, 5, 2, 6, 3, 7]); - // SAFETY: `Simd` is layout-compatible with `[f32; 8]`, and `Self` is `repr(C)` - // over `[Vec2; 4]`, eight `f32`s in the same memory order; the sizes match and every - // bit pattern is a valid `f32`. + // SAFETY: the cast relies on Simd's contiguous array element layout. `Self` is repr(C) over + // four transparent two-f32 arrays with the same interleaved order and no invalid component + // bit patterns. The transmute checks equal sizes. Under that SIMD layout contract, the + // initialized lanes are valid as a batch. unsafe { core::mem::transmute::, Self>(this) } } @@ -126,22 +140,21 @@ impl Vec2x4 { /// Returns all eight components as a single SIMD vector. /// - /// The lane order is the memory order: `x0 y0 x1 y1 x2 y2 x3 y3`. This compiles to a single - /// full-width vector load. + /// The lane order is the memory order: `x0 y0 x1 y1 x2 y2 x3 y3`. #[inline] #[must_use] pub const fn to_simd(self) -> Simd { - // SAFETY: `Simd` is layout-compatible with `[f32; 8]`, and `Self` is `repr(C)` - // over `[Vec2; 4]`, eight `f32`s in the same memory order; the sizes match and `Self` - // meets the SIMD alignment (both const-asserted below). Every bit pattern is a valid - // `f32`, so the reinterpretation is total in both directions. + // SAFETY: the cast relies on Simd's contiguous array element layout. `Self` contains eight + // initialized f32 components in lane order, without padding or additional validity + // constraints. Equal sizes are checked below. Under that SIMD layout contract, these bytes + // are valid as the returned SIMD value. unsafe { core::mem::transmute::>(self) } } /// Returns the component-wise minimum of the two batches. /// - /// NaN components lose. When exactly one operand is NaN in a component, the result takes the - /// other operand's component, following [`f32::min`]. + /// When exactly one operand is NaN in a component, the result takes the other operand's + /// component, following [`SimdFloat::simd_min`](core::simd::num::SimdFloat::simd_min). #[inline] #[must_use] pub fn min(self, other: Self) -> Self { @@ -150,8 +163,8 @@ impl Vec2x4 { /// Returns the component-wise maximum of the two batches. /// - /// NaN components lose. When exactly one operand is NaN in a component, the result takes the - /// other operand's component, following [`f32::max`]. + /// When exactly one operand is NaN in a component, the result takes the other operand's + /// component, following [`SimdFloat::simd_max`](core::simd::num::SimdFloat::simd_max). #[inline] #[must_use] pub fn max(self, other: Self) -> Self { @@ -174,7 +187,7 @@ impl Vec2x4 { /// Folds the batch into the component-wise minimum of its four vectors. /// - /// NaN components lose, following [`f32::min`]. + /// An axis ignores NaNs if any of its components are non-NaN, following [`Self::min`]. #[inline] #[must_use] pub fn reduce_min(self) -> Vec2 { @@ -193,7 +206,7 @@ impl Vec2x4 { /// Folds the batch into the component-wise maximum of its four vectors. /// - /// NaN components lose, following [`f32::max`]. + /// An axis ignores NaNs if any of its components are non-NaN, following [`Self::max`]. #[inline] #[must_use] pub fn reduce_max(self) -> Vec2 { @@ -212,8 +225,7 @@ impl Vec2x4 { /// Deinterleaves the batch into transposed (structure-of-arrays) order. /// - /// One shuffle pays the layout boundary cost. The result exposes the axis lane groups for - /// per-axis arithmetic. + /// The result exposes one lane group per axis for per-axis arithmetic. #[inline] #[must_use] pub fn transpose(self) -> Vec2x4T { @@ -225,7 +237,6 @@ impl Vec2x4 { } } -/// Adds the batches vector-wise: entry `i` of the result is `self[i] + other[i]`. impl Add for Vec2x4 { type Output = Self; @@ -235,7 +246,6 @@ impl Add for Vec2x4 { } } -/// Subtracts the batches vector-wise: entry `i` of the result is `self[i] - other[i]`. impl Sub for Vec2x4 { type Output = Self; @@ -245,7 +255,6 @@ impl Sub for Vec2x4 { } } -/// Negates every vector in the batch. impl Neg for Vec2x4 { type Output = Self; @@ -255,7 +264,6 @@ impl Neg for Vec2x4 { } } -/// Scales every vector in the batch uniformly. impl Mul for Vec2x4 { type Output = Self; @@ -266,7 +274,6 @@ impl Mul for Vec2x4 { } const impl From<[Vec2; 4]> for Vec2x4 { - /// Packs four vectors in their natural interleaved order. #[inline] fn from(vecs: [Vec2; 4]) -> Self { Self(vecs) @@ -281,13 +288,12 @@ const impl From for [Vec2; 4] { } const impl From> for Vec2x4 { - /// Reinterprets eight lanes in `x0 y0 x1 y1 x2 y2 x3 y3` order. #[inline] fn from(lanes: Simd) -> Self { - // SAFETY: `Simd` is layout-compatible with `[f32; 8]`, and `Self` is `repr(C)` - // over `[Vec2; 4]`, eight `f32`s in the same memory order; the sizes match and `Self` - // meets the SIMD alignment (both const-asserted below). Every bit pattern is a valid - // `f32`, so the reinterpretation is total in both directions. + // SAFETY: the cast relies on Simd's contiguous array element layout. `Self` contains eight + // f32 components in the same lane order, with no additional validity constraints. Equal + // sizes are checked below. Under that SIMD layout contract, the initialized lanes are valid + // as a batch. unsafe { core::mem::transmute::, Self>(lanes) } } } @@ -300,7 +306,6 @@ const impl From for Simd { } impl From for Vec2x4 { - /// Interleaves a structure-of-arrays batch back into whole vectors. #[inline] fn from(batch: Vec2x4T) -> Self { batch.transpose() @@ -310,11 +315,6 @@ impl From for Vec2x4 { impl Index for Vec2x4 { type Output = Vec2; - /// Returns a reference to the vector at `index`. - /// - /// # Panics - /// - /// This panics when `index ≥ 4`. #[inline] fn index(&self, index: usize) -> &Vec2 { &self.0[index] diff --git a/libs/@local/graph/atlas/src/math/vec2/mod.rs b/libs/@local/graph/atlas/src/math/vec2/mod.rs index 03ed2ab85e6..334e25f5c1c 100644 --- a/libs/@local/graph/atlas/src/math/vec2/mod.rs +++ b/libs/@local/graph/atlas/src/math/vec2/mod.rs @@ -3,23 +3,22 @@ //! The scalar type is [`Vec2`]. Vectorized code packs four vectors into one of two batch types, //! both 32 bytes and both aligned for [`Simd`](core::simd::Simd): //! -//! - [`Vec2x4`] stores the vectors interleaved as `x0 y0 x1 y1 ...`, the natural memory order of -//! `[Vec2; 4]`, so packing needs no shuffle and [`Vec2x4::get`] reads an individual vector. +//! - [`Vec2x4`] interleaves vectors as `x0 y0 x1 y1 ...`. Matching `[Vec2; 4]`'s natural memory +//! order permits packing without a shuffle and direct vector access through [`Vec2x4::get`]. //! - [`Vec2x4T`] is the transposed layout: all four `x` components followed by all four `y` -//! components. Use this when an operation treats the axes independently, such as distances, -//! bounding boxes, or axis-wise clamping: [`Vec2x4T::xs`] and [`Vec2x4T::ys`] each yield a full -//! [`Simd`](core::simd::Simd) lane group, so per-axis arithmetic runs without shuffles. +//! components. [`Vec2x4T::xs`] and [`Vec2x4T::ys`] each yield a full [`Simd`](core::simd::Simd) lane group for per-axis arithmetic without shuffles. Use this for +//! axis-independent operations such as distances, bounding boxes or axis-wise clamping. //! //! Converting `[Vec2; 4]` into [`Vec2x4T`] performs the deinterleave at that boundary, which is the //! usual tradeoff: pay the shuffle once on entry and keep the hot loop axis-parallel. //! -//! A borrowed point slice splits in place into batches via [`Vec2x4::from_slice`]: a bulk pass then -//! walks the aligned middle four vectors at a time and the unaligned edges one vector at a time, so -//! bulk passes over `&[Vec2]` vectorize without copying. +//! [`Vec2x4::from_slice`] borrows batches from a point slice without copying. Process the aligned +//! middle four vectors at a time and the unaligned edges one vector at a time. //! -//! Because both batch types match [`Simd`](core::simd::Simd) in size and meet its -//! alignment, [`to_simd`](Vec2x4T::to_simd) and the [`From`] conversions compile to a single -//! full-width vector load or store, with no intermediate copy and no split-load penalty. +//! Both batch types convert to eight SIMD lanes in their respective component order. Their size and +//! alignment checks support representation casts, without guaranteeing a particular load width or +//! instruction count. mod interleaved; #[cfg(test)] @@ -39,12 +38,19 @@ use super::{DVec2x4T, NonNegative, dvec2::DVec2, scalar::DNonNegative}; /// `[Vec2; N]` bit-compatible with a flat component buffer in interleaved order, and the zerocopy /// derives expose that reinterpretation without unsafe code. /// -/// Note that [`Hash`] hashes the raw bytes while equality follows `f32` semantics, so `-0.0` and -/// `0.0` compare equal but hash differently. +/// Multiplication of two vectors is component-wise. Use [`Self::dot`] for the scalar product. +/// Indexing selects `x` at zero and `y` at one, and panics for any other index. /// -/// # Examples +/// [`Hash`](core::hash::Hash) supplies the raw bytes to the hasher while equality follows `f32` +/// semantics. Signed zeros compare equal but supply different bytes to hashing. +/// +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::math::Vec2; +/// /// let vec = Vec2::new(1.0, 2.0); /// assert_eq!(vec.x(), 1.0); /// assert_eq!(vec.y(), 2.0); @@ -83,16 +89,20 @@ impl Vec2 { Self([value, value]) } - /// Wraps a borrowed slice in place as consecutive vectors. + /// Views consecutive component pairs as vectors without copying. /// - /// Vector `i` of the returned slice occupies components `2 · i` and `2 · i + 1`, so a row-major - /// `f32[T, 2]` matrix reads as its `T` points without copying. + /// A row-major `f32[T, 2]` matrix becomes a view of its `T` points. Vector `i` occupies + /// components `2 · i` and `2 · i + 1`. /// /// Returns [`None`] unless the length is a whole number of vectors. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::math::Vec2; + /// /// let components = [1.0, 2.0, 3.0, 4.0]; /// let points = Vec2::from_slice(&components).expect("two whole vectors"); /// assert_eq!(points, [Vec2::new(1.0, 2.0), Vec2::new(3.0, 4.0)]); @@ -100,24 +110,17 @@ impl Vec2 { /// ``` #[must_use] pub fn from_slice(components: &[f32]) -> Option<&[Self]> { - // The cast is checked: `Self` is `FromBytes`, `IntoBytes`, and `KnownLayout` over - // `[f32; 2]`, so the byte view reinterprets element-wise with identical layout, and - // an odd component count fails the conversion's size check. <[Self]>::ref_from_bytes(components.as_bytes()).ok() } - /// Wraps a mutable borrowed slice in place as consecutive vectors. + /// Mutably views consecutive component pairs as vectors without copying. /// - /// The mutable counterpart of [`Vec2::from_slice`], with the same component layout. A - /// write through a returned vector rewrites its two components where they stand, so a - /// row-major `f32[T, 2]` matrix mutates as its `T` points without copying. + /// The mutable counterpart of [`Vec2::from_slice`], with the same component layout. Writes + /// through the returned vectors update the original components. /// /// Returns [`None`] unless the length is a whole number of vectors. #[must_use] pub fn from_slice_mut(components: &mut [f32]) -> Option<&mut [Self]> { - // The cast is checked: `Self` is `FromBytes`, `IntoBytes`, and `KnownLayout` over - // `[f32; 2]`, so the byte view reinterprets element-wise with identical layout, and - // the returned borrow inherits the input's exclusive lifetime. <[Self]>::mut_from_bytes(components.as_mut_bytes()).ok() } @@ -150,9 +153,9 @@ impl Vec2 { /// Returns the perpendicular dot product, the `z` component of the 3D cross product. /// - /// The sign tells which side of `self` the other vector lies on. The result is positive when - /// `other` is counterclockwise from `self`, negative when clockwise, and zero when the vectors - /// are parallel. + /// The real determinant x₁y₂ − y₁x₂ is positive for counterclockwise orientation, negative for + /// clockwise orientation and zero for parallel vectors. The returned `f32` approximation can + /// lose this distinction through rounding, underflow or overflow. #[inline] #[must_use] pub const fn perp_dot(self, other: Self) -> f32 { @@ -164,7 +167,7 @@ impl Vec2 { /// Prefer this over [`length`](Self::length) when comparing magnitudes or feeding a squared /// metric. This avoids the square root. /// - /// Overflow escapes to `+∞` and asserts in debug builds, mirroring integer `+`. + /// Both components and the rounded sum of their squares must be finite. #[inline] #[must_use] pub(crate) const fn length_squared(self) -> NonNegative { @@ -172,6 +175,9 @@ impl Vec2 { } /// Returns the length of the vector. + /// + /// The squared length must satisfy [`Self::length_squared`]'s finite-result requirement, even + /// when the final length would fit in `f32`. #[inline] #[must_use] pub(crate) fn length(self) -> NonNegative { @@ -180,8 +186,8 @@ impl Vec2 { /// Returns the squared Euclidean distance to `other`. /// - /// Never NaN and never negative for finite points. Overflow escapes to `+∞` and asserts in - /// debug builds, mirroring integer `+`. + /// Both points and the rounded sum of squared coordinate differences must be finite. Use + /// [`Self::distance_squared_wide`] to cover the full finite `f32` coordinate range. #[inline] #[must_use] pub(crate) const fn distance_squared(self, other: Self) -> NonNegative { @@ -193,8 +199,8 @@ impl Vec2 { /// Returns the Euclidean distance to `other`. /// - /// The square root of [`distance_squared`](Self::distance_squared), and it carries the same - /// escape contract: an escaped `+∞` survives the root. + /// The squared distance must satisfy [`Self::distance_squared`]'s finite-result requirement, + /// even when the final distance would fit in `f32`. #[inline] #[must_use] pub(crate) fn distance(self, other: Self) -> NonNegative { @@ -203,27 +209,25 @@ impl Vec2 { /// Returns the squared Euclidean distance to `other`, accumulated in `f64`. /// - /// Both points widen exactly before the subtraction, so the reading carries no `f32` - /// arithmetic, and every operation rounds separately. This is the one metric of the - /// k-nearest-neighbour readouts: a consumer that compares its own readings against a - /// readout's computes them here, so tie sets never depend on the call site. - /// - /// A squared distance of finite points is finite and non-negative, so the reading returns - /// as [`DNonNegative`]. Finite inputs are the caller's contract. + /// Both points must be finite. Their components widen exactly before subtraction, with every + /// arithmetic operation rounded separately in `f64`. Coordinate differences need not be exact, + /// but the squared distance remains finite and non-negative throughout the finite `f32` input + /// range. #[inline] #[must_use] pub(crate) const fn distance_squared_wide(self, other: Self) -> DNonNegative { - // In domain with no check: each widened coordinate difference of finite `f32` points - // stays below 2¹³⁰ and the sum of their squares below 2²⁶¹, far from `f64` overflow. A - // sum of squares is non-negative. The `new_unchecked` debug assert catches a non-finite - // input. + // Finite f32 coordinates have magnitude below 2¹²⁸. Widened differences have magnitude at + // most 2¹²⁹ and their squared sum at most 2²⁵⁹, far below f64 overflow. Each nonzero + // difference is at least 2⁻¹⁴⁹ in magnitude, also keeping its square in the normal f64 + // range. Therefore the separately rounded sum satisfies DNonNegative's domain. DNonNegative::new_unchecked(DVec2::from(self).distance_squared(DVec2::from(other))) } /// Linearly interpolates from `self` toward `other`. /// - /// At `factor == 0.0` the result is `self`, at `factor == 1.0` it is `other`; values outside - /// `[0, 1]` extrapolate along the same line. + /// Evaluates self + (other − self) · factor component-wise. Factors outside [0, 1] extrapolate. + /// Floating-point rounding can miss the endpoint at factor one, and an overflowing difference + /// can produce a non-finite result even at factors zero or one. #[inline] #[must_use] pub const fn lerp(self, other: Self, factor: f32) -> Self { @@ -232,8 +236,8 @@ impl Vec2 { /// Returns the component-wise minimum of the two vectors. /// - /// NaN components lose. When exactly one operand is NaN in a component, the result takes the - /// other operand's component, following [`f32::min`]. + /// When exactly one operand is NaN in a component, the result takes the other operand's + /// component, following [`f32::min`]. #[inline] #[must_use] pub const fn min(self, other: Self) -> Self { @@ -242,8 +246,8 @@ impl Vec2 { /// Returns the component-wise maximum of the two vectors. /// - /// NaN components lose. When exactly one operand is NaN in a component, the result takes the - /// other operand's component, following [`f32::max`]. + /// When exactly one operand is NaN in a component, the result takes the other operand's + /// component, following [`f32::max`]. #[inline] #[must_use] pub const fn max(self, other: Self) -> Self { @@ -262,7 +266,7 @@ impl Vec2 { /// # Panics /// /// This panics when a component of `low` exceeds the matching component of `high`, or when a - /// bound is NaN, following [`f32::clamp`]. The check runs in every build profile. + /// bound is NaN, following [`f32::clamp`]. #[inline] #[must_use] pub const fn clamp(self, low: Self, high: Self) -> Self { @@ -321,9 +325,6 @@ const impl Neg for Vec2 { } } -/// Component-wise (Hadamard) product. -/// -/// For the scalar product, use [`Vec2::dot`]. const impl Mul for Vec2 { type Output = Self; @@ -407,18 +408,13 @@ const impl From for [f32; 2] { const impl Index for Vec2 { type Output = f32; - /// Returns the component at `index`, where `0` is `x` and `1` is `y`. - /// - /// # Panics - /// - /// This panics when `index ≥ 2`. #[inline] fn index(&self, index: usize) -> &f32 { &self.0[index] } } -/// The SIMD views of a point slice, each splitting the batches from the scalar rest. +/// Batch views and iterators over a point slice. pub(crate) trait Vec2SliceExt { /// Views the slice as aligned interleaved batches between a prefix and a suffix. /// @@ -437,11 +433,10 @@ pub(crate) trait Vec2SliceExt { &[Vec2], ); - /// Iterates transposed four-point batches widened to double precision, beside the widened - /// scalar remainder. + /// Iterates double-precision batches and the widened scalar remainder. /// - /// The widening is exact for every finite `f32` component, so a double-precision - /// accumulation over the batches reads the same points the slice stores. + /// Each batch transposes four consecutive points. Widening is exact for every finite `f32` + /// component. fn iter_transposed_wide( &self, ) -> ( diff --git a/libs/@local/graph/atlas/src/math/vec2/tests.rs b/libs/@local/graph/atlas/src/math/vec2/tests.rs index 8387947a7f8..26d1c6f9edf 100644 --- a/libs/@local/graph/atlas/src/math/vec2/tests.rs +++ b/libs/@local/graph/atlas/src/math/vec2/tests.rs @@ -137,8 +137,9 @@ fn batch_dot_and_distance_match_scalar_lanes() { let distances = lhs.distance_squared(rhs); let lengths = lhs.length_squared(); - // The sample values are exact in f32, so FMA contraction changes - // nothing and the comparison can be exact. + // The coordinates are multiples of 1/4 with magnitude at most 8. These products and sums are + // multiples of 1/16 with magnitude at most 512, exactly representable in f32. Therefore fusion + // leaves the fixture's results unchanged. for lane in 0..4 { assert_eq!(dots[lane], POINTS[lane].dot(other[lane])); assert_eq!(distances[lane], POINTS[lane].distance_squared(other[lane])); @@ -161,8 +162,9 @@ fn batch_perp_dot_matches_scalar_lanes() { let perps = lhs.perp_dot(rhs); let reversed = rhs.perp_dot(lhs); - // The sample values are exact in f32, so FMA contraction changes - // nothing and the comparison can be exact. + // The coordinates are multiples of 1/4 with magnitude at most 8. Each product and difference + // fits exactly in f32. Therefore the fused and separately rounded expressions agree on these + // fixtures. for lane in 0..4 { assert_eq!(perps[lane], POINTS[lane].perp_dot(other[lane])); // The perpendicular product is antisymmetric lane-wise. @@ -202,8 +204,6 @@ fn vec2_index_out_of_bounds() { #[test] fn from_slice_mut_writes_through_to_the_components() { - // The mutable form aliases the same storage, so a write through a vector rewrites its - // components where they stand. let mut components = [1.0, 2.0, 3.0, 4.0]; let points = Vec2::from_slice_mut(&mut components).expect("two whole vectors"); points[1] = Vec2::new(-3.0, -4.0); @@ -292,26 +292,23 @@ fn natural_operators_match_scalar_operators() { } } -/// A coordinate bounded to a well-conditioned range. -/// -/// The laws below are algebraic contracts. The example-based tests above pin overflow behaviour. +/// Generates finite coordinates whose products and squared differences avoid overflow. fn coordinate() -> impl Strategy { -1e5_f32..1e5 } -/// An arbitrary in-range vector. +/// Generates a vector with independently sampled bounded coordinates. fn vec2_strategy() -> impl Strategy { (coordinate(), coordinate()).prop_map(|(x, y)| Vec2::new(x, y)) } -/// Arbitrary in-range vectors, one per batch lane. +/// Generates four vectors with independently sampled bounded coordinates. fn vec2_array_strategy() -> impl Strategy { proptest::array::uniform4(vec2_strategy()) } -/// The dot product commutes bit for bit: both orders multiply and add the same values. -/// -/// Coordinates lie in `-1e5..1e5`. +// Swapping operands preserves each rounded product and their addition order. The bounded +// coordinates avoid overflow, permitting an exact numeric comparison. #[property_test] fn dot_is_commutative( #[strategy = vec2_strategy()] left: Vec2, @@ -320,10 +317,9 @@ fn dot_is_commutative( prop_assert_eq!(left.dot(right), right.dot(left)); } -/// The perpendicular product is antisymmetric. -/// -/// Swapping the operands negates the result exactly, because IEEE negation of a difference is -/// exact. Coordinates lie in `-1e5..1e5`. +// Swapping operands preserves the rounded products and reverses their subtraction. Round-to-nearest +// is symmetric under negation for nonzero finite results, and signed zeros compare equal. Therefore +// the bounded scalar results are numerically antisymmetric. #[property_test] fn perp_dot_is_antisymmetric( #[strategy = vec2_strategy()] left: Vec2, @@ -332,9 +328,6 @@ fn perp_dot_is_antisymmetric( prop_assert_eq!(left.perp_dot(right), -right.perp_dot(left)); } -/// Distance is symmetric, and the distance from a point to itself is exactly zero. -/// -/// Coordinates lie in `-1e5..1e5`. #[property_test] fn distance_is_symmetric_with_zero_self_distance( #[strategy = vec2_strategy()] left: Vec2, @@ -344,11 +337,10 @@ fn distance_is_symmetric_with_zero_self_distance( prop_assert_eq!(left.distance(left), 0.0); } -/// Lerp hits its endpoints. -/// -/// Factor zero is exact; factor one holds up to rounding scaled by the operands' magnitude (the -/// interpolation computes `from + (to - from) · factor`, which rounds twice). Coordinates lie in -/// `-1e5..1e5`. +// Bounded coordinates keep the difference finite. Factor zero reproduces the start numerically, +// while factor one evaluates from + (to − from) with two potentially inexact operations. The +// endpoint comparison allows 8 · f32::EPSILON times the greatest input magnitude, with an absolute +// floor of 8 · f32::EPSILON. #[property_test] fn lerp_hits_endpoints_on_arbitrary_vectors( #[strategy = vec2_strategy()] from: Vec2, @@ -373,9 +365,6 @@ fn lerp_hits_endpoints_on_arbitrary_vectors( ); } -/// Batch arithmetic operators match the scalar operators bit for bit in every lane. -/// -/// SIMD IEEE arithmetic is scalar arithmetic per lane. Coordinates lie in `-1e5..1e5`. #[property_test] fn batch_operators_match_scalar_lanes_on_arbitrary_inputs( #[strategy = vec2_array_strategy()] lhs: [Vec2; 4], @@ -398,10 +387,9 @@ fn batch_operators_match_scalar_lanes_on_arbitrary_inputs( } } -/// Batch reductions match the scalar reductions per lane up to FMA contraction. -/// -/// The contraction's rounding scales with the products' magnitude rather than the (possibly -/// cancelled) result. Coordinates lie in `-1e5..1e5`. +// cancellation can make a dot or perpendicular product much smaller than its terms. Their error +// tolerances use the products' scale to account for different rounding under fused multiply-add. +// Squared sums have no cancellation and use the result's magnitude. #[property_test] fn batch_reductions_match_scalar_lanes_on_arbitrary_inputs( #[strategy = vec2_array_strategy()] lhs: [Vec2; 4], @@ -441,10 +429,6 @@ fn batch_reductions_match_scalar_lanes_on_arbitrary_inputs( } } -/// Layout conversions round-trip bit for bit. -/// -/// `[Vec2; 4] -> Vec2x4 -> Vec2x4T -> Vec2x4 -> [Vec2; 4]` reproduces every coordinate's exact -/// bits. #[property_test] fn layout_round_trips_are_bit_exact(#[strategy = vec2_array_strategy()] points: [Vec2; 4]) { let transposed = Vec2x4T::from(Vec2x4::from(points)); @@ -456,12 +440,6 @@ fn layout_round_trips_are_bit_exact(#[strategy = vec2_array_strategy()] points: } } -/// The tests the `miri` nextest profile selects. -/// -/// Each test here drives a path that reinterprets or realigns memory. That covers slice views over -/// component storage, the layout conversions between the interleaved and transposed batches, the -/// reductions over the natural layout, and lane extraction with its inverse. The profile selects by -/// module path, so moving a test in or out of this module is the whole edit. mod miri { use core::simd::Simd; @@ -519,7 +497,6 @@ mod miri { let window = &points[offset..]; let (prefix, batches, suffix) = Vec2x4::from_slice(window); - // Every point lands in exactly one part, in order. assert_eq!( prefix.len() + 4 * batches.len() + suffix.len(), window.len(), diff --git a/libs/@local/graph/atlas/src/math/vec2/transposed.rs b/libs/@local/graph/atlas/src/math/vec2/transposed.rs index c2563ea7d6a..c667fb5de2f 100644 --- a/libs/@local/graph/atlas/src/math/vec2/transposed.rs +++ b/libs/@local/graph/atlas/src/math/vec2/transposed.rs @@ -1,8 +1,7 @@ //! The transposed (structure-of-arrays) batch of four 2D vectors. //! -//! This layout exists for axis-independent arithmetic. Each axis's four components form one -//! lane group, so per-axis operations run without shuffles in the hot loop, and the -//! deinterleave from natural order is paid once at conversion. +//! Each axis's four-component lane group supports per-axis arithmetic without shuffles. Conversion +//! from natural order performs the deinterleave once. use core::{ ops::{Add, Mul, Neg, Sub}, @@ -16,18 +15,22 @@ use crate::math::{dvec2::DVec2x4T, kernel::mul_add_f32x4, scalar::DNonNegative}; /// Four 2D vectors packed in transposed (structure-of-arrays) order. /// /// Storage places all four `x` values before all four `y` values: `x0 x1 x2 x3 y0 y1 y2 y3`. The -/// value is aligned for [`Simd`](Simd), and [`xs`](Self::xs) and [`ys`](Self::ys) each -/// return a full [`Simd`](Simd) lane group, so axis-independent arithmetic over the batch -/// needs no shuffles. +/// value is aligned for [`Simd`](Simd). [`xs`](Self::xs) and [`ys`](Self::ys) each return a +/// full [`Simd`](Simd) lane group for axis-independent arithmetic without shuffles. /// -/// Construct a batch from `[Vec2; 4]` via [`From`]; that conversion performs the deinterleave from -/// the vectors' natural memory order. After per-axis arithmetic, reassemble a batch with -/// [`from_lanes`](Self::from_lanes). +/// Construct a batch from `[Vec2; 4]` via [`From`] to deinterleave the vectors' natural memory +/// order. After per-axis arithmetic, reassemble a batch with [`from_lanes`](Self::from_lanes). +/// Arithmetic operators act component-wise, with scalar multiplication scaling every component. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private and uses nightly portable SIMD. /// /// ```ignore /// # #![feature(portable_simd)] +/// use core::simd::Simd; +/// +/// use crate::math::{Vec2, Vec2x4T}; /// /// let batch = Vec2x4T::from([ /// Vec2::new(1.0, 5.0), @@ -42,7 +45,6 @@ use crate::math::{dvec2::DVec2x4T, kernel::mul_add_f32x4, scalar::DNonNegative}; /// // Scale both axes, then repack. /// let scaled = Vec2x4T::from_lanes(batch.xs() * Simd::splat(2.0), batch.ys() * Simd::splat(2.0)); /// assert_eq!(scaled.get(0), Vec2::new(2.0, 10.0)); -/// # use core::simd::Simd; /// ``` #[derive( Debug, @@ -59,7 +61,7 @@ use crate::math::{dvec2::DVec2x4T, kernel::mul_add_f32x4, scalar::DNonNegative}; pub struct Vec2x4T([f32; 8]); impl Vec2x4T { - /// Replicates a `Vec2` value across all lanes of `Self`. + /// Creates a batch holding four copies of `value`. pub const fn splat(value: Vec2) -> Self { Self([ value.x(), @@ -81,9 +83,10 @@ impl Vec2x4T { #[must_use] pub const fn from_lanes(xs: Simd, ys: Simd) -> Self { let this = [xs, ys]; - // SAFETY: `[Simd; 2]` lays out the `x` lane group followed by the `y` lane - // group, exactly `Self`'s `repr(C)` `[f32; 8]` memory order; sizes match and every - // bit pattern is a valid `f32`. + // SAFETY: the cast relies on each four-lane Simd having its array element layout. The + // source array places the initialized x group before the y group, matching Self's f32 + // array. The transmute checks equal sizes and Self has no additional validity constraints. + // Under that SIMD layout contract, the resulting batch is valid. unsafe { core::mem::transmute::<[Simd; 2], Self>(this) } } @@ -101,10 +104,10 @@ impl Vec2x4T { let this = &raw const *self; let this = this.cast::(); - // SAFETY: `Self` is `repr(C)` over `[f32; 8]` whose first four elements are the `x` - // lane group, `Simd` is layout-compatible with `[f32; 4]`, and `Self`'s - // 32-byte alignment satisfies `Simd`'s; the borrow covers bytes owned by - // `self` and inherits its lifetime. + // SAFETY: the reference relies on Simd's array element layout. Self's first four f32 + // elements are initialized, and its 32-byte alignment meets the half-width alignment bound + // checked below. The pointer retains self's provenance and shared borrow lifetime. Under + // that SIMD layout contract, the x group is valid for the returned shared reference. unsafe { &*this.cast::>() } } @@ -122,9 +125,11 @@ impl Vec2x4T { let this = &raw const *self; let this = this.cast::(); - // SAFETY: elements `4..8` of `Self`'s `repr(C)` `[f32; 8]` storage are the `y` lane - // group; the 16-byte offset from the 32-byte-aligned base satisfies `Simd`'s - // alignment, and the borrow covers bytes owned by `self` and inherits its lifetime. + // SAFETY: the reference relies on Simd's array element layout. Adding four f32 elements + // stays within self's allocation and selects its initialized y group. The 16-byte offset + // from a 32-byte-aligned base meets the half-width alignment bound checked below. The + // pointer retains self's provenance and shared borrow lifetime. Under that SIMD layout + // contract, the y group is valid for the returned shared reference. unsafe { &*this.add(4).cast::>() } } @@ -140,9 +145,10 @@ impl Vec2x4T { reason = "the suggested `From` conversion is not const-callable" )] pub const fn into_lanes(self) -> (Simd, Simd) { - // SAFETY: `Self` is `repr(C)` over `[f32; 8]`, the `x` lane group followed by the `y` - // lane group, exactly `[Simd; 2]`'s memory order; sizes match and every bit - // pattern is a valid `f32`. + // SAFETY: the cast relies on each four-lane Simd having its array element layout. Self + // contains the initialized x group followed by the y group, without padding. The transmute + // checks equal sizes. Under that SIMD layout contract, both groups are valid as the + // returned SIMD values. let [xs, ys] = unsafe { core::mem::transmute::; 2]>(self) }; (xs, ys) @@ -164,21 +170,21 @@ impl Vec2x4T { /// Returns all eight components as a single SIMD vector. /// - /// The lane order is the memory order: `x0 x1 x2 x3 y0 y1 y2 y3`. This compiles to a single - /// full-width vector load. + /// The lane order is the memory order: `x0 x1 x2 x3 y0 y1 y2 y3`. #[inline] #[must_use] pub const fn to_simd(self) -> Simd { - // SAFETY: `Self` is `repr(C)` over `[f32; 8]`, which is layout-compatible with - // `Simd` (sizes const-asserted below); every bit pattern is a valid `f32`, so - // the reinterpretation is total. + // SAFETY: the cast relies on Simd's contiguous array element layout. Self contains eight + // initialized f32 lanes without padding, and equal sizes are checked below. Under that SIMD + // layout contract, these bytes are valid as the returned SIMD value. unsafe { core::mem::transmute::>(self) } } /// Returns the four pairwise dot products as SIMD lanes. /// - /// Lane `i` holds the dot product of the batches' `i`-th vectors. On targets with native FMA - /// one instruction performs the multiply-add, rounding once instead of twice. + /// Lane `i` approximates the dot product of the batches' `i`-th vectors. The y product rounds + /// first, then the x product and addition are fused with one rounding, including on targets + /// without native FMA. This can differ from [`Vec2::dot`]'s separate roundings. #[inline] #[must_use] pub fn dot(self, other: Self) -> Simd { @@ -187,11 +193,10 @@ impl Vec2x4T { /// Returns the four pairwise perpendicular dot products as SIMD lanes. /// - /// Lane `i` holds the perpendicular dot product of the batches' `i`-th vectors, with the sign - /// semantics of [`Vec2::perp_dot`]: the lane is positive when `other`'s vector is - /// counterclockwise from this batch's and negative when clockwise. Parallel vectors yield zero. - /// On targets with native FMA one instruction performs the multiply-add, rounding once instead - /// of twice. + /// Lane `i` approximates x₁y₂ − y₁x₂ for the batches' `i`-th vectors, with the geometric + /// interpretation of [`Vec2::perp_dot`]. The y₁x₂ product rounds first, then x₁y₂ and + /// subtraction are fused with one rounding. This can leave a nonzero rounding residual even for + /// parallel vectors and does not certify orientation or collinearity. #[inline] #[must_use] pub fn perp_dot(self, other: Self) -> Simd { @@ -200,8 +205,10 @@ impl Vec2x4T { /// Returns the four pairwise squared Euclidean distances as SIMD lanes. /// - /// Lane `i` holds the squared distance between the batches' `i`-th vectors. On targets with - /// native FMA one instruction performs the multiply-add, rounding once instead of twice. + /// Lane `i` approximates the squared distance between the batches' `i`-th vectors. Coordinate + /// differences and the squared y difference round first, then the squared x difference and + /// addition are fused with one rounding. Finite coordinates can still overflow during the + /// calculation. #[inline] #[must_use] pub fn distance_squared(self, other: Self) -> Simd { @@ -213,19 +220,19 @@ impl Vec2x4T { /// Returns the four pairwise squared Euclidean distances, accumulated in `f64`. /// - /// Reading `i` equals `self[i].distance_squared_wide(other[i])` bit for bit: the widened - /// lanes subtract, square, and sum with the scalar metric's separate roundings, so a lane - /// readout and a scalar readout select the same rows under the same ties. Unlike - /// [`distance_squared`](Self::distance_squared), nothing fuses. Finite inputs are the - /// caller's contract, as for the scalar form. + /// Both batches must contain finite points. Widening before subtraction covers the full finite + /// `f32` coordinate range with finite, non-negative results. Each lane uses the separately + /// rounded expression of [`Vec2::distance_squared_wide`], rather than the fused expression of + /// [`Self::distance_squared`]. #[inline] #[must_use] pub(crate) fn distance_squared_wide(self, other: Self) -> [DNonNegative; 4] { let readings = DVec2x4T::from(self).distance_squared(DVec2x4T::from(other)); <[f64; 4]>::from(readings).map(|reading| { - // In domain with no check: the scalar metric's own bound applies per component, and - // a sum of squares is non-negative. + // Finite f32 coordinates give f64 differences of magnitude at most 2¹²⁹. The squared + // sum is non-negative and at most 2²⁵⁹, within f64's range. Therefore each reading + // satisfies DNonNegative's domain. DNonNegative::new_unchecked(reading) }) } @@ -239,7 +246,7 @@ impl Vec2x4T { /// Interleaves the batch back into natural (array-of-structures) order. /// - /// One shuffle pays the layout boundary cost. The result stores whole vectors again. + /// The result stores each vector's x and y components consecutively. #[inline] #[must_use] pub fn transpose(self) -> Vec2x4 { @@ -251,8 +258,6 @@ impl Vec2x4T { } } -/// Adds the batches vector-wise: the result's `i`-th vector is the sum of the operands' `i`-th -/// vectors. impl Add for Vec2x4T { type Output = Self; @@ -262,8 +267,6 @@ impl Add for Vec2x4T { } } -/// Subtracts the batches vector-wise: the result's `i`-th vector is the difference of the operands' -/// `i`-th vectors. impl Sub for Vec2x4T { type Output = Self; @@ -273,7 +276,6 @@ impl Sub for Vec2x4T { } } -/// Negates every vector in the batch. impl Neg for Vec2x4T { type Output = Self; @@ -283,7 +285,6 @@ impl Neg for Vec2x4T { } } -/// Scales every vector in the batch uniformly. impl Mul for Vec2x4T { type Output = Self; @@ -294,7 +295,6 @@ impl Mul for Vec2x4T { } impl From<[Vec2; 4]> for Vec2x4T { - /// Deinterleaves four vectors into structure-of-arrays order. #[inline] fn from(vecs: [Vec2; 4]) -> Self { let this = Vec2x4::from(vecs); @@ -303,12 +303,12 @@ impl From<[Vec2; 4]> for Vec2x4T { } const impl From> for Vec2x4T { - /// Reinterprets eight lanes in `x0 x1 x2 x3 y0 y1 y2 y3` order. #[inline] fn from(lanes: Simd) -> Self { - // SAFETY: `Simd` is layout-compatible with `[f32; 8]`, `Self`'s `repr(C)` - // storage (sizes const-asserted below); every bit pattern is a valid `f32`, so the - // reinterpretation is total. + // SAFETY: the cast relies on Simd's contiguous array element layout. Self contains eight + // f32 components in lane order, with no additional validity constraints, and equal sizes + // are checked below. Under that SIMD layout contract, the initialized lanes are valid as a + // batch. unsafe { core::mem::transmute::, Vec2x4T>(lanes) } } } @@ -321,16 +321,15 @@ const impl From for Simd { } impl From for Vec2x4T { - /// Deinterleaves an array-of-structures batch by axis. #[inline] fn from(batch: Vec2x4) -> Self { batch.transpose() } } -// The batch must be usable as backing storage for `Simd`, which requires identical size -// and at least its alignment. The `align(32)` supplies that alignment. The lane views borrow -// `Simd` groups at byte offsets 0 and 16, so the half-width alignment must not exceed 16. +// The batch must match `Simd`'s size and meet its alignment, supplied by `align(32)`. +// Borrowed `Simd` groups begin at byte offsets 0 and 16. Their alignment must not exceed 16 +// bytes to keep both group addresses aligned. const _: () = assert!(align_of::>() <= 16); const _: () = assert!(size_of::() == size_of::>()); const _: () = assert!(align_of::() >= align_of::>()); diff --git a/libs/@local/graph/atlas/src/math/vecn/mod.rs b/libs/@local/graph/atlas/src/math/vecn/mod.rs index 788f4e222f6..c8bcb994a0f 100644 --- a/libs/@local/graph/atlas/src/math/vecn/mod.rs +++ b/libs/@local/graph/atlas/src/math/vecn/mod.rs @@ -1,10 +1,9 @@ //! High-dimensional vectors for embeddings, with SIMD-aligned heap storage. //! -//! [`VecN`] wraps an `[f32; N]` without changing its layout, so borrowed embedding data can be -//! viewed as a vector for free. [`BoxedVecN`] copies a vector into a heap allocation aligned for -//! [`f32x8`] and hands out [`AlignedVecN`] references to it: the alignment guarantees that -//! [`AlignedVecN::lanes`] loads every 8-lane group from an aligned address, never splitting a cache -//! line. +//! [`VecN`] views borrowed embedding arrays without copying. [`BoxedVecN`] provides owned storage +//! aligned for [`f32x8`], and [`AlignedVecN::lanes`] borrows that storage as SIMD groups with a +//! scalar remainder. Alignment constrains addresses without guaranteeing a cache-line size or +//! particular generated load instructions. use alloc::alloc::Global; use core::{ @@ -23,10 +22,9 @@ mod tests; /// An `N`-dimensional vector of `f32` components. /// -/// A [`VecN`] is guaranteed to have the same layout as `[f32; N]`, so borrowed arrays convert in -/// place through [`from_ref`](Self::from_ref) and [`from_mut`](Self::from_mut) without copying. -/// This is the working type for embedding vectors; move one into a [`BoxedVecN`] when SIMD kernels -/// need aligned storage. +/// A [`VecN`] is guaranteed to have the same layout as an array of `N` `f32` components. Borrow +/// arrays in place through [`from_ref`](Self::from_ref) and [`from_mut`](Self::from_mut), without +/// copying. Use [`BoxedVecN`] when SIMD kernels need owned aligned storage. #[derive( Debug, Copy, @@ -49,41 +47,46 @@ impl VecN { Self(components) } - /// Wraps a borrowed array in place, without copying. + /// Views a borrowed component array without copying. #[inline] #[must_use] pub const fn from_ref(value: &[f32; N]) -> &Self { zerocopy::transmute_ref!(value) } - /// Wraps a mutably borrowed array in place, without copying. + /// Mutably views a borrowed component array without copying. #[inline] #[must_use] pub const fn from_mut(value: &mut [f32; N]) -> &mut Self { let ptr = (&raw mut *value).cast::(); - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, so the cast preserves layout - // and validity. The wrapper inherits the mutable borrow unchanged. + // SAFETY: repr(transparent) gives Self the array's size, alignment and validity. The + // pointer derives from an initialized, exclusively borrowed array and retains its + // provenance and lifetime. Therefore the returned mutable reference is valid for that + // borrow. unsafe { &mut *ptr } } - /// Wraps a borrowed slice of arrays in place, without copying. + /// Views a slice of component arrays as vectors without copying. #[inline] #[must_use] pub const fn wrap_slice(values: &[[f32; N]]) -> &[Self] { let data = values.as_ptr().cast::(); - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, so the element layouts are - // identical and the slice reinterprets in place with its length preserved. + // SAFETY: repr(transparent) gives Self the array element's size, alignment and validity. + // The source slice supplies one valid initialized range, including a non-null aligned + // pointer for empty slices or zero-sized elements. The cast preserves its count, provenance + // and shared lifetime. Therefore the returned slice is valid for the source borrow. unsafe { core::slice::from_raw_parts(data, values.len()) } } - /// Wraps a mutably borrowed slice of arrays in place, without copying. + /// Mutably views a slice of component arrays as vectors without copying. #[inline] #[must_use] pub const fn wrap_slice_mut(values: &mut [[f32; N]]) -> &mut [Self] { let data = values.as_mut_ptr().cast::(); - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, so the element layouts are - // identical and the slice reinterprets in place with its length preserved. The wrapper - // inherits the mutable borrow unchanged. + // SAFETY: repr(transparent) gives Self the array element's size, alignment and validity. + // The source slice supplies one valid initialized range, including a non-null aligned + // pointer for empty slices or zero-sized elements. The cast preserves its count, provenance + // and exclusive lifetime. Therefore the returned slice is valid for the source borrow. unsafe { core::slice::from_raw_parts_mut(data, values.len()) } } @@ -107,9 +110,8 @@ impl VecN { /// Reinterprets the vector as SIMD-aligned and mutable, when its address allows. /// - /// Returns [`None`] when the vector does not happen to sit at an address aligned to - /// `align_of::()` bytes. For storage whose alignment comes from construction rather than - /// luck, use [`BoxedVecN`]. + /// Returns [`None`] unless the address is aligned to `align_of::()` bytes. [`BoxedVecN`] + /// provides that alignment at construction. #[inline] #[must_use] pub fn try_as_aligned_mut(&mut self) -> Option<&mut AlignedVecN> { @@ -118,9 +120,9 @@ impl VecN { /// Returns the dot product of the two vectors, accumulated in double precision. /// - /// This sums the products in `f64` and rounds to `f32` once at the end, so the result carries a - /// single rounding regardless of the dimension. A naive single-precision sum instead - /// accumulates error that grows with `N`. + /// Finite components widen exactly and their products fit exactly in `f64`. Summation still + /// rounds in double precision before one final narrowing to `f32`. The result can overflow + /// during that narrowing. A zero-dimensional vector gives zero. #[inline] #[must_use] pub fn dot(&self, other: &Self) -> f32 { @@ -129,20 +131,24 @@ impl VecN { /// Returns the squared Euclidean length, accumulated in double precision. /// - /// The result is non-negative and carries a single rounding to `f32`, like [`dot`](Self::dot). + /// For finite components the result is non-negative, possibly infinity after narrowing, with + /// the rounding behavior of [`Self::dot`]. A zero-dimensional vector gives zero. #[inline] #[must_use] pub fn norm_squared(&self) -> f32 { narrow_accumulated(self.dot_accumulated(self)) } - /// Returns the cosine distance `1 - cos(angle)` between the vectors, in `[0, 2]`. + /// Approximates the cosine distance between finite vectors, clamped to `[0, 2]`. /// - /// Zero at parallel vectors, one at orthogonal vectors, two at opposite vectors. One fused pass - /// computes the dot product and both squared norms with double-precision accumulators. + /// Both vectors must have finite components. The model is 1 − ⟨x, y⟩/(‖x‖‖y‖), with zero at + /// parallel nonzero vectors, one at orthogonal vectors and two at opposite vectors. Dot + /// products and norms accumulate in double precision, but rounding can lose distinctions + /// between nearly parallel vectors. /// - /// The zero vector has no direction: the distance between two zero vectors is zero, and the - /// distance between a zero vector and any other vector is one. + /// The zero vector has no direction. The distance between two zero vectors is defined as zero, + /// and the distance between a zero vector and any other vector as one. The empty pair follows + /// the two-zero-vectors case. #[expect( clippy::float_cmp, reason = "a squared norm is exactly zero precisely for the zero vector; the degenerate \ @@ -195,9 +201,9 @@ impl VecN { /// Returns the dot product with a double-precision vector. /// - /// This is the mixed-precision kernel for optimizers that keep their coefficients in `f64` - /// while the data stays `f32`. Each component widens exactly, and the result stays in full - /// double precision. + /// Finite `f32` components widen exactly. Products with the double-precision coefficients and + /// their sum round in `f64`, with no final narrowing. Finite inputs can still overflow the + /// double-precision calculation. #[inline] #[must_use] pub fn dot_wide(&self, coefficients: &DVecN) -> f64 { @@ -222,18 +228,15 @@ impl VecN { /// Sums the products of the two vectors' components in double precision. /// - /// Each `f32` component widens exactly, the products accumulate in `f64`, and the returned sum - /// carries no narrowing. This is the exact-product kernel behind [`dot`](Self::dot) and the - /// entry kernel of Gram matrices over `f32` data whose downstream arithmetic runs in `f64`. + /// Two finite `f32` significands multiply within 48 bits, and their exponent range fits inside + /// `f64`. Widening before multiplication therefore gives exact products. Accumulating and + /// reducing those products still rounds, and the returned sum carries no final narrowing. The + /// empty sum is zero. /// - /// The fold shape never varies. It takes eight lanes at a time into two interleaved fused - /// accumulators, then one horizontal reduction, then a scalar remainder. Equal inputs therefore - /// reduce to identical bits. - // Lane-width choice: `f64x8` is wider than 128-bit NEON registers, so - // the compiler unrolls it fourfold; with the two independent - // accumulators that keeps sixteen f64 FMA chains in flight, which - // covers the latency-times-throughput product of current cores. On - // AVX-2 the same shape is a two-register unroll. + /// The result depends on the summation grouping, including the SIMD horizontal reduction. It is + /// not a cross-target bitwise reproducibility contract. + // two interleaved eight-lane accumulators expose independent multiply-add chains. The target + // and compiler decide how the groups map to machine registers. #[inline] pub(crate) fn dot_accumulated(&self, other: &Self) -> f64 { let (chunks_left, remainder_left) = self.0.as_chunks::<8>(); @@ -276,7 +279,9 @@ const impl AsRef for VecN { } } -/// Widens an 8-component chunk to double-precision lanes; exact for every `f32`. +/// Widens eight components to double-precision lanes. +/// +/// Every finite `f32` component is represented exactly. #[expect( clippy::inline_always, reason = "SIMD values cross non-inlined call boundaries through memory; inlining into the \ @@ -287,7 +292,9 @@ fn widen(chunk: [f32; 8]) -> f64x8 { f32x8::from_array(chunk).cast() } -/// Rounds a double-precision accumulator to the working precision. +/// Rounds a double-precision accumulator to `f32`. +/// +/// Values beyond the finite rounding range become signed infinity, and NaN remains NaN. #[expect( clippy::cast_possible_truncation, reason = "the narrowing is the operation: the single rounding from the f64 accumulator to the \ @@ -306,8 +313,8 @@ const fn narrow_accumulated(value: f64) -> f32 { /// `align_of::()`. The transparent layout means any array that happens to be aligned can be /// wrapped in place. /// -/// The payoff is [`lanes`](Self::lanes): every 8-lane load comes from an aligned address, so -/// iteration over the vector never splits a cache line. +/// [`Self::lanes`] borrows complete SIMD groups from the aligned base and returns any trailing +/// components separately. // No `FromBytes`/`FromZeros`: a byte-level constructor would let // `zerocopy::transmute_ref!` produce references to unaligned arrays, // bypassing the alignment invariant. @@ -327,8 +334,10 @@ impl AlignedVecN { #[inline] #[must_use] pub const unsafe fn from_ref_unchecked(value: &[f32; N]) -> &Self { - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, and the alignment invariant is - // the caller's contract. + // SAFETY: repr(transparent) preserves the initialized array's layout and validity. The + // caller supplies the additional SIMD address alignment, and the cast retains the source + // provenance and shared lifetime. Therefore the reference meets both Rust's validity rules + // and Self's alignment invariant. unsafe { &*ptr::from_ref(value).cast::() } } @@ -341,15 +350,17 @@ impl AlignedVecN { #[inline] #[must_use] pub const unsafe fn from_mut_unchecked(value: &mut [f32; N]) -> &mut Self { - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, and the alignment invariant is - // the caller's contract. + // SAFETY: repr(transparent) preserves the initialized array's layout and validity. The + // caller supplies the additional SIMD address alignment, and the cast retains the source + // provenance and exclusive lifetime. Therefore the reference meets both Rust's validity + // rules and Self's alignment invariant. unsafe { &mut *ptr::from_mut(value).cast::() } } /// Wraps a borrowed array, checking its alignment. /// - /// Returns [`None`] when `value` is not aligned to `align_of::()` bytes. Stack arrays - /// and plain boxes meet that alignment only by luck. Obtain aligned storage from [`BoxedVecN`]. + /// Returns [`None`] when `value` is not aligned to `align_of::()` bytes. Obtain storage + /// with that alignment from [`BoxedVecN`]. #[must_use] pub fn from_ref(value: &[f32; N]) -> Option<&Self> { if !value.as_ptr().is_aligned_to(align_of::()) { @@ -362,8 +373,8 @@ impl AlignedVecN { /// Wraps a mutable array, checking its alignment. /// - /// Returns [`None`] when `value` is not aligned to `align_of::()` bytes. Stack arrays - /// and plain boxes meet that alignment only by luck. Obtain aligned storage from [`BoxedVecN`]. + /// Returns [`None`] when `value` is not aligned to `align_of::()` bytes. Obtain storage + /// with that alignment from [`BoxedVecN`]. #[must_use] pub fn from_mut(value: &mut [f32; N]) -> Option<&mut Self> { if !value.as_ptr().is_aligned_to(align_of::()) { @@ -376,14 +387,14 @@ impl AlignedVecN { /// Wraps a borrowed slice in place as consecutive aligned vectors. /// - /// Vector `i` of the returned slice occupies components `N · i` through `N · i + N - 1`, so a - /// row-major `f32[T, N]` matrix reads as its `T` rows with every SIMD kernel available on each. + /// A row-major `f32[T, N]` matrix becomes a view of its `T` rows, with every SIMD kernel + /// available on each. Vector `i` occupies components `N · i` through `N · i + N - 1`. /// /// Returns [`None`] unless every vector satisfies the alignment invariant: `components` starts /// at an address aligned to `align_of::()` bytes, one vector's `N · 4` bytes are a - /// multiple of that alignment (`N % 8 == 0` at the widest, 32-byte alignment) so the base - /// alignment carries to every row, and the length is a whole number of vectors. `N` must be - /// nonzero. + /// multiple of that alignment, and the length is a whole number of vectors. These conditions + /// carry the base alignment to every row. Instantiating this method with `N == 0` fails its + /// compile-time assertion. #[must_use] pub fn from_slice(components: &[f32]) -> Option<&[Self]> { const { assert!(N != 0) }; @@ -403,9 +414,10 @@ impl AlignedVecN { let chunks_ptr = &raw const *chunks; let ptr = chunks_ptr as *const [Self]; - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, so the chunk slice - // reinterprets element-wise, and the checks above place every element a multiple of - // `align_of::()` bytes past an aligned base, which is the alignment invariant. + // SAFETY: repr(transparent) preserves each chunk's layout and validity. The checks + // establish an aligned base and an alignment-preserving row stride. The cast retains the + // initialized slice's element count, provenance and shared lifetime. Therefore every + // returned row satisfies Self's alignment invariant throughout the borrow. Some(unsafe { &*ptr }) } @@ -433,10 +445,10 @@ impl AlignedVecN { let chunks_ptr = &raw mut *chunks; let ptr = chunks_ptr as *mut [Self]; - // SAFETY: `Self` is a transparent wrapper around `[f32; N]`, so the chunk slice - // reinterprets element-wise, the checks above place every element a multiple of - // `align_of::()` bytes past an aligned base, which is the alignment invariant, and - // the wrapper inherits the exclusive borrow unchanged. + // SAFETY: repr(transparent) preserves each chunk's layout and validity. The checks + // establish an aligned base and an alignment-preserving row stride. The cast retains the + // initialized slice's element count, provenance and exclusive lifetime. Therefore every + // returned row satisfies Self's alignment invariant throughout the borrow. Some(unsafe { &mut *ptr }) } @@ -464,9 +476,8 @@ impl AlignedVecN { /// /// The first slice reinterprets the storage in place as full [`f32x8`] groups, in order: group /// `i` holds components `8 · i` through `8 · i + 7`. The second slice holds the trailing `N % - /// 8` components that do not fill a group; it is empty whenever the dimension is a multiple of - /// 8, which embedding dimensions in practice are. The type's alignment invariant guarantees no - /// misaligned prefix exists, so no components precede the groups. + /// 8` components that do not fill a group. The type's alignment invariant excludes a misaligned + /// prefix. #[inline] #[must_use] pub fn lanes(&self) -> (&[f32x8], &[f32]) { @@ -482,8 +493,8 @@ impl AlignedVecN { /// Returns the components as mutable aligned 8-lane groups plus a mutable scalar remainder. /// - /// The split is the same as [`lanes`](Self::lanes); writes through either slice update the - /// vector in place, so SIMD kernels can transform embeddings without a staging copy. + /// The split is the same as [`lanes`](Self::lanes). Writes through either slice update the + /// vector in place. #[inline] #[must_use] pub fn lanes_mut(&mut self) -> (&mut [f32x8], &mut [f32]) { @@ -497,9 +508,6 @@ impl AlignedVecN { (lanes, suffix) } - // The arithmetic delegates to the `VecN` kernels over the same - // pointer, so every load still reads an aligned address. - /// Returns the dot product of the two vectors, accumulated in double precision. /// /// See [`VecN::dot`]. @@ -580,15 +588,24 @@ impl ToOwned for AlignedVecN { /// An owned `N`-dimensional vector in a heap allocation aligned for [`f32x8`]. /// -/// The buffer is allocated with `align_of::()` alignment regardless of `N`, so dereferencing -/// always yields an [`AlignedVecN`]. This is the intended long-term storage for embeddings: -/// allocate once, then hand out aligned references to SIMD kernels for the lifetime of the box. +/// The buffer provides `align_of::()` alignment regardless of `N`, including zero. It can be +/// borrowed as an [`AlignedVecN`] throughout the box's lifetime. Cloning creates a separate buffer, +/// while cloning into an existing box reuses its allocation. The allocator is retained until that +/// buffer is released. /// -/// # Examples +/// Allocation failure in the infallible constructors and trait conversions is handled by +/// [`handle_alloc_error`](alloc::alloc::handle_alloc_error). They panic if the required layout +/// cannot be represented. +/// +/// # Example +/// +/// This in-crate example is ignored because the module is private and uses nightly portable SIMD. /// /// ```ignore /// # #![feature(portable_simd)] -/// # use std::simd::num::SimdFloat as _; +/// use std::simd::num::SimdFloat as _; +/// +/// use crate::math::{BoxedVecN, VecN}; /// /// let embedding = BoxedVecN::new(&VecN::new([0.5_f32; 32])); /// @@ -605,6 +622,12 @@ pub struct BoxedVecN { impl BoxedVecN { /// Copies the vector into a new aligned allocation in the global allocator. + /// + /// Allocation failure is handled by [`handle_alloc_error`](alloc::alloc::handle_alloc_error). + /// + /// # Panics + /// + /// Panics if the required layout cannot be represented. See [`Self::new_in`]. #[inline] #[must_use] pub(crate) fn new(value: &VecN) -> Self { @@ -614,7 +637,12 @@ impl BoxedVecN { /// Creates the zero vector in a new aligned allocation in the global allocator. /// /// Every component is `0.0` and the buffer is valid for in-place filling through - /// [`as_array_mut`](AlignedVecN::as_array_mut). + /// [`as_array_mut`](AlignedVecN::as_array_mut). Allocation failure is handled by + /// [`handle_alloc_error`](alloc::alloc::handle_alloc_error). + /// + /// # Panics + /// + /// Panics if the required layout cannot be represented. See [`Self::zero_in`]. #[inline] #[must_use] pub(crate) fn zero() -> Self { @@ -623,9 +651,14 @@ impl BoxedVecN { } impl BoxedVecN { - /// The allocation layout: `N` components, padded to the alignment of [`f32x8`]. + /// Computes the allocation layout for `N` components with SIMD alignment. /// - /// Allocation and deallocation must agree on this. + /// Raising alignment preserves the byte size, without adding trailing padding. Allocation and + /// deallocation must use this same layout. + /// + /// # Panics + /// + /// Panics if the component array or its required alignment exceeds the layout size limit. #[inline] fn layout() -> Layout { Layout::array::(N) @@ -635,8 +668,11 @@ impl BoxedVecN { /// Creates the zero vector in a new aligned allocation in `alloc`. /// - /// This aborts the process through [`handle_alloc_error`](std::alloc::handle_alloc_error) when - /// the allocator cannot provide the buffer. + /// Allocation failure is handled by [`handle_alloc_error`](alloc::alloc::handle_alloc_error). + /// + /// # Panics + /// + /// Panics if the required layout cannot be represented. #[inline] #[must_use] pub(crate) fn zero_in(alloc: A) -> Self { @@ -654,8 +690,12 @@ impl BoxedVecN { /// Copies the vector into a new aligned allocation in `alloc`. /// - /// This aborts the process through [`handle_alloc_error`](std::alloc::handle_alloc_error) when - /// the allocator cannot provide the buffer. + /// Allocation failure is handled by [`handle_alloc_error`](alloc::alloc::handle_alloc_error). + /// Use [`Self::try_new_in`] to receive an allocation error. + /// + /// # Panics + /// + /// Panics if the required layout cannot be represented. #[inline] #[must_use] pub(crate) fn new_in(value: &VecN, alloc: A) -> Self { @@ -666,20 +706,27 @@ impl BoxedVecN { this } - /// Copies the vector into a new aligned allocation in `alloc`, surfacing allocation failure. + /// Tries to copy the vector into a new aligned allocation in `alloc`. /// /// # Errors /// - /// Returns [`AllocError`] when the allocator cannot provide the buffer. The failing call leaks - /// no memory. + /// Returns [`AllocError`] when the allocator cannot provide the buffer. + /// + /// # Panics + /// + /// Panics if the required layout cannot be represented. #[inline] pub(crate) fn try_new_in(value: &VecN, alloc: A) -> Result { let layout = Self::layout(); let allocation = alloc.allocate(layout)?; let ptr = allocation.cast::(); - // SAFETY: the allocation above covers at least `N` components and cannot overlap the - // borrowed source. + // SAFETY: copy_nonoverlapping is an untyped copy that preserves initialization state. It + // requires aligned source and destination pointers valid for N-component read and write + // ranges, with no overlap for a nonzero copy. The source array reference supplies N + // initialized f32 values. allocate supplies a separate buffer with the requested size and + // alignment, including a non-null aligned pointer when N is zero. Therefore the copy + // initializes the destination without aliasing the source. unsafe { ptr::copy_nonoverlapping(value.as_array().as_ptr(), ptr.as_ptr(), N); } @@ -692,17 +739,22 @@ const impl Deref for BoxedVecN { type Target = AlignedVecN; fn deref(&self) -> &Self::Target { - // SAFETY: `ptr` owns an initialized buffer of `N` components for as long as `self` lives, - // allocated with the alignment of `f32x8` by `layout`. + // SAFETY: the array reference requires initialized aligned storage valid for the borrow. + // Constructors initialize all N components by zeroing or copying, retain the allocator, and + // request f32x8 alignment even for N = 0. No shared method deallocates or mutates the + // buffer. Therefore the array reference and its AlignedVecN view remain valid for the + // shared borrow of self. unsafe { AlignedVecN::from_ref_unchecked(&*self.ptr.as_ptr().cast::<[f32; N]>()) } } } const impl DerefMut for BoxedVecN { fn deref_mut(&mut self) -> &mut Self::Target { - // SAFETY: `ptr` owns an initialized buffer of `N` components for as long as `self` lives, - // allocated with the alignment of `f32x8` by `layout`; the exclusive borrow of `self` - // guards the exclusive reference. + // SAFETY: the array reference requires initialized, aligned and exclusively accessible + // storage. Constructors initialize all N components and retain a buffer with f32x8 + // alignment, including when N = 0. The exclusive borrow of its owning box excludes other + // buffer access for the returned lifetime. Therefore both the mutable array reference and + // its AlignedVecN view are valid. unsafe { AlignedVecN::from_mut_unchecked(&mut *self.ptr.as_ptr().cast::<[f32; N]>()) } } } @@ -726,11 +778,13 @@ impl Clone for BoxedVecN { } fn clone_from(&mut self, source: &Self) { - // Both buffers share the same layout for a given `N`, so this - // copies into the existing allocation instead of reallocating. - // - // SAFETY: both pointers own initialized buffers of `N` components, and two live boxes - // cannot alias. + // SAFETY: copy_nonoverlapping is an untyped copy that preserves initialization state. It + // requires aligned pointers valid for N-component read and write ranges, with no overlap + // for a nonzero copy. Both boxes retain separately owned buffers of N aligned f32 values, + // and the source array reference supplies initialized components. The mutable destination + // borrow excludes aliasing with source. For N = 0 both pointers remain non-null and + // aligned. Therefore the copy reuses the destination allocation while preserving its + // initialized components. unsafe { ptr::copy_nonoverlapping(source.as_array().as_ptr(), self.ptr.as_ptr(), N); } @@ -789,18 +843,23 @@ const impl PartialEq for BoxedVecN { impl Drop for BoxedVecN { #[inline] fn drop(&mut self) { - // SAFETY: every constructor allocates `ptr` from `alloc` with `Self::layout()`, the - // layout passed here, and nothing has deallocated it since. + // SAFETY: deallocate requires a currently allocated pointer and a matching allocator + // layout. Constructors retain the allocating allocator and its buffer, which no other + // operation deallocates or transfers. Self::layout is unchanged for N. Therefore Drop + // releases the buffer exactly once with its original allocator and layout. unsafe { self.alloc.deallocate(self.ptr.cast::(), Self::layout()); } } } -// SAFETY: the buffer is exclusively owned and its `f32` components are `Send` and `Sync`; the -// allocator's own thread-safety carries the bound. +// SAFETY: Send permits transferring ownership between threads. The box exclusively owns its f32 +// buffer, whose components are Send, and A: Send permits moving the retained allocator with it. +// Therefore the buffer and its eventual deallocation can transfer with the box. unsafe impl Send for BoxedVecN {} -// SAFETY: shared access only exposes `&[f32; N]`, which is `Sync`; the allocator's own -// thread-safety carries the bound. +// SAFETY: Sync requires shared access to avoid unsynchronized mutation. Shared box methods expose +// immutable f32 components and access the retained allocator only through shared methods, with A: +// Sync. Mutation and deallocation require exclusive ownership. Therefore sharing the box introduces +// no mutable buffer aliases. unsafe impl Sync for BoxedVecN {} diff --git a/libs/@local/graph/atlas/src/math/vecn/tests.rs b/libs/@local/graph/atlas/src/math/vecn/tests.rs index 28e4b90eea9..b372afc9702 100644 --- a/libs/@local/graph/atlas/src/math/vecn/tests.rs +++ b/libs/@local/graph/atlas/src/math/vecn/tests.rs @@ -15,7 +15,7 @@ use proptest::{prop_assert, prop_assert_eq, prop_assume, property_test, strategy use crate::math::{AlignedVecN, BoxedDVecN, BoxedVecN, DVecN, VecN}; -/// Deterministic, sign-varying components crossing multiple 8-lane chunks. +/// Generates sign-varying components from the index and an offset. fn scattered(offset: f32) -> [f32; N] { core::array::from_fn(|index| { let value = f32::from(u8::try_from(index % 200).expect("bounded by modulus")); @@ -24,7 +24,7 @@ fn scattered(offset: f32) -> [f32; N] { }) } -/// Plain-f64 reference for every product-sum kernel under test. +/// Sums separately rounded `f64` products over the slices' common prefix. fn reference_dot(left: &[f32], right: &[f64]) -> f64 { left.iter() .zip(right) @@ -34,8 +34,8 @@ fn reference_dot(left: &[f32], right: &[f64]) -> f64 { #[test] fn is_finite_rejects_any_non_finite_component() { - // N = 11 crosses one full 8-lane chunk plus a remainder. Index 3 - // poisons the lane path and index 9 poisons the remainder path. + // N = 11 includes one full eight-lane chunk and a remainder. Indices 3 and 9 exercise the two + // paths separately. let finite: [f32; 11] = scattered(0.25); assert!(VecN::new(finite).is_finite()); @@ -84,8 +84,15 @@ fn aligned_dot_wide_matches_f64_reference() { ); } +// The dimensions cover an empty sum, a scalar remainder alone, an exact chunk and multiple chunks +// with and without a remainder. #[test] fn dot_matches_f64_reference_across_chunk_sizes() { + /// Compares the narrowed dot product with a scalar `f64` accumulation. + /// + /// # Panics + /// + /// Panics if the absolute error exceeds 10⁻⁶ · (|expected| + 1). fn check() { let left: [f32; N] = scattered(0.5); let right: [f32; N] = scattered(-1.25); @@ -149,7 +156,7 @@ fn cosine_distance_matches_known_geometry() { x_axis[0] = 2.0; y_axis[1] = 0.5; - // Orthogonal: one. Parallel (any positive scaling): zero. Opposite: two. + // the axis-aligned fixture has exactly representable dot products and norms assert_eq!( VecN::new(x_axis).cosine_distance(VecN::from_ref(&y_axis)), 1.0, @@ -174,6 +181,7 @@ fn cosine_distance_of_zero_vectors_follows_the_contract() { assert_eq!(unit.cosine_distance(&zero), 1.0); } +/// Over 100 components the cosine distance agrees with the clamped `f64` reference within `1e-6`. #[test] fn cosine_distance_matches_f64_reference() { let left: [f32; 100] = scattered(0.75); @@ -207,6 +215,7 @@ fn dot_wide_matches_f64_reference() { ); } +/// The aligned box's `dot`, `norm_squared` and `cosine_distance` return exactly the `VecN` results. #[test] fn aligned_kernels_agree_with_vecn() { let left: [f32; 24] = scattered(1.5); @@ -225,17 +234,15 @@ fn aligned_kernels_agree_with_vecn() { ); } -/// Components bounded to the well-conditioned `-1e3..1e3` range, in the fixed dimension 19. +/// Generates 19 bounded components spanning two SIMD chunks and a remainder. /// -/// Two full 8-lane chunks plus a remainder of three, so every fold exercises both the batched body -/// and the remainder. +/// The magnitude bound avoids overflow, while mixed signs still allow cancellation. fn components_strategy() -> impl Strategy { proptest::array::uniform19(-1e3_f32..1e3) } -/// The dot product commutes bit for bit. -/// -/// Both orders accumulate the same products in the same order. +// Swapping operands preserves every rounded product and the accumulation grouping. These bounded +// inputs permit an exact numeric comparison. #[property_test] fn dot_is_commutative( #[strategy = components_strategy()] left: [f32; 19], @@ -247,15 +254,11 @@ fn dot_is_commutative( ); } -/// The squared norm is non-negative: it accumulates squares. #[property_test] fn norm_squared_is_non_negative(#[strategy = components_strategy()] components: [f32; 19]) { prop_assert!(VecN::new(components).norm_squared() >= 0.0); } -/// Cosine distance lies in `[0, 2]`. -/// -/// The distance from a non-zero vector to itself is zero up to rounding. #[property_test] fn cosine_distance_stays_in_range_and_vanishes_on_self( #[strategy = components_strategy()] left: [f32; 19], @@ -271,9 +274,8 @@ fn cosine_distance_stays_in_range_and_vanishes_on_self( prop_assert!(left.cosine_distance(&left) < 1e-6); } -/// The fused SIMD dot product matches a plain-f64 reference loop. -/// -/// The tolerance is relative, plus a small absolute floor for cancelled sums. +// Different summation groupings can disagree after cancellation. The comparison permits 10⁻⁶ +// relative error plus an absolute allowance of 10⁻³. #[property_test] fn dot_matches_a_plain_f64_reference( #[strategy = components_strategy()] left: [f32; 19], @@ -290,10 +292,8 @@ fn dot_matches_a_plain_f64_reference( ); } -/// The checking slice wrapper refuses an offset view of aligned storage. -/// -/// Eight `f32` components fill exactly one `f32x8` group, so the width condition holds and only -/// the address decides. One component in, the view sits four bytes past the boundary. +// shifting an eight-component slice by one component preserves its width and offsets its address by +// four bytes, isolating the SIMD alignment check. #[test] fn aligned_from_slice_mut_checks_the_address() { let mut boxed = BoxedVecN::new(&VecN::new([7.0_f32; 16])); @@ -303,18 +303,13 @@ fn aligned_from_slice_mut_checks_the_address() { assert!(AlignedVecN::<8>::from_slice_mut(&mut array[1..9]).is_none()); } -/// Hashes one value with the std default hasher. +/// Hashes one value with [`DefaultHasher`]. fn hash_of(value: impl Hash) -> u64 { let mut hasher = DefaultHasher::new(); value.hash(&mut hasher); hasher.finish() } -/// The tests the `miri` nextest profile selects. -/// -/// Each test here wraps, boxes or lane-splits component storage in place and checks the alignment -/// invariant those views carry. The profile selects by module path, so moving a test in or out of -/// this module is the whole edit. mod miri { use core::{ iter, @@ -346,7 +341,6 @@ mod miri { f32::from(u8::try_from(index).expect("test dimensions are small")) }); - // Allocate sixteen boxes so one aligned pointer cannot be luck. let boxes: Vec> = iter::repeat_with(|| BoxedVecN::new(VecN::from_ref(&source))) .take(16) @@ -424,6 +418,7 @@ mod miri { assert_eq!(maximum, 15.0); } + /// `lanes` over eleven components yields one full group and a three-component remainder. #[test] fn lanes_split_off_partial_group_as_remainder() { let source: [f32; 11] = core::array::from_fn(|index| { @@ -444,7 +439,7 @@ mod miri { fn aligned_vecn_rejects_misaligned_storage() { let boxed = BoxedVecN::new(&VecN::new([0.0_f32; 16])); - // The box's own storage meets the f32x8 alignment, so wrapping it succeeds. + // the box provides an f32x8-aligned base assert!(AlignedVecN::<16>::from_ref(boxed.as_array()).is_some()); // One component past an aligned base breaks the f32x8 alignment. @@ -474,6 +469,7 @@ mod miri { assert_eq!(clone.as_array()[7], 8.0); } + /// `clone_from` copies the contents into the existing buffer without reallocating. #[test] fn boxed_vecn_clone_from_reuses_the_allocation() { let source = BoxedVecN::from([9.0_f32; 8]); @@ -526,7 +522,7 @@ mod miri { fn try_as_aligned_agrees_between_shared_and_mutable() { let mut boxed = BoxedVecN::from([3.0_f32; 8]); - // Boxed storage meets the f32x8 alignment, so both reinterpretations succeed. + // boxed storage provides f32x8 alignment for the shared and mutable views assert!(VecN::from_ref(boxed.as_array()).try_as_aligned().is_some()); let vecn = VecN::from_mut(boxed.as_array_mut()); @@ -564,14 +560,10 @@ mod miri { let matrix = BoxedVecN::<32>::zero(); let components = matrix.as_array().as_slice(); - // A zero dimension fails compilation outright; only the runtime - // conditions remain to certify. - - // A partial trailing row cannot be a vector. + // N = 12 can fail the row-stride alignment check before the partial-row check. assert!(AlignedVecN::<12>::from_slice(components).is_none()); - // One component past an aligned base breaks the alignment: a single - // `f32` is narrower than `f32x8`'s alignment on every supported target. + // the offset is four bytes, below the SIMD alignment on the tested target assert!(AlignedVecN::<8>::from_slice(&components[1..25]).is_none()); // An empty slice at an aligned base yields zero rows. @@ -585,7 +577,7 @@ mod miri { let low = BoxedVecN::new(&VecN::new([0.5_f32, 1.5])); let high = BoxedVecN::new(&VecN::new([1.0_f32, 1.5])); - // A fixed-key DefaultHasher makes distinctness deterministic for fixed inputs. + // these two fixtures produce different hashes with the tested DefaultHasher assert_ne!(hash_of(&low), hash_of(&high)); assert_eq!(format!("{low:?}"), "AlignedVecN([0.5, 1.5])"); } diff --git a/libs/@local/graph/atlas/src/morton/mod.rs b/libs/@local/graph/atlas/src/morton/mod.rs index 4a0b0527641..ff58c65489b 100644 --- a/libs/@local/graph/atlas/src/morton/mod.rs +++ b/libs/@local/graph/atlas/src/morton/mod.rs @@ -1,19 +1,17 @@ //! Z-order keys: two 32-bit axes interleaved into one sortable `u64`. //! -//! The module is crate-internal, with one deliberate seam: the `bench` facade re-exports -//! [`Depth`], [`MortonKey`], and [`MortonCell`] so the benchmark targets speak the same typed -//! vocabulary as production instead of raw integers. The items are therefore `pub` while the -//! module is not, and they reach a consumer only through that feature-gated door. Examples carry -//! `ignore` and spell each call as an in-crate caller writes it. +//! The crate-internal module exports [`Depth`], [`MortonKey`], and [`MortonCell`] through the +//! `bench` facade when that feature is enabled. Examples carry `ignore` and use in-crate paths. //! //! [`MortonKey::new`] interleaves the bits of an `(x, y)` pair, `x` into the even bits and `y` into -//! the odd bits, so that comparing keys compares positions along the Z-order curve. Every -//! axis-aligned power-of-two cell of the grid is one contiguous key range, so a sorted key array -//! answers cell queries with two binary searches. +//! the odd bits. Comparing keys compares positions along the Z-order curve. Each depth-d cell fixes +//! the leading d bits of each axis, hence the leading 2d key bits. Its remaining bits span one +//! contiguous key range. A sorted key array answers these prefix-aligned cell queries with two +//! binary searches. //! //! [`Depth`] counts subdivisions. Depth 0 is the whole domain, each step quarters a cell, and depth //! 32 pins both axes to a single key. A tile address `(z, x, y)` names the cell -//! [`MortonCell::new(z, x, y)`](MortonCell::new); the cell containing an existing key is +//! [`MortonCell::new(z, x, y)`](MortonCell::new). The cell containing an existing key is //! [`MortonKey::cell`]. Cells subdivide in key order via [`MortonCell::children`]. #![cfg_attr( @@ -44,14 +42,12 @@ impl Depth { /// Wraps a subdivision count. /// - /// Returns [`None`] above [`Depth::MAX`]. + /// Depth d cells are the squares of a 2ᵈ × 2ᵈ grid over the axis domain. [`Depth::MIN`] is the + /// whole domain. [`Depth::MAX`] fixes all 32 bits of both axes and identifies one key. /// /// # Examples /// - /// ```ignore - /// assert!(Depth::new(16).is_some()); - /// assert_eq!(Depth::new(33), None); - /// ``` + /// Returns [`None`] above [`Depth::MAX`], the key width of one axis. #[inline] #[must_use] pub const fn new(depth: u8) -> Option { @@ -62,7 +58,9 @@ impl Depth { Some(Self(depth)) } - /// Returns the subdivision count. + /// Returns the depth whose grid a tile zoom addresses. + /// + /// Zoom z and depth z name the same 2ᶻ × 2ᶻ grid, and both types end at [`Depth::MAX`]. #[inline] #[must_use] pub const fn get(self) -> u8 { @@ -71,15 +69,21 @@ impl Depth { /// Adds `steps` subdivisions, saturating at [`Depth::MAX`]. /// - /// The domain is capped, so the sum clamps instead of overflowing: the same contract as - /// [`u8::saturating_add`], with the ceiling at the key width rather than the type width. + /// Like [`u8::saturating_add`], the sum clamps instead of overflowing. The ceiling is the key + /// width rather than the type width. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore - /// let depth = Depth::new(30).unwrap(); - /// assert_eq!(depth.saturating_add(1).get(), 31); - /// assert_eq!(depth.saturating_add(9), Depth::MAX); + /// use crate::math::{Log2}; + /// use crate::morton::{Depth}; + /// + /// let depth = Depth::new(30); + /// let steps = Log2::new(9).expect("9 should fit the exponent domain"); + /// assert_eq!(depth.saturating_add(Log2::ONE).get(), 31); + /// assert_eq!(depth.saturating_add(steps), Depth::MAX); /// ``` #[inline] #[must_use] @@ -101,13 +105,17 @@ impl Depth { /// A Z-order key interleaving two 32-bit axes into one `u64`. /// -/// `x` occupies the even bits and `y` the odd bits, starting at bit 0, so key order is Z-order -/// curve order and every [`MortonCell`] is one contiguous key range. Every bit pattern is a valid -/// key. +/// `x` occupies the even bits and `y` the odd bits, starting at bit 0. Key order follows the +/// Z-order curve, and every [`MortonCell`] is one contiguous key range. Every bit pattern is a +/// valid key. +/// +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::morton::{MortonKey}; +/// /// assert_eq!(MortonKey::new(1, 0).to_bits(), 0b01); /// assert_eq!(MortonKey::new(0, 1).to_bits(), 0b10); /// assert_eq!(MortonKey::new(3, 5).coordinates(), [3, 5]); @@ -147,14 +155,18 @@ impl MortonKey { /// Returns the cell index at `depth`. /// - /// The leading `2 · depth` key bits, a value below `4^depth` that is dense over the depth's - /// grid. + /// The leading 2d key bits, where d is `depth`. These values densely index the cells from zero + /// through 4ᵈ − 1. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::morton::{Depth, MortonKey}; + /// /// let key = MortonKey::new(0b10 << 30, 0b11 << 30); - /// assert_eq!(key.prefix(Depth::new(2).unwrap()), 0b1110); + /// assert_eq!(key.prefix(Depth::new(2)), 0b1110); /// ``` #[inline] #[must_use] @@ -167,12 +179,16 @@ impl MortonKey { /// Returns the deepest depth at which this key and `other` share a cell. /// - /// A depth-`d` cell is the leading `2 · d` key bits, so the shared depth counts the agreed + /// A depth-`d` cell is the leading `2 · d` key bits. The shared depth counts the agreed /// leading bit pairs. Equal keys share every grid and return [`Depth::MAX`]. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore + /// use crate::morton::{Depth, MortonKey}; + /// /// let key = MortonKey::new(0, 0); /// assert_eq!(key.shared_depth(key), Depth::MAX); /// assert_eq!(key.shared_depth(MortonKey::new(0, 1 << 31)).get(), 0); @@ -217,19 +233,24 @@ pub struct MortonCell { /// /// The bits below the prefix are zero. min: u64, + /// The grid the cell belongs to. depth: Depth, } impl MortonCell { - /// Wraps the cell at `(x, y)` of the depth's grid. + /// Selects the cell at `(x, y)` of the depth's grid. /// - /// The grid spans `2^depth` cells per axis; returns [`None`] when either coordinate lies + /// At depth d the grid spans 2ᵈ cells per axis. Returns [`None`] when either coordinate lies /// outside it. /// - /// # Examples + /// # Example + /// + /// This in-crate example is ignored because the module is private. /// /// ```ignore - /// let depth = Depth::new(3).unwrap(); + /// use crate::morton::{Depth, MortonCell}; + /// + /// let depth = Depth::new(3); /// assert!(MortonCell::new(depth, 7, 0).is_some()); /// assert_eq!(MortonCell::new(depth, 8, 0), None); /// ``` @@ -281,7 +302,7 @@ impl MortonCell { /// Returns the four child cells in key order. /// - /// Child `i` holds the keys whose next axis bits are `x = i & 1` and `y = i >> 1`; the + /// Child `i` holds the keys whose next axis bits are `x = i & 1` and `y = i >> 1`. The /// children's ranges partition the parent's in that order. Returns [`None`] at [`Depth::MAX`]. #[must_use] pub const fn children(self) -> Option<[Self; 4]> { diff --git a/libs/@local/graph/atlas/src/morton/tests.rs b/libs/@local/graph/atlas/src/morton/tests.rs index eaf80c271ee..3756a26b1eb 100644 --- a/libs/@local/graph/atlas/src/morton/tests.rs +++ b/libs/@local/graph/atlas/src/morton/tests.rs @@ -2,6 +2,7 @@ use proptest::{prop_assert, prop_assert_eq, prop_assert_ne, property_test}; use super::{Depth, MortonCell, MortonKey}; +/// Borrowed and copied decoding admit exactly the constructor's zoom domain. #[test] fn curve_start_matches_the_hand_table() { // The Z-order curve over the 4 x 4 grid, keys 0..16 by hand: @@ -31,6 +32,8 @@ fn curve_start_matches_the_hand_table() { } } +/// Saturated axes interleave to the all-ones key, and a single saturated axis to its alternating +/// bit mask. #[test] fn extremes_interleave_exactly() { assert_eq!(MortonKey::new(0, 0).to_bits(), 0); @@ -39,6 +42,7 @@ fn extremes_interleave_exactly() { assert_eq!(MortonKey::new(0, u32::MAX).to_bits(), 0xAAAA_AAAA_AAAA_AAAA); } +/// `Depth::try_new` admits exactly `0..=32`, mapping the ends to `MIN` and `MAX`. #[test] fn depth_admits_the_documented_domain() { assert_eq!(Depth::new(0), Some(Depth::MIN)); @@ -71,6 +75,8 @@ fn full_depth_cell_is_one_key() { ); } +/// `MortonCell::new` admits coordinates below `2^depth` on each axis and refuses the first +/// coordinate at or beyond it. #[test] fn cell_addresses_reject_coordinates_outside_the_grid() { let depth = Depth::new(3).expect("3 subdivisions lie below the maximum of 32"); @@ -80,6 +86,8 @@ fn cell_addresses_reject_coordinates_outside_the_grid() { assert_eq!(MortonCell::new(depth, 0, 8), None); } +/// A key's prefix at depth `d` is its leading `2d` interleaved bits: empty at depth zero and the +/// whole key at maximum depth. #[test] fn prefixes_index_the_depth_grid_in_key_order() { // Cell (x = 2, y = 3) of the depth-2 grid: axis bits sit at the @@ -99,7 +107,7 @@ fn shared_depth_counts_the_agreed_bit_pairs() { assert_eq!(key.shared_depth(key), Depth::MAX); // y's top bit differs: the keys part at the first subdivision. assert_eq!(key.shared_depth(MortonKey::new(0, 1 << 31)).get(), 0); - // x's second bit differs: one agreed bit pair, so one shared subdivision. + // x's second bit differs: one agreed bit pair and one shared subdivision. assert_eq!(key.shared_depth(MortonKey::new(1 << 30, 0)).get(), 1); } diff --git a/libs/@local/graph/atlas/src/offload.rs b/libs/@local/graph/atlas/src/offload.rs index b3d3b49029c..7eaeb2629cc 100644 --- a/libs/@local/graph/atlas/src/offload.rs +++ b/libs/@local/graph/atlas/src/offload.rs @@ -1,11 +1,8 @@ -//! CPU-bound work spawned onto rayon and answered on tokio. +//! CPU-bound tasks with asynchronous result collection. //! -//! The async surfaces stay responsive by running their heavy computation - response assembly, -//! schedule and census construction - on rayon workers rather than runtime threads. This module -//! exists so every such hand-off shares one panic posture: [`run`] executes the closure behind -//! [`catch_unwind`](std::panic::catch_unwind) and answers through a oneshot channel, so a panic -//! reaches the caller as [`OffloadError::Panicked`] instead of aborting the process, which is -//! rayon's response to a panic no join point observes. +//! [`run`] submits a closure to Rayon, keeping its computation off the async executor. Await the +//! returned [`OffloadHandle`] for its result, or use [`OffloadHandle::try_join`] to check for +//! completion without waiting. use alloc::borrow::Cow; use core::{any::Any, error::Error, fmt, panic::UnwindSafe}; @@ -16,12 +13,9 @@ use core::{any::Any, error::Error, fmt, panic::UnwindSafe}; /// error. #[derive(Debug)] pub(crate) enum OffloadError { - /// The work panicked, and this holds the payload's text when the payload was one. + /// The computation panicked, with a message for string panic payloads. Panicked(Option>), - /// The worker vanished without answering. - /// - /// The channel closed with no result sent, which happens only when the pool drops the job - /// without running it, as at process teardown. + /// The worker closed the channel without a result, or the handle already returned it. Vanished, } @@ -39,10 +33,8 @@ impl Error for OffloadError {} /// Runs `work` on a rayon worker and returns its value, answering a panic as an error. /// -/// The future resolves when the work completes. Dropping the future first - a cancelled request, -/// an abandoned resolution - drops the computed value on the worker and nothing else happens: the -/// work itself always runs to completion once spawned, and the rejected value's drop stays inside -/// an unwind boundary, so even a panicking destructor cannot abort the pool. +/// Submission starts the job independently of polling the returned handle. The worker enters the +/// current tracing span for both the computation and cleanup of an undeliverable result. /// /// # Errors /// @@ -57,13 +49,10 @@ pub(crate) async fn run( rayon::spawn(move || { let result = std::panic::catch_unwind(work); - // A send failure means the caller's future was dropped and nothing wants the value: the - // rejected result drops right here on the worker. That drop runs inside its own unwind - // boundary, because a panicking destructor would otherwise reach the pool as a panic no - // join point observes, which aborts the process. + // A rejected result can panic during drop after the computation's unwind boundary has + // ended. // - // AssertUnwindSafe: the panic is swallowed after the caller has gone, so nothing - // observes the sender's or the value's state after the unwind. + // AssertUnwindSafe: the closure consumes the sender and result. let _cancelled = std::panic::catch_unwind(core::panic::AssertUnwindSafe(|| { let _rejected: Result<(), _> = sender.send(result); })); @@ -76,10 +65,13 @@ pub(crate) async fn run( } } -/// Extracts a panic payload's text. +/// Extracts a string panic message and discards other payloads. /// -/// A `panic!` with a message carries `&'static str` or `String`. Any other payload type has no -/// text to extract and answers [`None`]. +/// Returns the text of an `&'static str` or [`String`] payload, or [`None`] for any other type. +/// +/// # Panics +/// +/// Panics if a non-string payload's destructor panics. fn panic_message(panic: Box) -> Option> { match panic.downcast_ref::<&'static str>() { Some(&message) => Some(Cow::Borrowed(message)), @@ -93,18 +85,12 @@ fn panic_message(panic: Box) -> Option> { pub(crate) mod tests { use super::{OffloadError, run}; - /// A completed computation answers its value. #[tokio::test] async fn completed_work_answers_its_value() { let value = run(|| 6 * 7).await.expect("the work completes"); assert_eq!(value, 42); } - /// A panicking computation answers an error that holds the payload, and the process survives. - /// - /// The survival is the point: a bare `rayon::spawn` would abort the process on this panic, - /// because the pool has no join point to observe it. The follow-up call witnesses that the - /// pool keeps serving after the caught panic. #[tokio::test] async fn panicking_work_answers_an_error_without_aborting() { let error = run(|| -> u32 { panic!("the fixture panicked on purpose") }) @@ -120,7 +106,6 @@ pub(crate) mod tests { assert_eq!(value, 7); } - /// A formatted panic payload crosses as its rendered text. #[tokio::test] async fn formatted_panic_payload_keeps_its_text() { let error = run(|| -> u32 { panic!("row {} is out of range", 41) }) @@ -133,7 +118,6 @@ pub(crate) mod tests { assert_eq!(payload, "row 41 is out of range"); } - /// A payload that is not text answers the panic without one. #[tokio::test] async fn textless_panic_payload_answers_none() { let error = run(|| -> u32 { std::panic::panic_any(41_u64) }) @@ -146,12 +130,7 @@ pub(crate) mod tests { ); } - /// A cancelled caller whose rejected value panics on drop does not abort the pool. - /// - /// A failed send drops the computed value on the rayon worker, and that drop runs inside an - /// unwind boundary. Dropped bare, the destructor's panic would reach the pool with no join - /// point to observe it, and rayon aborts a process on such a panic. Under nextest, a - /// regression here therefore fails this one test with its own process's SIGABRT. + /// The worker catches a string panic from a rejected value's destructor. #[tokio::test] async fn cancelled_send_with_panicking_destructor_does_not_abort() { /// Signals that its drop ran, then panics inside it. @@ -186,7 +165,6 @@ pub(crate) mod tests { release.send(()).expect("the worker waits on this release"); - // The worker computed the value, failed the send, and ran the panicking destructor. drop_witness .recv_timeout(core::time::Duration::from_secs(10)) .expect("the rejected value's destructor runs on the worker"); diff --git a/libs/@local/graph/atlas/src/postgres/card/associations.rs b/libs/@local/graph/atlas/src/postgres/card/associations.rs index a1019aad71a..33be77b65ae 100644 --- a/libs/@local/graph/atlas/src/postgres/card/associations.rs +++ b/libs/@local/graph/atlas/src/postgres/card/associations.rs @@ -1,5 +1,4 @@ -//! The endpoint-association facts, covering the current source types that constrain each -//! relation. +//! The endpoint-association facts, covering the current source types that constrain each relation. use hash_graph_postgres_store::store::postgres::query::{ Aliased, Binder, BoundStatement, ColumnName, CommonTableExpression, Constant, Correlation, @@ -31,7 +30,7 @@ const REF_KEY: &str = "$ref"; /// Matches the version suffix after a base id, anchored to consume the whole remainder. /// -/// Travels as a bound parameter, so the statement text carries no quoted literal. +/// Travels as a bound parameter. The statement text carries no quoted literal. const VERSION_SUFFIX: &str = "^v/[0-9]+$"; /// Matches a trailing version suffix, for erasing it from a versioned URL. const TRAILING_VERSION_SUFFIX: &str = "v/[0-9]+$"; @@ -697,6 +696,11 @@ fn cardinality(value: Option) -> Option { } /// Applies one association row to its relation's facts. +/// +/// # Errors +/// +/// Returns the store's error when a selected column does not read back at the type the decode +/// asks for. An absent target list is a null column that defaults to the empty list. fn apply_row( row: &Row, columns: &AssociationColumns, @@ -756,6 +760,11 @@ fn apply_row( /// version contributes the newest constraint. Allowed targets resolve per base id to their /// latest current prose, ordered by target id. A target reference whose base id is no longer /// current drops out. +/// +/// # Errors +/// +/// Returns the store's error when it rejects the read, then when a row does not decode. `facts` +/// keeps whatever earlier rows already contributed. pub(super) async fn association_rows( transaction: &Transaction<'_>, axes: TemporalAxes, @@ -791,10 +800,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn statement_text() { let axes = TemporalAxes::now(); diff --git a/libs/@local/graph/atlas/src/postgres/card/examples.rs b/libs/@local/graph/atlas/src/postgres/card/examples.rs index c01f89c77fb..9aeb38a7676 100644 --- a/libs/@local/graph/atlas/src/postgres/card/examples.rs +++ b/libs/@local/graph/atlas/src/postgres/card/examples.rs @@ -27,7 +27,7 @@ use crate::dataset::TemporalAxes; /// The field separator inside a stable hash's input, keeping the hashed tuple unambiguous. /// -/// Travels as a bound parameter, so the statement text carries no quoted literal. +/// Travels as a bound parameter. The statement text carries no quoted literal. const FIELD_SEPARATOR: &str = "|"; /// The direct-type marker of a source whose edition lists no direct types. @@ -39,9 +39,8 @@ const NO_DIRECT_TYPE: &str = ""; /// The columns of the example pipeline's CTEs. /// /// The pipeline's stages annotate one logical row: `links` carries the instance identity, -/// `raw_examples` adds the endpoints and their labels, and the scoring stages add frequencies -/// and ranks. One vocabulary serves every stage, so a stage alias can cite exactly the pipeline's -/// columns. +/// `raw_examples` adds the endpoints and their labels, and the scoring stages add frequencies and +/// ranks. One vocabulary serves every stage. A stage alias can cite exactly the pipeline's columns. #[derive(Debug, Copy, Clone, PartialEq, Eq)] enum Example { /// The relation's 1-based position in the type table. @@ -272,11 +271,18 @@ fn instances(axes: Axes) -> SelectStatement { .build() } -/// The endpoint aliases `raw_examples` resolves one edge's target through. +/// The edge row `raw_examples` resolves an example's left endpoint through, as its target. +/// +/// The edge table is joined once per endpoint and the edition cache once per resolved endpoint. +/// Each of those two tables therefore occurs twice in the statement, and each occurrence needs a +/// name of its own to be referred to. const LEFT_EDGE: Aliased = Aliased::of(Table::EntityEdge, "left_edge"); +/// The edge row `raw_examples` reads an example's right endpoint from. const RIGHT_EDGE: Aliased = Aliased::of(Table::EntityEdge, "right_edge"); +/// The cached edition of the left endpoint, which carries its display labels. const SOURCE_CACHE: Aliased = Aliased::of(Table::EntityEditionCache, "source_cache"); +/// The cached edition of the right endpoint, which carries its display labels. const TARGET_CACHE: Aliased = Aliased::of(Table::EntityEditionCache, "target_cache"); @@ -290,8 +296,8 @@ fn raw_example_outputs( field_separator: Placeholder, no_direct_type: Placeholder, ) -> Vec { - // The base ids of the source's direct types lead its closure array, `direct_types` many, so - // the slice's first element is the representative type, and NULL when the edition lists none. + // The base ids of the source's direct types lead its closure array, `direct_types` many. The + // slice's first element is the representative type, and NULL when the edition lists none. let direct_type = Expression::ArrayElement { expr: Box::new(Expression::ArraySlice { expr: Box::new(SOURCE_CACHE.column(&EntityEditionCache::BaseUrls)), @@ -549,8 +555,8 @@ fn scored_examples() -> SelectStatement { /// Builds the `stratified_examples` table: one row per endpoint pair, ranked per subgroup. fn stratified_examples() -> SelectStatement { - // ln(1 + ): a frequency counts at least the row it annotates, so the sum - // stays integral and its logarithm equals the fractional form's. + // ln(1 + ): a frequency counts at least the row it annotates. The sum stays + // integral and its logarithm equals the fractional form's. let log_frequency = |column: Example| { Expression::from(Function::Ln(Box::new( Expression::from(Constant::U32(1)).add(SCORED_EXAMPLES.column(&column)), @@ -792,6 +798,11 @@ fn pool_bound(count: usize, factor: usize) -> i64 { } /// Applies one example row to its relation's facts. +/// +/// # Errors +/// +/// Returns the store's error when a selected column does not read back at the type the decode +/// asks for. fn apply_row( row: &Row, columns: &ExampleColumns, @@ -840,6 +851,11 @@ fn apply_row( /// row survives per endpoint pair, frequencies count each endpoint's occurrences among the /// relation's instances before that dedup, and pooling bounds transfer per source-direct-type /// subgroup first, then per relation, in a deterministic hash order. +/// +/// # Errors +/// +/// Returns the store's error when it rejects the read, then when a row does not decode. `facts` +/// keeps whatever earlier rows already contributed. pub(super) async fn example_rows( transaction: &Transaction<'_>, axes: TemporalAxes, @@ -879,10 +895,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn statement_text() { let axes = TemporalAxes::now(); diff --git a/libs/@local/graph/atlas/src/postgres/card/mod.rs b/libs/@local/graph/atlas/src/postgres/card/mod.rs index 04ddd899fed..159a4fb8300 100644 --- a/libs/@local/graph/atlas/src/postgres/card/mod.rs +++ b/libs/@local/graph/atlas/src/postgres/card/mod.rs @@ -2,8 +2,8 @@ //! //! [`corpus_facts`] gathers everything the card builder consumes for every type in the dataset's //! type table, inside the frozen transaction and at its temporal axes. One query per fact kind -//! (prose, ancestors, associations, examples) covers the whole table at once, so the expensive -//! scans over the entity tables amortize across all cards instead of repeating per type: +//! (prose, ancestors, associations, examples) covers the whole table at once. The expensive scans +//! over the entity tables amortize across all cards instead of repeating per type: //! //! - each type's prose and its (depth, id)-ordered ancestor chain, which omits the type itself and //! the link root; @@ -52,25 +52,32 @@ use crate::dataset::{ /// Content-affecting controls for card extraction. /// -/// A card is deterministic in the dataset's temporal axes and these parameters, so a generation +/// A card is deterministic in the dataset's temporal axes and these parameters. The generation /// records both. The dataset starts from the defaults. #[derive(Debug, Copy, Clone, PartialEq, Eq, Default)] pub(crate) struct CardParameters { /// The most examples one finished card presents. + /// + /// Defaults to 8. pub example_count: usize = 8, /// Example candidates fetched per source-type subgroup. /// /// A multiple of [`example_count`](Self::example_count). /// - /// The pool bounds what the query transfers, and the diverse selector consumes candidates - /// from each subgroup in a deterministic order, so the pool is the slack it has for - /// rejecting duplicates and conflicts. The selector never reaches rows beyond the pool. + /// The pool bounds what the query transfers, and the diverse selector consumes candidates from each subgroup in a deterministic order. The pool is the slack it has for rejecting duplicates and conflicts. The selector never reaches rows beyond the pool. + /// + /// Defaults to 8. A default card then draws from eight candidates per subgroup for each + /// example it presents. pub subgroup_pool_factor: usize = 8, /// Example candidates fetched per relation across all subgroups. /// /// A multiple of [`example_count`](Self::example_count). + /// + /// Defaults to 32. pub pool_factor: usize = 32, /// Token budgets for structural truncation. + /// + /// Defaults to [`CardsConfig`]'s own budgets. pub budgets: CardsConfig = CardsConfig { .. }, } @@ -191,7 +198,7 @@ impl RelationFacts { /// Gathers card facts for every type in `types` inside `transaction`. /// -/// The `n`-th returned facts belong to `types[n]`, so the result aligns with ontology row order. +/// The `n`-th returned facts belong to `types[n]`. The result aligns with ontology row order. /// /// # Errors /// @@ -222,6 +229,12 @@ pub(crate) async fn corpus_facts( } /// Resolves the 1-based `ordinality` column into an index over `facts`. +/// +/// # Panics +/// +/// This panics when `ordinality` is below one or past the end of `facts`. Every caller reads it +/// from a `WITH ORDINALITY` position over the unnested type table that sized `facts`. Either +/// value means the statement stopped agreeing with that table. fn fact_at(facts: &mut [RelationFacts], ordinality: i64) -> &mut RelationFacts { let index = usize::try_from(ordinality - 1).expect("WITH ORDINALITY yields positions starting at one"); diff --git a/libs/@local/graph/atlas/src/postgres/card/prose.rs b/libs/@local/graph/atlas/src/postgres/card/prose.rs index 461dba8cc38..e61625583e9 100644 --- a/libs/@local/graph/atlas/src/postgres/card/prose.rs +++ b/libs/@local/graph/atlas/src/postgres/card/prose.rs @@ -96,6 +96,11 @@ fn prose_statement(types: &(impl ToSql + Sync)) -> BoundStatement<'_, ProseColum /// The `n`-th returned facts belong to `types[n]`: the statement orders by the unnest /// ordinality, and the row count check makes a violated referential contract loud. /// +/// # Errors +/// +/// Returns the store's error when it rejects the read, then when a row's prose columns do not +/// read back at the type the decode asks for. +/// /// # Panics /// /// This panics when a type in `types` resolves no versioned type row, which the store's foreign @@ -152,8 +157,9 @@ struct AncestorColumns { ancestor_id: usize, } -/// Builds the ancestor statement, walking the all-depth inheritance table to each type's -/// ancestors. +/// Builds the ancestor statement. +/// +/// It walks the all-depth inheritance table to each type's ancestors. /// /// # SQL /// @@ -256,6 +262,11 @@ fn ancestor_statement(types: &(impl ToSql + Sync)) -> BoundStatement<'_, Ancesto /// /// The store's inheritance table holds no self rows, and the statement excludes other versions /// of a type's own base id along with it. The link root contributes no prose. +/// +/// # Errors +/// +/// Returns the store's error when it rejects the read, then when a row does not decode. `facts` +/// keeps the ancestors earlier rows already contributed. pub(super) async fn ancestor_rows( transaction: &Transaction<'_>, types: &[Uuid], @@ -306,10 +317,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered prose statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn prose_statement_text() { let types = vec![Uuid::nil()]; @@ -326,10 +333,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered ancestor statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn ancestor_statement_text() { let types = vec![Uuid::nil()]; diff --git a/libs/@local/graph/atlas/src/postgres/classification.rs b/libs/@local/graph/atlas/src/postgres/classification.rs index 7e1ba1cadea..ee33715a4ff 100644 --- a/libs/@local/graph/atlas/src/postgres/classification.rs +++ b/libs/@local/graph/atlas/src/postgres/classification.rs @@ -1,9 +1,9 @@ //! The node-versus-link classification over requested identities, under the serving-time regime. //! -//! The statement executes on its caller's own connection at axes taken at the call, so a -//! verdict describes the store as it stands rather than as any fit observed it. -//! [`Classification`] states the type law the verdict applies. Result identity keys each -//! answer, and the caller counts the answers against its requests. +//! The statement executes on its caller's own connection at axes taken at the call. A verdict +//! describes the store as it stands rather than as any fit observed it. [`Classification`] states +//! the type law the verdict applies. Result identity keys each answer, and the caller counts the +//! answers against its requests. use hash_graph_postgres_store::store::postgres::query::{ Aliased, Binder, BoundStatement, CommonTableExpression, Expression, FromItem, Placeholder, @@ -41,10 +41,10 @@ pub(crate) struct ClassificationColumns { /// The node-versus-link verdict for one resolved identity. /// -/// The verdict applies the type law that also draws the corpus's node scope. An entity is a -/// link exactly when its type closure reaches the link entity type, whatever edges it holds. -/// Endpoints are entity identities, never resolved against any generation's rows, so the -/// verdict holds for entities no generation has fitted. +/// The verdict applies the type law that also draws the corpus's node scope. An entity is a link +/// exactly when its type closure reaches the link entity type, whatever edges it holds. Endpoints +/// are entity identities, never resolved against any generation's rows. The verdict holds for +/// entities no generation has fitted. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum Classification { /// A non-link entity. @@ -60,11 +60,11 @@ pub(crate) enum Classification { /// Builds the classification lookup over the requested identities. /// -/// The statement decides the node-versus-link split for every requested identity that resolves -/// at the bound axes, and delivers a link's outgoing attachment endpoints in the same row. The -/// endpoint joins are outer, so a link with an absent or incomplete attachment pair still -/// answers, with its missing endpoints SQL NULL. The caller counts the answers against its -/// requests. A missing row is an identity that is draft-only, archived, or absent at the axes. +/// The statement decides the node-versus-link split for every requested identity that resolves at +/// the bound axes, and delivers a link's outgoing attachment endpoints in the same row. The +/// endpoint joins are outer. A link with an absent or incomplete attachment pair still answers, +/// with its missing endpoints SQL NULL. The caller counts the answers against its requests. A +/// missing row is an identity that is draft-only, archived, or absent at the axes. /// /// # SQL /// @@ -163,12 +163,18 @@ pub(crate) fn classification_statement<'params>( } /// Decodes one classification row. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Query`] when a selected column does not read back at the type +/// the decode asks for. An endpoint half that is null is not a failure: the two halves are null +/// together, and the pair collapses to no endpoint. pub(crate) fn decode_classification( row: &Row, columns: &ClassificationColumns, ) -> Result<(ArchivedEntityId, Classification), PostgresDatasetError> { - // Both columns of an endpoint come from one joined edge row, so they are null together and - // `zip` collapses exactly the no-edge case. + // Both columns of an endpoint come from one joined edge row. They are null together and `zip` + // collapses exactly the no-edge case. let endpoint = |web_id: usize, entity_uuid: usize| { let web_id: Option = row.try_get(web_id)?; let entity_uuid: Option = row.try_get(entity_uuid)?; @@ -221,12 +227,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. Reviewing a diff, hold it to the - /// statement's own contract: both attachment edges join outer, so a link with an absent - /// or incomplete attachment pair still answers. #[test] fn statement_text() { let axes = TemporalAxes::now(); diff --git a/libs/@local/graph/atlas/src/postgres/corpus.rs b/libs/@local/graph/atlas/src/postgres/corpus.rs index 0b70de6f792..a52a924b726 100644 --- a/libs/@local/graph/atlas/src/postgres/corpus.rs +++ b/libs/@local/graph/atlas/src/postgres/corpus.rs @@ -597,10 +597,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn statement_text() { let axes = TemporalAxes::now(); diff --git a/libs/@local/graph/atlas/src/postgres/edition_display.rs b/libs/@local/graph/atlas/src/postgres/edition_display.rs index 254a300fc33..080c7783def 100644 --- a/libs/@local/graph/atlas/src/postgres/edition_display.rs +++ b/libs/@local/graph/atlas/src/postgres/edition_display.rs @@ -1,13 +1,12 @@ //! The display lookup over requested editions, under the serving-time regime. //! -//! The statement executes on its caller's own connection and binds no temporal axes. An -//! edition id addresses one immutable row, so the answer is the same at any read. Every -//! requested edition answers exactly once, because every join is outer and the unnested -//! requests survive them. The representative cached type answers as a store uuid rather than a -//! generation ordinal, because an edition written after a fit can carry a type no generation -//! tabulated. Its nearest declared icon rides the same row, resolved through the -//! representative's current closed schema, so a register allocating a row for such a type has -//! the icon in hand at the allocation. +//! The statement executes on its caller's own connection and binds no temporal axes. An edition id +//! addresses one immutable row. The answer is the same at any read. Every requested edition answers +//! exactly once, because every join is outer and the unnested requests survive them. The +//! representative cached type answers as a store uuid rather than a generation ordinal, because an +//! edition written after a fit can carry a type no generation tabulated. Its nearest declared icon +//! rides the same row, resolved through the representative's current closed schema. A register +//! allocating a row for such a type has the icon in hand at the allocation. use hash_graph_postgres_store::store::postgres::query::{ Aliased, Binder, BoundStatement, ColumnName, Correlation, Expression, FromItem, Function, @@ -77,12 +76,12 @@ pub(crate) struct EditionDisplayColumns { /// Builds the display lookup over the requested editions. /// -/// The statement answers every requested edition exactly once, because every join is outer, -/// each joins at most one row - the edition cache through its primary key, the representative -/// type through the unique `(base_url, version)` pair, its type row through the ontology id, -/// and the icon lateral through its own `LIMIT 1` - and the unnested requests survive them all. -/// The statement binds no temporal axes. An edition id addresses one immutable row, so the -/// answer is the same at any read. +/// The statement answers every requested edition exactly once, because every join is outer, each +/// joins at most one row - the edition cache through its primary key, the representative type +/// through the unique `(base_url, version)` pair, its type row through the ontology id, and the +/// icon lateral through its own `LIMIT 1` - and the unnested requests survive them all. The +/// statement binds no temporal axes. An edition id addresses one immutable row. The answer is the +/// same at any read. /// /// # SQL /// @@ -193,10 +192,16 @@ pub(crate) struct DisplayParts { /// Decodes one edition-display row. /// -/// A row without a resolved representative type decodes as [`None`], so the caller's next read -/// cycle retries it: the register turns the representative into its ontology row, and an absent +/// A row without a resolved representative type decodes as [`None`]. The caller's next read cycle +/// retries it: the register turns the representative into its ontology row, and an absent /// representative leaves nothing to resolve. A row whose cache holds no label carries the empty /// label, and a representative whose chain declares no icon carries the empty icon. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Query`] when a selected column does not read back at the type +/// the decode asks for. A missing label, icon or representative is a null column rather than a +/// failure. pub(crate) fn decode_edition_display( row: &Row, columns: &EditionDisplayColumns, @@ -232,12 +237,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. Reviewing a diff, hold it to the - /// statement's own contract: every join is outer and joins at most one row, so every - /// requested edition answers exactly once. #[test] fn statement_text() { let edition_ids = vec![Uuid::nil()]; diff --git a/libs/@local/graph/atlas/src/postgres/embeddings.rs b/libs/@local/graph/atlas/src/postgres/embeddings.rs index c4a790549b6..b65643b2667 100644 --- a/libs/@local/graph/atlas/src/postgres/embeddings.rs +++ b/libs/@local/graph/atlas/src/postgres/embeddings.rs @@ -1,13 +1,14 @@ -//! The embedding lookups over requested identities. +//! Batched whole-entity embedding queries with full-width and projector projections. //! -//! One builder produces both statements, so the request join and the answer shape are a single -//! definition and the projection is the only difference between them. The canonical lookup -//! executes on the dataset's frozen-snapshot transaction and answers the stored whole-entity -//! embedding at full width. The projector lookup executes on its caller's own connection at -//! serving time and answers the embedding's -//! l2-normalized projector prefix, bit-identical to the representation row a fit reads for the -//! same stored embedding. Result identity keys each answer, and the caller counts the answers -//! against its requests. +//! Use [`canonical_embedding_statement`] for stored vectors at full width, or +//! [`projector_embedding_statement`] for the normalized prefix used as a dataset representation. +//! +//! Pass equal-length UUID arrays, pairing each web ID with the entity UUID at the same position. +//! Each result includes that identity. Match results by identity rather than result position: the +//! query has no ordering guarantee and omits identities without a whole-entity embedding. +//! +//! The lookup selects by entity identity, independently of edition selection. Visibility follows +//! the connection or transaction executing the statement. use hash_graph_postgres_store::store::postgres::query::{ Aliased, Binder, BoundStatement, Expression, SelectList, SelectStatement, SimpleSelect, Table, @@ -26,17 +27,20 @@ use crate::{ math::BoxedVecN, }; -/// The output columns of one embedding lookup. +/// Column positions for decoding an embedding query's identity and vector. pub(crate) struct EmbeddingLookupColumns { - /// The web the entity belongs to. + /// Position of the web ID. pub web_id: usize, - /// The entity's identity within its web. + /// Position of the entity UUID within that web. pub entity_uuid: usize, - /// The requested projection of the whole-entity embedding. + /// Position of the projected embedding. pub embedding: usize, } -/// Builds an embedding lookup over the requested identities, with the caller's projection. +/// Selects whole-entity embeddings with a configurable vector projection. +/// +/// `web_ids` and `entity_uuids` bind as UUID arrays paired by position. `projection` selects the +/// vector expression while the identity columns and whole-entity filter remain fixed. /// /// # SQL /// @@ -98,11 +102,10 @@ fn embedding_lookup<'params>( BoundStatement::new(&statement, binder, columns) } -/// Builds the canonical-embedding lookup over the requested identities. +/// Selects the stored full-width embedding for each requested entity. /// -/// The statement delivers the full-width embedding for every requested identity the store -/// holds a whole-entity embedding for. The caller counts the answers against its requests. A -/// missing row is an identity whose whole-entity embedding the store does not hold. +/// The query returns every matching whole-entity embedding unchanged. The UUID arrays must pair web +/// IDs and entity UUIDs by position. Decode results with [`decode_canonical_embedding`]. pub(crate) fn canonical_embedding_statement<'params>( web_ids: &'params (impl ToSql + Sync), entity_uuids: &'params (impl ToSql + Sync), @@ -112,12 +115,14 @@ pub(crate) fn canonical_embedding_statement<'params>( }) } -/// Builds the projector-input lookup over the requested identities. +/// Selects each requested entity's embedding prefix for projection. +/// +/// [`normalized_prefix`] selects the leading [`PROJECTOR_DIMENSIONS`] components and applies +/// pgvector's L2 normalization in the database. For the same stored embedding, the result is +/// bit-identical to its dataset representation. /// -/// The request shape is the canonical lookup's. The output is the embedding's l2-normalized -/// projector prefix through the node stream's own expression, so the connection carries -/// unit-norm prefixes and nothing wider, and an answer is bit-identical to the representation -/// row a fit reads for the same stored embedding. +/// The UUID arrays must pair web IDs and entity UUIDs by position. Decode results with +/// [`decode_projector_embedding`]. pub(crate) fn projector_embedding_statement<'params>( web_ids: &'params (impl ToSql + Sync), entity_uuids: &'params (impl ToSql + Sync), @@ -125,7 +130,12 @@ pub(crate) fn projector_embedding_statement<'params>( embedding_lookup(web_ids, entity_uuids, normalized_prefix) } -/// Decodes one canonical-embedding row. +/// Reads an entity identity and full-width vector from a query result. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Query`] if a selected column is missing or cannot decode, +/// including a vector with a width other than [`CANONICAL_DIMENSIONS`]. pub(crate) fn decode_canonical_embedding( row: &Row, columns: &EmbeddingLookupColumns, @@ -143,7 +153,12 @@ pub(crate) fn decode_canonical_embedding( )) } -/// Decodes one projector-input row. +/// Reads an entity identity and projector-input vector from a query result. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Query`] if a selected column is missing or cannot decode, +/// including a vector with a width other than [`PROJECTOR_DIMENSIONS`]. pub(crate) fn decode_projector_embedding( row: &Row, columns: &EmbeddingLookupColumns, @@ -170,7 +185,6 @@ mod tests { projector_embedding_statement, }; - /// Both lookups cite exactly the parameters they bind. #[test] fn statements_cite_their_whole_bind_list() { let web_ids = vec![Uuid::nil()]; @@ -183,10 +197,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered canonical lookup, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn canonical_statement_text() { let web_ids = vec![Uuid::nil()]; @@ -195,12 +205,6 @@ mod tests { insta::assert_snapshot!(canonical_embedding_statement(&web_ids, &entity_uuids).sql); } - /// The rendered projector lookup, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. Reviewing a diff, hold it to the - /// statement's own contract: the projection is the node stream's own normalized-prefix - /// expression, so an answer stays bit-identical to the representation row a fit reads. #[test] fn projector_statement_text() { let web_ids = vec![Uuid::nil()]; diff --git a/libs/@local/graph/atlas/src/postgres/id.rs b/libs/@local/graph/atlas/src/postgres/id.rs index 5ec46e24dfd..e6908816ff5 100644 --- a/libs/@local/graph/atlas/src/postgres/id.rs +++ b/libs/@local/graph/atlas/src/postgres/id.rs @@ -1,3 +1,12 @@ +//! Fixed-size store identities for dataset records and mapped identity tables. +//! +//! These identities have byte alignment and compare and hash by their stored bytes. UUID components +//! retain the byte order of [`uuid::Uuid::as_bytes`], independent of the host's endianness. +//! +//! [`ArchivedEntityId`] combines a web ID and an entity UUID to identify a non-draft entity. +//! [`ArchivedOntologyTypeUuid`] identifies a versioned ontology type by its URL-derived UUID. Both +//! implement [`Key`] for storage in identity tables. + use core::ops::Deref; use type_system::{ @@ -14,9 +23,10 @@ use crate::{ file::identity::{Key, KeyKind}, }; -/// The byte-level form of an [`EntityUuid`]. +/// An entity UUID stored as 16 bytes. /// -/// The derived order is uuid-byte order. +/// Conversion to and from [`EntityUuid`] preserves all UUID bytes. Ordering is lexicographic in +/// [`uuid::Uuid::as_bytes`] order. #[derive( Debug, Copy, @@ -41,7 +51,7 @@ impl ArchivedEntityUuid { Self(bytes) } - /// Returns the raw uuid bytes. + /// Returns the UUID bytes in their stored order. pub(crate) const fn to_bytes(self) -> [u8; 16] { self.0 } @@ -85,9 +95,10 @@ impl Deref for ArchivedEntityUuid { } } -/// The byte-level form of a [`WebId`]. +/// A web ID stored as 16 UUID bytes. /// -/// The derived order is uuid-byte order. +/// Conversion to and from [`WebId`] preserves all UUID bytes. Ordering is lexicographic in +/// [`uuid::Uuid::as_bytes`] order. #[derive( Debug, Copy, @@ -112,7 +123,7 @@ impl ArchivedWebId { Self(bytes) } - /// Returns the raw uuid bytes. + /// Returns the UUID bytes in their stored order. pub(crate) const fn to_bytes(self) -> [u8; 16] { self.0 } @@ -157,11 +168,15 @@ impl Deref for ArchivedWebId { } } -/// The byte-level form of a non-draft entity identity. +/// A non-draft entity identity stored as a web ID followed by an entity UUID. +/// +/// The representation is 32 bytes with no padding. Ordering compares web ID bytes first, then +/// entity UUID bytes. Serialization uses [`EntityId`]'s `web_id~entity_uuid` string form. /// -/// Drafts never enter a dataset's scope, so the identity is the web and entity components alone. +/// # Warning /// -/// The derived order is identity-byte order: web id bytes, then entity uuid bytes. +/// Conversion from [`EntityId`] discards any draft ID. Conversion back always sets `draft_id` to +/// [`None`]. Only non-draft entity identities round-trip without loss. #[derive( Debug, Copy, @@ -184,7 +199,7 @@ pub(crate) struct ArchivedEntityId { pub entity_uuid: ArchivedEntityUuid, } -// No multi-byte fields: both components are byte arrays, so no byte order arises. +// UUID components are byte arrays with no host-endian integer fields. crate::dataset::offline::portable::self_archived!(ArchivedEntityId); impl From for ArchivedEntityId { @@ -212,7 +227,10 @@ impl Key for ArchivedEntityId { const KIND: KeyKind = KeyKind::EntityId; } -/// The byte-level form of an [`OntologyTypeUuid`]. +/// A versioned ontology type's identity stored as 16 UUID bytes. +/// +/// Construct it from a [`uuid::Uuid`] or derive it from a [`VersionedUrl`] with +/// [`from_url`](Self::from_url). The bytes represent the same identity as [`OntologyTypeUuid`]. #[derive( Debug, Copy, @@ -228,14 +246,14 @@ impl Key for ArchivedEntityId { #[repr(transparent)] pub(crate) struct ArchivedOntologyTypeUuid([u8; 16]); -// No multi-byte fields: the identity is a byte array, so no byte order arises. +// the UUID is a byte array with no host-endian integer fields. crate::dataset::offline::portable::self_archived!(ArchivedOntologyTypeUuid); impl ArchivedOntologyTypeUuid { - /// Derives the identity of the versioned type `url` names. + /// Derives a UUID v5 from the versioned type URL. /// - /// Any [`VersionedUrl`] spelling the same versioned type derives the same identity, so - /// equality on the result is equality of the named type. + /// The derivation uses the URL namespace and the UTF-8 bytes of `url`'s string form, matching + /// [`OntologyTypeUuid::from_url`]. #[inline] pub(crate) fn from_url(url: &VersionedUrl) -> Self { Self::from(OntologyTypeUuid::from_url(url).into_uuid()) diff --git a/libs/@local/graph/atlas/src/postgres/legends.rs b/libs/@local/graph/atlas/src/postgres/legends.rs index 0f2e64c65f6..fbca1beadd2 100644 --- a/libs/@local/graph/atlas/src/postgres/legends.rs +++ b/libs/@local/graph/atlas/src/postgres/legends.rs @@ -45,7 +45,7 @@ fn representative_ordinal() -> Expression { /// Builds the joins resolving the edition cache's representative type to its type-table ordinal. /// -/// Both joins are outer, so a missing cache entry or a representative type outside the type table +/// Both joins are outer. A missing cache entry or a representative type outside the type table /// leaves the ordinal SQL NULL for the decoder to refuse. /// /// # SQL @@ -256,6 +256,16 @@ pub(crate) fn edge_legend_statement<'params>( /// /// An edition without a cached label decodes as the empty label. A row whose representative type /// resolves to no type-table ordinal fails the decode. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError`]: [`Query`] when a selected column does not read back at the +/// type the decode asks for, then [`Representative`] for a null representative, then [`Ordinal`] +/// for a representative the store returns as a negative ordinal. +/// +/// [`Query`]: PostgresDatasetError::Query +/// [`Representative`]: PostgresDatasetError::Representative +/// [`Ordinal`]: PostgresDatasetError::Ordinal pub(crate) fn decode_legend( row: &Row, columns: &LegendColumns, @@ -298,12 +308,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered node-legend statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. Reviewing a diff, hold it to the - /// statement's own contract: the ordering is the node stream's, so positions agree under - /// the frozen snapshot. #[test] fn node_statement_text() { let axes = TemporalAxes::now(); @@ -312,12 +316,6 @@ mod tests { insta::assert_snapshot!(node_legend_statement(&axes, &types).sql); } - /// The rendered edge-legend statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. Reviewing a diff, hold it to the - /// statement's own contract: the ordering is the edge stream's link identity, so - /// positions agree under the frozen snapshot. #[test] fn edge_statement_text() { let axes = TemporalAxes::now(); diff --git a/libs/@local/graph/atlas/src/postgres/node_types.rs b/libs/@local/graph/atlas/src/postgres/node_types.rs index c60e495a14d..a8930b1b654 100644 --- a/libs/@local/graph/atlas/src/postgres/node_types.rs +++ b/libs/@local/graph/atlas/src/postgres/node_types.rs @@ -116,6 +116,11 @@ pub(crate) fn node_type_statement<'params>( } /// Converts a column of SQL ordinals into ontology row references. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Ordinal`] carrying the first ordinal that is not a `u64`, which +/// is a negative value, since the signed column is the only width mismatch a store can deliver. pub(crate) fn ontology_rows( ordinals: Vec, ) -> Result, PostgresDatasetError> { @@ -130,6 +135,16 @@ pub(crate) fn ontology_rows( } /// Decodes one direct-type row. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError`]: [`Query`] when a selected column does not read back at the +/// type the decode asks for, then [`Ordinal`] when the ordinal column carries a value that is not +/// a row reference. An edition with no direct types does not fail here, because the statement +/// coalesces its absent row to the empty ordinal array. +/// +/// [`Query`]: PostgresDatasetError::Query +/// [`Ordinal`]: PostgresDatasetError::Ordinal pub(crate) fn decode_node_types( row: &Row, columns: &NodeTypeColumns, @@ -166,10 +181,6 @@ mod tests { assert_placeholders_dense(&statement.sql, statement.parameters.len()); } - /// The rendered statement, pinned as the text the store receives. - /// - /// The pin makes any rendering change a visible snapshot diff in review instead of a - /// silent swap of what runs against the store. #[test] fn statement_text() { let axes = TemporalAxes::now(); diff --git a/libs/@local/graph/atlas/src/postgres/ontology.rs b/libs/@local/graph/atlas/src/postgres/ontology.rs index 1f45a3331fb..a1385a83b5e 100644 --- a/libs/@local/graph/atlas/src/postgres/ontology.rs +++ b/libs/@local/graph/atlas/src/postgres/ontology.rs @@ -1,10 +1,9 @@ //! The ontology payload lookups, under the frozen-snapshot regime. //! -//! Both lookups take the dataset's transaction directly and execute their statements -//! themselves, reshaping the rows before streaming: the supertype rows gather into per-type -//! parent lists, and the icon rows decode into owned icons. The answers arrive in ontology row -//! order, the order the bound type table fixes, so a position here names the same type at -//! every consumer of the table. +//! Both lookups take the dataset's transaction directly and execute their statements themselves, +//! reshaping the rows before streaming: the supertype rows gather into per-type parent lists, and +//! the icon rows decode into owned icons. The answers arrive in ontology row order, the order the +//! bound type table fixes. A position here names the same type at every consumer of the table. use futures::{Stream, stream}; use hash_graph_postgres_store::store::postgres::query::{ @@ -27,13 +26,22 @@ use crate::{ /// Opens the ontology stream: each type's direct supertypes, in ontology row order. /// -/// Parents outside the type table cannot occur: the store materializes closures per edition, so -/// every depth-0 parent of a reachable type is itself reachable. +/// Sort `types` by UUID: each row's source and target reach their row through a binary search +/// over it. The stream drops a parent the slice does not hold. It therefore yields the whole +/// parent relation only for a slice closed under depth-0 parents. The store materializes closures +/// per edition, which is what closes the table this dataset supplies. +/// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Query`] when the store rejects the read or an inheritance row +/// does not decode. This call reads every row before it builds the stream, and the stream itself +/// yields no further failures. /// /// # Panics /// -/// This panics when the store returns a source outside the type table, which the statement's -/// own filter forbids. +/// This panics when the binary search misses a returned source, which an unsorted slice causes +/// even for a source the slice does hold. The statement's own filter keeps returned sources +/// inside the table. /// /// # SQL /// @@ -132,6 +140,12 @@ pub(crate) async fn ontology<'t>( /// survives them. The lateral picks the nearest declared icon in the type's closed schema, and /// a chain without one answers SQL NULL for the decoder to default. /// +/// # Errors +/// +/// Returns [`PostgresDatasetError::Query`] when the store rejects the read. The stream decodes +/// each row as the caller pulls it. A column that does not read back fails at that item rather +/// than here. +/// /// # SQL /// /// ```sql diff --git a/libs/@local/graph/atlas/src/postgres/sql.rs b/libs/@local/graph/atlas/src/postgres/sql.rs index 7e316e41a25..e75a39d628a 100644 --- a/libs/@local/graph/atlas/src/postgres/sql.rs +++ b/libs/@local/graph/atlas/src/postgres/sql.rs @@ -6,12 +6,12 @@ //! //! - [`current_identity_join`], [`time_axis_conjunction`] and [`edition_conjunction`] hold //! "current" to one definition across every statement that resolves an entity or its edition. -//! - [`type_mapping`] unnests the bound type table with its ordinality, so the store re-derives -//! every type's position and both ends share the ordinal map by construction. +//! - [`type_mapping`] unnests the bound type table with its ordinality. The store re-derives every +//! type's position, and both ends share the ordinal map by construction. //! - [`Axes`] and [`AttachmentVocabulary`] bind the axis points and the link-attachment //! discriminants as the store's own wire-typed parameters. -//! - [`json_text`] and [`json_field`] route JSON keys through [`PathToken`], so a key renders -//! through the store's own quoting and carries a name at the site that uses it. +//! - [`json_text`] and [`json_field`] route JSON keys through [`PathToken`]. A key renders through +//! the store's own quoting and carries a name at the site that uses it. //! - [`first_label`] reads the edition cache's first display label, the one spelling every legend //! and display statement shares. //! - [`nearest_declared_icon`] picks a type's nearest declared icon out of its closed schema, the @@ -35,8 +35,8 @@ use crate::dataset::TemporalAxes; /// The temporal-axes placeholders every currency condition consumes. /// -/// One bind per axis, shared by every fragment of the statement, so the statement carries the -/// axes once however many conditions cite them. +/// One bind per axis, shared by every fragment of the statement. The statement carries the axes +/// once however many conditions cite them. #[derive(Debug, Copy, Clone)] pub(crate) struct Axes { /// The transaction-time point. @@ -57,7 +57,7 @@ impl Axes { /// The link-attachment discriminants, bound as their store-typed values. /// -/// The values travel as parameters of the store's own enum types, so the statement compares the +/// The statement binds the values as parameters of the store's own enum types. It compares the /// `kind` and `direction` columns against values the wire protocol type-checks instead of against /// quoted literals a schema migration can silently strand. #[derive(Debug, Copy, Clone)] @@ -162,15 +162,15 @@ pub(crate) fn first_label(cache: Aliased) -> Expression { cache.column(&EntityEditionCache::Labels).array_element(1) } -/// The `uuid[]` type, for casting a bound identity array where inference needs the annotation. +/// Returns the `uuid[]` cast type where a bound identity array needs an explicit annotation. pub(crate) fn uuid_array() -> PostgresType { PostgresType::Array(Box::new(PostgresType::Uuid)) } /// Extracts the text at a JSON key. /// -/// The key travels as a [`PathToken`], so it renders through the store's own key quoting. Pass -/// a named constant, so the key's meaning has a name at the site that uses it. +/// The key renders as a quoted field name through [`PathToken`]. Pass a named constant: the key's +/// meaning has a name at the site that uses it. /// /// # SQL /// @@ -186,8 +186,8 @@ pub(crate) fn json_text(expression: impl Into, key: &'static str) -> /// Extracts the `jsonb` at a JSON key. /// -/// The key travels as a [`PathToken`], so it renders through the store's own key quoting. Pass -/// a named constant, so the key's meaning has a name at the site that uses it. +/// The key renders as a quoted field name through [`PathToken`]. Pass a named constant: the key's +/// meaning has a name at the site that uses it. /// /// # SQL /// @@ -231,10 +231,9 @@ impl DatabaseColumn<'_> for Mapping { /// Builds the `mapping` rows: the bound type table unnested beside each type's ordinality. /// -/// The type table travels as one bound array and the store re-derives the position of every -/// type, so both ends share the ordinal map by construction. Every statement that resolves -/// ordinals builds its FROM item here, which is what makes the shared derivation one -/// declaration. +/// The builder binds the type table as one array and the store re-derives the position of every +/// type. Both ends share the ordinal map by construction. Every statement that resolves ordinals +/// builds its FROM item here, which is what makes the shared derivation one declaration. /// /// # SQL /// @@ -298,8 +297,7 @@ impl DatabaseColumn<'_> for NearestIcon { /// The subquery picks the nearest declared icon in the joined type's closed schema: ascending /// inheritance depth, position in the `allOf` array breaking ties, one row at most. A chain /// without a declared icon answers no row, which a LEFT LATERAL join turns into SQL NULL for -/// the decoder to default. The selection rule mirrors the serving side's type-icon resolution -/// in `serve::hydrate`'s tile hydration query, and a change to either belongs in both. +/// the decoder to default. Every read of the rule builds it here, from this one definition. /// /// # SQL /// @@ -373,6 +371,11 @@ pub(crate) fn nearest_declared_icon(types: Aliased) -> SelectStatem /// statement at execution with an unread-parameter error. The scan also catches a placeholder /// rendered without a bind, which the kit cannot produce but a hand-assembled statement could /// reintroduce. +/// +/// # Panics +/// +/// Panics when the placeholders `sql` cites are not exactly `$1` through `$parameter_count`, +/// which is either failure above. #[cfg(test)] // Every statement module's tests assert placeholder density. #[track_caller] pub(crate) fn assert_placeholders_dense(sql: &str, parameter_count: usize) { diff --git a/libs/@local/graph/atlas/src/postgres/vector.rs b/libs/@local/graph/atlas/src/postgres/vector.rs index 60bf96d3158..f722eb42f6f 100644 --- a/libs/@local/graph/atlas/src/postgres/vector.rs +++ b/libs/@local/graph/atlas/src/postgres/vector.rs @@ -45,7 +45,7 @@ pub(crate) fn normalized_prefix(embedding: Aliased) -> Express pub(crate) enum VectorDecodeError { /// The four-byte header is truncated. Header, - /// The header's dimensions or the payload length disagree with the expected shape. + /// The dimensions or payload size are wrong, or the reserved header word is nonzero. Shape { /// The compile-time component count. expected: usize, @@ -79,6 +79,11 @@ impl Error for VectorDecodeError {} pub(crate) struct PgVector(pub BoxedVecN); impl<'value, const N: usize> FromSql<'value> for PgVector { + /// Decodes pgvector's binary wire form into an aligned `N`-component vector. + /// + /// # Errors + /// + /// Returns [`VectorDecodeError`] for a truncated header or a value with the wrong vector shape. #[expect( clippy::big_endian_bytes, reason = "pgvector's binary protocol uses network byte order" @@ -107,7 +112,7 @@ impl<'value, const N: usize> FromSql<'value> for PgVector { })); } - // The components decode straight into the aligned buffer; the + // The components decode straight into the aligned buffer. The // shape check above pinned their count to exactly `N`. let mut decoded = BoxedVecN::::zero(); for (slot, &bytes) in decoded @@ -122,6 +127,7 @@ impl<'value, const N: usize> FromSql<'value> for PgVector { } fn accepts(ty: &Type) -> bool { + // pgvector's OID is assigned per database when the extension is installed. ty.name() == "vector" } } diff --git a/libs/@local/graph/atlas/src/postgres/vocabulary.rs b/libs/@local/graph/atlas/src/postgres/vocabulary.rs index bf90648f118..a48e67847af 100644 --- a/libs/@local/graph/atlas/src/postgres/vocabulary.rs +++ b/libs/@local/graph/atlas/src/postgres/vocabulary.rs @@ -47,6 +47,11 @@ impl CorpusTable { self.as_str().into() } + /// Returns the table as a query correlation, unqualified by a schema. + /// + /// These tables are common table expressions rather than schema objects: the name lives in + /// the statement's own `WITH` clause, and a schema qualifier would send the resolver looking + /// for a real table of that name instead. pub(crate) fn reference(self) -> TableReference<'static> { TableReference { schema: None, @@ -54,6 +59,10 @@ impl CorpusTable { } } + /// Returns `reference` as a column of this table, qualified by the table's own correlation. + /// + /// The qualifier is what keeps a column name that two joined relations share from being + /// ambiguous where both are in scope. pub(crate) fn column(self, reference: impl DatabaseColumn<'static> + Copy) -> Expression { Expression::ColumnReference(ColumnReference { correlation: Some(self.reference()), diff --git a/libs/@local/graph/atlas/src/progress.rs b/libs/@local/graph/atlas/src/progress.rs index 0c3d3b70fb2..b5dd9f24dff 100644 --- a/libs/@local/graph/atlas/src/progress.rs +++ b/libs/@local/graph/atlas/src/progress.rs @@ -1,24 +1,12 @@ -//! Observation of a running fit. +//! Progress reports for fitting and generation admission. //! -//! [`Progress`] is the seam operator surfaces render from. It carries the pipeline's observations -//! to whatever the operator is watching, whether that is nothing, a log stream, or a live -//! dashboard. The observations are stage boundaries, batch counters, convergence readouts, and -//! quality probes. The trait observes and never steers. Every value flows outward, and a run -//! behaves identically under any observer. Each method has an empty default body, so an observer -//! implements exactly the observations it renders and the rest monomorphize to no-ops that cost -//! nothing. +//! [`Progress`] receives observations during a run: [`Stage`] completion, [`Batch`] counters, +//! neighbour-list update rates and fit or quality measurements. Implement callbacks for the +//! observations your log or display needs, or use [`NoProgress`] to ignore them. //! -//! Observations travel as the pipeline's own types wherever one exists - [`CardEmbeddingStats`], -//! [`RecallSpotCheck`], [`LossBreakdown`], [`QualityMetric`], re-exported here - and as this -//! module's observation vocabulary ([`Stage`], [`Batch`], [`DescentIteration`]) where the pipeline -//! reports something no artifact records. -//! -//! An observer crosses the run's thread seams - the async ingest half and the rayon compute half - -//! so implementations are cloneable and shareable by construction; a renderer typically holds the -//! sending half of a channel and does its drawing elsewhere. Hot loops report at batch cadence, -//! never per row. -//! -//! [`NoProgress`] is the silent observer, for runs nothing watches. +//! Callbacks execute in the task reporting the observation. A display can enqueue observations for +//! another task to render, keeping its I/O off the fitting path. [`Progress::Detached`] provides an +//! owned observer for reporting work that cannot borrow the original observer. use crate::{ math::Vec2, @@ -59,25 +47,19 @@ pub enum Stage { } impl Stage { - /// Every stage, in the order the runner drives them. - /// - /// A renderer showing the run's remaining work needs the order before the run reaches it, so - /// this constant states the sequence once instead of leaving a renderer to infer it from - /// arrival. + /// Every stage, in pipeline order. #[expect( clippy::cast_possible_truncation, reason = "the index runs over the variant count, an order of magnitude inside u8" )] pub const ALL: [Self; core::mem::variant_count::()] = - // SAFETY: every variant is a unit variant of a `repr(u8)` enum. The discriminants are - // therefore exactly `0..variant_count`, and `from_fn` calls the closure once per index of - // that range. + // SAFETY: A fieldless `repr(u8)` enum has u8 layout and admits its declared discriminants. + // These variants use consecutive implicit discriminants starting at zero, and `from_fn` + // supplies exactly those indices. Therefore every converted index is a valid `Stage` + // value. core::array::from_fn(const |index| unsafe { core::mem::transmute(index as u8) }); - /// The stage's name, in the vocabulary a run reports it under. - /// - /// One lowercase word per stage, so a log line, a rail row, and a report name the same stage - /// the same way. + /// Returns the lowercase name used in progress output. #[must_use] pub const fn label(self) -> &'static str { match self { @@ -106,109 +88,127 @@ pub struct Batch { pub total: usize, } -/// One NN-Descent iteration's convergence reading. +/// One NN-Descent iteration's update-rate observation. /// -/// The construction stops when `accepted_per_entry` falls to `threshold`. +/// The constructor rounds the configured rate's count threshold up before comparing accepted +/// updates. It can also stop at its iteration limit. The reported `threshold` is the configured +/// rate before count rounding. #[derive(Debug, Copy, Clone, PartialEq)] pub struct DescentIteration { /// One-based index of the completed iteration. pub iteration: usize, /// Neighbour updates the iteration accepted, per stored list entry. /// - /// Not a share of anything: a local join offers a pair to both sides and one iteration can - /// displace the same entry more than once, so an early reading stands above `1`. The reading - /// measures convergence. It falls as the lists stop changing. + /// A local join offers a pair to both neighbour lists, and an entry can change more than once + /// in an iteration. The rate can exceed `1` and need not decrease between iterations. pub accepted_per_entry: f64, - /// The convergence threshold the reading is falling toward. + /// The configured update rate for the stopping criterion. pub threshold: f64, } -/// The observer of one run's progress. +/// An observer of stage completion and fit measurements. +/// +/// Observation callbacks have no-op defaults. Override the callbacks your output needs and +/// implement [`detach`](Self::detach) to supply an owned observer. Callbacks return no fitting +/// decisions. [`projector_sample_size`](Self::projector_sample_size) controls snapshot gathering. /// -/// Every method is an observation the pipeline reports as it happens; none returns anything the run -/// acts on, with one deliberate exception: [`projector_sample_size`](Self::projector_sample_size) -/// is a capability probe whose value is the observer's own appetite. The placement the run -/// publishes is identical under every observer. +/// Reporting is synchronous. A callback's blocking work delays the reporting task, and a panic can +/// interrupt it. Some operations report from parallel workers, requiring a shared observer to +/// support concurrent callbacks. #[expect( unused_variables, reason = "the default bodies observe nothing; the parameter names document each observation \ for implementors" )] pub trait Progress { - /// The observer a stage hands to machinery that owns its reporter. + /// An owned observer for work that cannot borrow this observer. /// - /// A backend that reports through a foreign builder cannot lend this observer. The builder - /// takes its reporter by value and keeps it for the call. Each observer answers with whatever - /// it can give away: [`NoProgress`] when nothing crosses, or a handle onto its own sink. What - /// crosses observes exactly what that answer observes. + /// Use [`NoProgress`] when detached work needs no reporting. A reporting implementation can + /// return an owned handle to the original observation destination. type Detached: Progress + Send + Sync + 'static; - /// Hands out this observer's detached half. + /// Returns an owned observer for independently reported work. fn detach(&self) -> Self::Detached; - /// The card-embedding stage resolved its reuse split: `stats.reused` unique texts serve from - /// the prior generation, `stats.embedded` go to the provider. + /// Reports the split between reusable and new card embeddings. + /// + /// `stats.reused` counts unique texts copied from the prior generation, and `stats.embedded` + /// counts unique texts requiring the provider. This precedes provider work, including when no + /// text needs embedding. fn embedding_started(&self, stats: &CardEmbeddingStats) {} - /// The provider finished another embedding chunk. + /// Reports progress after an embedding chunk completes. fn embedding_batch(&self, batch: Batch) {} - /// The corpus assembly derived its near-duplicate boundary. + /// Reports the near-duplicate boundary derived during corpus assembly. fn assembly_boundary_derived(&self, epsilon: f64) {} - /// The neighbour-table construction entered a named backend phase. + /// Reports the start of a neighbour-index backend phase. /// - /// The names are the backend's own open vocabulary (the HNSW backend reports its build steps), - /// passed through verbatim. + /// Phase names use the backend's vocabulary without translation. fn knn_build_phase(&self, phase: &str) {} - /// The neighbour-table construction inserted another batch of rows. + /// Reports progress through the index's input rows. + /// + /// The count tracks rows requested by the backend, not a committed transaction. fn knn_insert(&self, batch: Batch) {} - /// An NN-Descent iteration completed with its convergence reading. + /// Reports the update rate after an NN-Descent iteration. fn descent_iteration(&self, iteration: DescentIteration) {} - /// The neighbour-table readback covered another batch of rows. + /// Reports completed rows of the neighbour-table readback. + /// + /// Parallel workers can invoke this callback out of count order. `batch.done` counts completed + /// rows rather than identifying a row. fn knn_readback(&self, batch: Batch) {} - /// The construction's measured recall against the exact reference sample. + /// Reports measured neighbour recall against an exact reference sample. fn knn_recall(&self, check: &RecallSpotCheck) {} - /// The projector finished training step `step` of `steps` at the reported loss. + /// Reports the loss before a training step's optimizer update. + /// + /// `step` is zero-based within the full schedule, including during a resumed segment. `steps` + /// is the schedule's total step count. fn projector_step(&self, step: usize, steps: usize, loss: &LossBreakdown) {} - /// How many placement rows the observer wants sampled into - /// [`projector_snapshot`](Self::projector_snapshot) calls. + /// Requests the maximum number of placement rows in a snapshot. /// - /// The capability probe: `0`, the default, means the run never gathers a snapshot. The run - /// chooses the rows once at stage start, taking the landmark skeleton first and then an even - /// stride over the corpus, and every snapshot reports those same rows moving. The choice draws - /// no randomness, so an observer's appetite cannot move what the run publishes. + /// Returns `0` by default, disabling [`projector_snapshot`](Self::projector_snapshot) calls. A + /// positive budget selects at most that many rows at the start of each training segment. + /// Snapshots within that segment report positions for the same rows, with sampled landmarks + /// first and non-landmark rows after them. + /// + /// Selection uses an even spread within each group and consumes no training randomness. The + /// budget controls snapshot allocation and copying, not the rows used for training. Training + /// still performs its refresh computations when the budget is zero. fn projector_sample_size(&self) -> usize { 0 } - /// The sampled placement positions at a training refresh. + /// Reports sampled placement coordinates at a training refresh. /// - /// The landmark rows are `positions[..landmarks]`. + /// `positions[..landmarks]` contains the sampled landmark positions. The remaining positions + /// belong to non-landmark rows. fn projector_snapshot(&self, positions: &[Vec2], landmarks: usize) {} - /// The retrospective arrival replay projected another batch of sampled arrivals. + /// Reports progress through the sampled arrivals of a retrospective replay. fn replay_projection(&self, batch: Batch) {} - /// The classifier fit started over `folds` cross-validation folds. + /// Reports the start of classifier fitting over `folds` cross-validation folds. fn classifier_started(&self, folds: usize) {} - /// One classifier cross-validation fold completed. + /// Reports completion of the candidate fits for one cross-validation fold. + /// + /// `fold` is a zero-based index. Folds can complete out of index order. fn classifier_fold_completed(&self, fold: usize) {} - /// The classifier fit selected its regularization strength. + /// Reports the regularization strength selected by classifier fitting. fn classifier_regularization_selected(&self, regularization: f64) {} - /// The admission probe measured one quality metric. + /// Reports an admission metric's aggregate reading across the probe steps. fn quality_probe(&self, metric: QualityMetric, value: f64) {} - /// A pipeline stage completed. + /// Reports completion of a pipeline stage. fn stage_completed(&self, stage: Stage) {} } @@ -291,7 +291,7 @@ where } } -/// The silent observer, whose observations are all no-ops. +/// An observer that ignores progress and requests no snapshots. #[derive(Debug, Copy, Clone, Default, PartialEq, Eq)] pub struct NoProgress; diff --git a/libs/@local/graph/atlas/src/random/compat.rs b/libs/@local/graph/atlas/src/random/compat.rs index 76cc888c1c4..aa9617de870 100644 --- a/libs/@local/graph/atlas/src/random/compat.rs +++ b/libs/@local/graph/atlas/src/random/compat.rs @@ -1,10 +1,17 @@ use rand_core as rc10; use rand_core_06 as rc06; +/// An infallible generator exposed through the `rand_core` 0.6 traits. +/// +/// Matching word and byte-generation calls consume the underlying generator unchanged. Higher-level +/// distributions from different rand versions may consume those words differently. The +/// [`rc06::RngCore::try_fill_bytes`] implementation always returns `Ok` because [`rc10::Rng`] +/// requires an infallible error type. #[repr(transparent)] pub(crate) struct Compat(R); impl Compat { + /// Adapts `rng` without reseeding or consuming any output. pub(crate) const fn new(rng: R) -> Self where R: Sized, diff --git a/libs/@local/graph/atlas/src/random/mod.rs b/libs/@local/graph/atlas/src/random/mod.rs index 057f3836fe9..2bfc7bd92cc 100644 --- a/libs/@local/graph/atlas/src/random/mod.rs +++ b/libs/@local/graph/atlas/src/random/mod.rs @@ -1,26 +1,15 @@ -//! Sampling utilities. +//! Random samples and statistical sample-size estimates. //! -//! Unbiased bounded integers, subset sampling, and statistical acceptance verification. +//! [`uniform_below`] draws bounded integers, while [`sample_indices_vec`] and [`sample_ids`] sample +//! without replacement. Uniformity assumes uniform generator words and follows rand's +//! range-sampling guarantees. Replaying a seeded draw requires the same generator, sampling +//! implementation and sequence of calls. //! -//! Every sampler draws from a caller-provided [`Rng`], so any generator works and seeded runs -//! reproduce exactly: -//! -//! - [`uniform_below`] draws one unbiased integer below a bound. -//! - [`sample_indices_vec`] draws distinct indices without replacement, in memory proportional to -//! the sample rather than the population. -//! - [`sample_ids`] draws distinct ids of an id-indexed population, typing the draw by the -//! population it came from. -//! - [`keyed_rng`] builds an independent generator per `(seed, key, stream)`, which is what keeps -//! parallel draws deterministic. -//! - [`acceptance_sample_size`] reports how many uniformly sampled items must all pass to certify a -//! defect-rate bound at a confidence level. -//! - [`mean_sample_size`] reports how many uniformly sampled items estimate a mean within a margin -//! at a confidence level for a given per-item deviation. -//! - [`normal_quantile`] inverts the standard normal distribution and supplies the `z` factor that -//! [`mean_sample_size`] uses. -//! -//! The module is crate-internal. Its examples carry `ignore` and spell each call as an in-crate -//! caller writes it, and the module's tests assert every property the examples show. +//! [`keyed_rng`] derives a generator from `(seed, key, stream)` without shared mutable state. +//! [`acceptance_sample_size`] sizes an all-pass check against a defect-rate threshold. +//! [`mean_sample_size`] uses a normal approximation to size a mean estimate, with +//! [`normal_quantile`] supplying the quantile. The statistical models and floating-point limits are +//! documented on the sizing functions. use core::num::NonZero; @@ -38,22 +27,26 @@ mod compat; #[cfg(test)] mod tests; -/// Draws an unbiased uniform integer in `[0, bound)`. +/// Draws an integer in `[0, bound)`. +/// +/// The nonzero bound excludes an empty range. With uniform generator words, rand's range sampler +/// has a small mapping bias unless its `unbiased` feature is enabled. Rand bounds the affected +/// fraction of `u64` draws by 2⁻⁶⁴. This function inherits that sampling behavior. /// -/// Every value below the bound is exactly equally likely; the draw consumes a small bounded -/// expected number of generator words. The non-zero bound makes the empty range unrepresentable, so -/// the draw always succeeds. +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private. /// /// ```ignore -/// use core::num::NonZero; +/// # use crate::math::nz; /// /// use rand::SeedableRng as _; /// use rand_xoshiro::Xoshiro256PlusPlus; /// +/// use crate::random::uniform_below; +/// /// let mut rng = Xoshiro256PlusPlus::seed_from_u64(42); -/// let sides = NonZero::new(6).expect("a die has sides"); +/// let sides = nz!(6); /// let roll = uniform_below(&mut rng, sides) + 1; /// assert!((1..=6).contains(&roll)); /// ``` @@ -63,23 +56,26 @@ pub(crate) fn uniform_below(mut rng: impl Rng, bound: NonZero) -> u64 { rng.random_range(0..bound.get()) } -/// Samples `count` distinct indices from `[0, population)` uniformly at random. +/// Samples `count` distinct indices from `[0, population)` in shuffled order. /// -/// The sample draws without replacement, so every `count`-element subset of the population is -/// equally likely, and the returned order is itself uniformly random. Memory scales with the sample -/// size rather than the population. +/// Sampling is without replacement and follows rand's range-sampling guarantees. Sparse requests +/// use memory proportional to `count`. For denser requests, the sampler may allocate an index for +/// every member of the population. /// /// # Panics /// -/// This panics when `count` exceeds `population`. No `count`-element sample exists to draw, and an -/// oversized request is a caller bug rather than a runtime condition. +/// This panics when `count` exceeds `population`. +/// +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private. /// /// ```ignore /// use rand::SeedableRng as _; /// use rand_xoshiro::Xoshiro256PlusPlus; /// +/// use crate::random::sample_indices_vec; +/// /// let mut rng = Xoshiro256PlusPlus::seed_from_u64(42); /// let picked = sample_indices_vec(&mut rng, 1_000_000, 688); /// assert_eq!(picked.len(), 688); @@ -90,31 +86,37 @@ pub(crate) fn sample_indices_vec(mut rng: impl Rng, population: usize, count: us sample(&mut rng, population, count) } -/// Samples `count` distinct ids of an id-indexed population uniformly at random. +/// Samples `count` distinct typed positions in shuffled order. /// -/// The typed form of [`sample_indices_vec`]: the id domain comes from the population itself, so a -/// draw cannot pair one population's length with another population's id type. The sample draws -/// without replacement, the yielded order is itself uniformly random, and both forms consume the -/// identical generator stream, so a seeded draw is unchanged by adopting the typed form. Every -/// yielded id is a position of the population, below its length by the draw and within the id's -/// representation by the slice's own construction. The iterator reports its exact length, which -/// is `count`. +/// This has the sampling behavior and memory cost of [`sample_indices_vec`], with the population +/// length and ID type supplied together. Both forms consume the identical generator stream for +/// equal lengths, counts and starting generator states. The returned iterator initially has exactly +/// `count` elements. +/// +/// Every sampled position must be representable by `I`. [`IdSlice::from_raw`] preserves lengths +/// beyond the ID range and does not establish this condition. /// /// # Panics /// -/// This panics when `count` exceeds the population's length. No `count`-element sample exists to -/// draw, and an oversized request is a caller bug rather than a runtime condition. +/// This panics when `count` exceeds the population's length. Iterating the result panics if a +/// sampled position is outside `I`'s range. +/// +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use hashql_core::id::{IdSlice, newtype}; /// use rand::SeedableRng as _; /// use rand_xoshiro::Xoshiro256PlusPlus; /// +/// use crate::random::sample_ids; +/// +/// newtype!(struct SampleId(u32)); /// let mut rng = Xoshiro256PlusPlus::seed_from_u64(42); -/// let population = IdSlice::::from_raw(&[(); 1_000_000]); -/// let picked: Vec = sample_ids(&mut rng, population, 688).collect(); -/// assert_eq!(picked.len(), 688); +/// let population = IdSlice::::from_raw(&[(); 4096]); +/// let picked: Vec = sample_ids(&mut rng, population, 128).collect(); +/// assert_eq!(picked.len(), 128); /// ``` #[inline] pub(crate) fn sample_ids( @@ -127,8 +129,10 @@ pub(crate) fn sample_ids( .map(I::from_usize) } -/// The golden-ratio increment of `SplitMix64`: `2^64 / phi`, odd and therefore coprime to the word, -/// so consecutive keys land maximally spread before mixing. +/// The odd golden-ratio increment used by `SplitMix64`. +/// +/// Its value approximates 2⁶⁴/φ, where φ = (1 + √5)/2. Oddness makes multiplication invertible +/// modulo 2⁶⁴. const SPLITMIX64_GAMMA: u64 = 0x9E37_79B9_7F4A_7C15; /// The first `SplitMix64` finalizer multiplier (D. Stafford's "mix 13" variant). @@ -137,25 +141,30 @@ const SPLITMIX64_MIX_1: u64 = 0xBF58_476D_1CE4_E5B9; /// The second `SplitMix64` finalizer multiplier (D. Stafford's "mix 13" variant). const SPLITMIX64_MIX_2: u64 = 0x94D0_49BB_1331_11EB; -/// Builds a generator keyed by a seed and two stream indexes. +/// Builds a reproducible non-cryptographic generator from a seed and stream indexes. +/// +/// Equal `(seed, key, stream)` inputs initialize equal generator states. Giving each work item +/// stable inputs and its own generator makes its draws independent of scheduling, provided its +/// sequence of calls is unchanged. This does not make subsequent floating-point reductions +/// independent of execution order. /// -/// The key mixes through the `SplitMix64` finalizer. The golden-ratio increment spreads `key` -/// across the word, and the two multiply-xorshift rounds avalanche every input bit into the output, -/// so generators with adjacent keys are statistically independent. Parallel work draws one -/// generator per `(seed, key, stream)` instead of sharing a sequence, and a seeded run reproduces -/// exactly at any thread count. +/// Odd multiplication and right-xorshift mixing permute 64-bit words. Varying one coordinate while +/// holding the others fixed changes the mixed seed. When more than one coordinate varies, this +/// 64-bit derivation can produce the same seed. It guarantees neither distinct nor statistically +/// independent streams. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore /// use rand::RngExt as _; /// +/// use crate::random::keyed_rng; +/// /// let mut draws = keyed_rng(42, 7, 0); /// let mut replay = keyed_rng(42, 7, 0); /// assert_eq!(draws.random::(), replay.random::()); -/// -/// let mut sibling = keyed_rng(42, 8, 0); -/// assert_ne!(draws.random::(), sibling.random::()); /// ``` #[must_use] pub(crate) fn keyed_rng(seed: u64, key: u64, stream: u64) -> impl Rng { @@ -166,28 +175,35 @@ pub(crate) fn keyed_rng(seed: u64, key: u64, stream: u64) -> impl Rng { Xoshiro256PlusPlus::seed_from_u64(mixed ^ (mixed >> 31)) } -/// Computes the sample size certifying a defect-rate bound. +/// Estimates the sample size for an all-pass defect-rate check. /// -/// The number of uniformly sampled items to check so that an all-pass result certifies the bound at -/// a confidence level. +/// Let p = `defect_rate` and c = `confidence`, both in (0, 1). Under independent uniform sampling, +/// a population with defect fraction at least p passes all n checks with probability at most (1 − +/// p)ⁿ. Requiring (1 − p)ⁿ ≤ 1 − c gives the real-arithmetic budget n = ⌈ln(1 − c)/ln(1 − p)⌉. +/// Therefore accepting only after all n checks pass bounds the probability of accepting a +/// population at or above the threshold by 1 − c. This is a repeated-sampling error bound, not a +/// posterior probability about the population after observing the sample. /// -/// Checking this many uniformly sampled items and finding all of them valid establishes, with -/// probability at least `confidence`, that the true fraction of invalid items is below -/// `defect_rate`. +/// For a fixed finite population, uniform sampling without replacement only lowers the all-pass +/// probability. The same budget is conservative when n fits the population. When n exceeds the +/// population, checking it in full directly settles an all-pass criterion. /// -/// If the true defect fraction were at least `defect_rate`, the probability that `n` independent -/// uniform samples all pass would be at most `(1 - defect_rate)^n`; requiring that this is at most -/// `1 - confidence` gives the smallest sufficient count, `n = ceil(ln(1 - confidence) / ln(1 - -/// defect_rate))`. Sampling without replacement from a finite population only lowers the all-pass -/// probability, so the bound stays valid and conservative there too. +/// # Warning /// -/// # Examples +/// The returned budget uses floating-point logarithms, division and ceiling, followed by a +/// saturating conversion to [`usize`]. Rounding near an integer boundary can change the minimal +/// sufficient count. Extreme ratios can underflow to zero or saturate to [`usize::MAX`]. The +/// returned integer alone is not a certified upper bound on the real-arithmetic budget. +/// +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore -/// // Verifying 688 uniformly sampled embeddings out of one million (for -/// // example, that each is L2-normalized) and finding all of them valid -/// // gives 99.9% confidence that fewer than 1% of the full set fails -/// // that check. +/// use crate::{math::OpenUnitFraction, random::acceptance_sample_size}; +/// +/// // If at least 1% of a population is defective, 688 independent uniform +/// // draws all pass with probability at most 0.99⁶⁸⁸ < 0.001. /// let defect_rate = OpenUnitFraction::new(0.01).expect("one percent is interior"); /// let confidence = OpenUnitFraction::new(0.999).expect("the confidence is interior"); /// assert_eq!(acceptance_sample_size(defect_rate, confidence), 688); @@ -197,10 +213,9 @@ pub(crate) fn acceptance_sample_size( defect_rate: OpenUnitFraction, confidence: OpenUnitFraction, ) -> usize { - // Both logarithms are strictly negative, so the ratio is positive and - // finite; `ceil` yields the smallest count whose all-pass probability - // drops to `1 - confidence` or below, up to f64 rounding at exact - // power boundaries. + // ln_1p preserves a small fraction's correction when `1.0 - fraction` would round to one. The + // negative logarithms have a positive real ratio, but the f64 division can underflow or + // overflow. let samples = (confidence.ln_complement() / defect_rate.ln_complement()).ceil(); #[expect( @@ -214,36 +229,45 @@ pub(crate) fn acceptance_sample_size( samples } -/// Computes the sample size estimating the population mean within `margin`. +/// Estimates a mean's sample size using a one-sided normal approximation. /// -/// The number of uniformly sampled items whose mean reaches the one-sided confidence level, given -/// the per-item standard deviation. +/// Let σ = `deviation` ≥ 0 be the per-item standard deviation, m = `margin` > 0 the tolerated +/// error, and c = `confidence` ∈ (1/2, 1). For n independent, identically distributed observations, +/// the sample mean has standard error σ/√n. With z = Φ⁻¹(c), the normal model requires zσ/√n ≤ m, +/// giving n = ⌈(zσ/m)²⌉. Here Φ is the standard normal cumulative distribution. This model is exact +/// for normal observations with known σ and approximate when justified by the central limit +/// theorem. /// -/// The estimate's standard error is `deviation / √n`, so `n = ceil((z · deviation / margin)^2)` -/// with `z` the standard normal quantile of `confidence` keeps the probability of a sampling error -/// beyond `margin` (in one direction) at most `1 - confidence`, by the central limit theorem. This -/// sizes aggregate-mean criteria. [`acceptance_sample_size`] sizes an all-pass criterion instead. -/// An acceptance budget guarantees nothing about a mean's error, so neither sizing rule substitutes -/// for the other. +/// A finite-variance assumption alone gives no finite-sample accuracy guarantee for the normal +/// approximation. For observations in [a, b], σ ≤ (b − a)/2 provides a distribution-free bound on +/// the deviation, but does not turn this sizing rule into a distribution-free confidence guarantee. +/// A pilot estimate of σ adds estimation uncertainty that this formula does not account for. Use +/// [`acceptance_sample_size`] for an all-pass criterion rather than a mean's error. /// -/// The deviation is the caller's to supply: bounded-per-item means admit the distribution-free -/// bound (half the range), and a pilot sample's measured deviation sizes the final sample without -/// baking a population constant into configuration (Stein's two-stage procedure). +/// # Warning /// -/// Every parameter carries its domain in the type, so every call has a defined sample size. A -/// ratio too large for `f64` saturates to [`usize::MAX`], which no corpus reaches. +/// [`normal_quantile`] and the budget arithmetic are approximate. Intermediate multiplication can +/// overflow even when the real ratio fits, and tiny ratios or their squares can underflow to zero. +/// The final conversion saturates to [`usize::MAX`]. The function returns zero for zero deviation +/// or confidence 1/2. A zero budget does not provide an observed mean. Confidence below 1/2 is +/// accepted by the type, but squaring the negative quantile does not implement the one-sided sizing +/// derivation above. /// -/// # Examples +/// # Example +/// +/// This in-crate example is ignored because the module is private. /// /// ```ignore -/// // Estimating a mean within one percentage point at 99% one-sided -/// // confidence, with a measured per-item deviation of 0.32. +/// use crate::{math::{DNonNegative, DPositive, OpenUnitFraction}, random::mean_sample_size}; +/// +/// // Planning a mean estimate within one percentage point at approximate +/// // 99% one-sided confidence, using a deviation estimate of 0.32. /// let deviation = DNonNegative::new(0.32).expect("the deviation is non-negative"); /// let margin = DPositive::new(0.01).expect("the margin is positive"); /// let confidence = OpenUnitFraction::new(0.99).expect("the confidence is interior"); /// assert_eq!(mean_sample_size(deviation, margin, confidence), 5542); /// -/// // A deviation of zero needs no sample at all. +/// // The formula returns zero when the supplied deviation is zero. /// assert_eq!(mean_sample_size(DNonNegative::ZERO, margin, confidence), 0); /// ``` #[must_use] @@ -266,19 +290,24 @@ pub(crate) fn mean_sample_size( samples } -/// Inverts the standard normal cumulative distribution. +/// Approximates a standard normal quantile. +/// +/// For p = `probability` ∈ (0, 1), the target is the finite z satisfying Φ(z) = p, where Φ is the +/// standard normal cumulative distribution. Acklam's rational approximation uses a central +/// polynomial ratio and a tail ratio after the transformation q = √(−2 ln p), with symmetry for the +/// upper tail. The returned value approximates z without a subsequent refinement step. /// -/// Returns the value `z` with `Phi(z) = probability`: the boundary a standard normal variable stays -/// below with exactly the given probability. Computed by Acklam's rational approximation, whose -/// relative error stays below `1.15e-9` over the normal range of the open interval and below -/// `1.8e-9` on subnormal probabilities - beyond any sampling design's sensitivity. +/// The upper-tail calculation uses [`f64::ln_1p`] to evaluate ln(1 − p). At the median p = 1/2, the +/// result is zero. Floating-point evaluation and the rational approximation do not guarantee exact +/// inversion or correct rounding. /// -/// The probability carries the open unit interval in its type, and every interior probability -/// has a finite quantile. +/// # Example /// -/// # Examples +/// This in-crate example is ignored because the module is private. /// /// ```ignore +/// use crate::{math::OpenUnitFraction, random::normal_quantile}; +/// /// let median = normal_quantile(OpenUnitFraction::new(0.5).expect("the median is interior")); /// assert!(median.abs() < 1e-9); /// @@ -339,6 +368,8 @@ pub(crate) fn normal_quantile(probability: OpenUnitFraction) -> f64 { } /// Evaluates a polynomial by Horner's rule, leading coefficient first. +/// +/// An empty coefficient array represents the zero polynomial. fn horner(coefficients: [f64; N], x: f64) -> f64 { coefficients .into_iter() diff --git a/libs/@local/graph/atlas/src/random/tests.rs b/libs/@local/graph/atlas/src/random/tests.rs index 80e68389cfc..be3e5a1143c 100644 --- a/libs/@local/graph/atlas/src/random/tests.rs +++ b/libs/@local/graph/atlas/src/random/tests.rs @@ -12,6 +12,7 @@ use super::{ }; use crate::math::{OpenUnitFraction, d_non_negative, d_positive, open_unit_fraction}; +/// Creates a reproducible generator for a test case. fn rng(seed: u64) -> Xoshiro256PlusPlus { Xoshiro256PlusPlus::seed_from_u64(seed) } @@ -38,8 +39,9 @@ fn uniform_below_residue_balance() { counts[usize::try_from(value).expect("a value below seven fits usize")] += 1; } - // Expected 10_000 per residue; ±10% is ~26 standard deviations, so a - // failure indicates bias rather than bad luck with the fixed seed. + // Modeling each draw as independent and uniform over the seven residues, the expected + // count per residue is 10_000, with standard deviation about 92.6 (√(70_000 · 1/7 · 6/7)). The + // ±10% margin is about 10.8 standard deviations under that model. for (residue, &count) in counts.iter().enumerate() { assert!( (9_000..=11_000).contains(&count), @@ -64,8 +66,6 @@ fn uniform_below_seed_determinism() { .collect(); assert_eq!(first, second); - // The stream must actually vary; a constant stream would make the - // equality above vacuous. assert!( first .array_windows::<2>() @@ -81,7 +81,7 @@ fn acceptance_sample_size_hand_checked() { acceptance_sample_size(open_unit_fraction!(0.01), open_unit_fraction!(0.95)), 299 ); - // The doc example uses one-in-a-hundred defects at 99.9% confidence. + // ln(0.001)/ln(0.99) ≈ 687.32, rounded up to 688. assert_eq!( acceptance_sample_size(open_unit_fraction!(0.01), open_unit_fraction!(0.999)), 688 @@ -92,11 +92,9 @@ fn acceptance_sample_size_hand_checked() { ); } -/// The returned count is sufficient and minimal. -/// -/// `n` all-pass samples push the false-acceptance probability to the target or below, and `n - 1` -/// samples do not. This is the function's entire contract, certified over the whole in-domain -/// parameter space. +// The real-arithmetic budget satisfies (1 − p)ⁿ ≤ 1 − c < (1 − p)ⁿ⁻¹. These sampled parameter +// ranges keep counts representable by i32 and away from extreme underflow or saturation. The +// probability comparison allows an absolute tolerance of 10⁻¹². #[property_test] fn acceptance_sample_size_sufficient_minimal( #[strategy = 1e-6_f64..0.5] defect_rate: f64, @@ -116,7 +114,6 @@ fn acceptance_sample_size_sufficient_minimal( } } -/// Stricter requirements never shrink the sample. #[property_test] fn acceptance_sample_size_monotone( #[strategy = 1e-5_f64..0.4] defect_rate: f64, @@ -132,7 +129,6 @@ fn acceptance_sample_size_monotone( prop_assert!(looser_defect <= base); } -/// Draws respect any bound, including awkward ones near overflow. #[property_test] fn uniform_below_in_range(#[strategy = any::()] seed: u64, #[strategy = 1_u64..] bound: u64) { let bound = NonZero::new(bound).expect("the strategy starts at one"); @@ -141,12 +137,9 @@ fn uniform_below_in_range(#[strategy = any::()] seed: u64, #[strategy = 1_u prop_assert!(value < bound.get()); } -/// The quantile matches tabulated standard normal values. -/// -/// The tabulated cases cover both rational-approximation regions. #[test] fn normal_quantile_tabulated() { - // Central region. + // common tabulated quantiles, including the upper tail at 0.99 for (probability, expected) in [ (0.5, 0.0), (0.75, 0.674_489_750_196_082), @@ -163,7 +156,7 @@ fn normal_quantile_tabulated() { ); } - // Tail regions (the approximation switches at 0.02425). + // additional tail values (the lower-tail approximation switches at 0.02425) for (probability, expected) in [ (0.999, 3.090_232_306_167_813), (0.000_1, -3.719_016_485_455_68), @@ -178,7 +171,6 @@ fn normal_quantile_tabulated() { } } -/// The quantile is antisymmetric about the median. #[test] fn normal_quantile_antisymmetry() { for probability in [0.001, 0.02425, 0.1, 0.3, 0.49] { @@ -192,12 +184,9 @@ fn normal_quantile_antisymmetry() { } } -/// The mean sample size follows the closed form and its monotonicity laws. -/// -/// Tighter margins and higher confidence grow the sample, smaller deviations shrink it. #[test] fn mean_sample_size_closed_form() { - // ceil((2.326348 · 0.32 / 0.012)^2) = ceil(3848.4) hand-checked. + // ⌈(2.326348 · 0.32 / 0.012)²⌉ = ⌈3848.46…⌉ = 3849. assert_eq!( mean_sample_size( d_non_negative!(0.32), @@ -214,7 +203,7 @@ fn mean_sample_size_closed_form() { ), 5542 ); - // A deviation of zero needs no sample. + // zero deviation gives a zero budget in the sizing formula assert_eq!( mean_sample_size( d_non_negative!(0.0), @@ -248,12 +237,11 @@ fn mean_sample_size_closed_form() { assert!(higher_confidence > base); assert!(smaller_deviation < base); - // Halving the margin exactly quadruples the requirement before - // rounding, so allow one count of ceiling slack. + // Halving the margin quadruples the real-arithmetic budget before ceiling. The integer budgets + // allow four counts of slack for rounding. assert!(tighter_margin >= base * 4 - 4 && tighter_margin <= base * 4 + 4); } -/// A sample carries the requested count of distinct indices, every one inside the population. #[test] fn sample_indices_vec_without_replacement() { let population = 1_000_000; @@ -263,8 +251,6 @@ fn sample_indices_vec_without_replacement() { assert_eq!(picked.len(), count); - // Distinctness is the without-replacement contract, so the set size must - // equal the sample length rather than merely bound it. let distinct: BTreeSet = picked.iter().collect(); assert_eq!(distinct.len(), count); assert!(distinct.iter().all(|&index| index < population)); @@ -272,10 +258,10 @@ fn sample_indices_vec_without_replacement() { hashql_core::id::newtype! { /// A position within the test population. + /// struct SampleId(u32) } -/// A sample carries the requested count of distinct ids, every one inside the population. #[test] fn sample_ids_without_replacement() { let population = IdSlice::::from_raw(&[(); 4096]); @@ -285,21 +271,16 @@ fn sample_ids_without_replacement() { assert_eq!(picked.len(), count); - // Distinctness is the without-replacement contract, so the set size must - // equal the sample length rather than merely bound it. let distinct: BTreeSet = picked.iter().copied().collect(); assert_eq!(distinct.len(), count); assert!(distinct.iter().all(|id| id.as_usize() < population.len())); } -/// One seed draws identical positions through the typed and untyped forms. #[test] fn sample_ids_stream_parity() { let population = IdSlice::::from_raw(&[(); 4096]); let count = 128; - // Seeded draws are pinned wherever a replay consumes them, so adopting - // the typed form must not move a single drawn value. let typed: Vec = sample_ids(rng(42), population, count) .map(SampleId::as_usize) .collect(); @@ -310,7 +291,6 @@ fn sample_ids_stream_parity() { assert_eq!(typed, raw); } -/// One key replays its own stream exactly. #[test] fn keyed_rng_replay() { let mut first = keyed_rng(42, 7, 0); @@ -324,8 +304,6 @@ fn keyed_rng_replay() { .collect(); assert_eq!(draws, replay); - // A constant stream would satisfy the equality above without replaying - // anything, so the draws must actually vary. assert!( draws .array_windows::<2>() @@ -333,9 +311,9 @@ fn keyed_rng_replay() { ); } -/// Each coordinate of `(seed, key, stream)` selects its own stream. #[test] fn keyed_rng_stream_separation() { + /// Draws the first 32 values of the generator keyed by `(seed, key, index)`. fn stream(seed: u64, key: u64, index: u64) -> Vec { let mut rng = keyed_rng(seed, key, index); core::iter::repeat_with(|| rng.random::()) diff --git a/libs/@local/graph/atlas/src/runs/mod.rs b/libs/@local/graph/atlas/src/runs/mod.rs index cdb6c91ab60..33f0891989a 100644 --- a/libs/@local/graph/atlas/src/runs/mod.rs +++ b/libs/@local/graph/atlas/src/runs/mod.rs @@ -1,40 +1,32 @@ -//! Compressed runs of items over dense key domains. +//! Packed variable-length lists indexed by dense keys. //! -//! [`Runs`] is the shared form of a recurring artifact and pipeline shape: many -//! short item lists, one per key of a dense domain, stored as two flat columns. -//! The items column holds every list back to back in key order, and the -//! fencepost column records where each list begins, so key `i`'s list is the -//! contiguous stretch `items[posts[i]..posts[i + 1]]`. Storage stays two -//! allocations at any key count. A run borrows as one slice, and either column -//! writes to an artifact region as it is. +//! [`Runs`] stores each key's list, called a run, in a shared items column. An offset column +//! locates the runs without allocating a separate buffer for each key. [`RunsView`] borrows the +//! same representation, including from mapped artifact regions. //! -//! # Vocabulary and invariants +//! # Representation //! -//! A run is one key's item list. The fencepost column carries one offset per -//! run plus a closing offset. It anchors at zero, never decreases, and closes -//! at the items column's length. Construction establishes these rules once, -//! so accessors index on them without rechecking. +//! The offset column, `posts`, contains one start offset per run and a final end offset. Key `i` +//! selects `items[posts[i]..posts[i + 1]]`. Equal adjacent offsets represent an empty run. //! -//! # Construction +//! For example, `posts = [0, 2, 2, 5]` and `items = [4, 7, 1, 6, 8]` represent the runs `[4, 7]`, +//! `[]` and `[1, 6, 8]` at keys 0, 1 and 2. //! -//! [`Runs::from_pairs`] counting-sorts unsorted `(key, item)` pairs into runs -//! in linear time. [`RunsBuilder`] appends whole runs when the producer -//! already visits keys in order. [`Runs::from_parts`] validates columns that -//! already exist. +//! Valid offsets start at zero, never decrease, and end at the items column's length. An empty key +//! domain still has the offset column `[0]`. Offsets are little-endian 64-bit values, ready for +//! writing to an artifact region. //! -//! # Value columns +//! # Construction //! -//! Per-item values live in parallel columns beside the items column, never -//! inside it. [`Runs::span`] returns a run's index range, and slicing a -//! parallel column with it yields the values of exactly that run. One -//! structure thereby serves any number of aligned columns, and each column -//! stays a plain array in memory and on disk. +//! - Use [`Runs::from_pairs`] to group `(key, item)` pairs supplied in any order. +//! - Use [`RunsBuilder`] to append whole runs in key order. +//! - Use [`Runs::from_parts`] or [`RunsView::from_parts`] to validate existing columns. //! -//! # Mapped artifacts +//! # Parallel columns //! -//! [`RunsView`] is the borrowed counterpart over a mapped artifact's regions: -//! the same fencepost law over columns a read-only file mapping owns, with -//! the fenceposts at their persisted little-endian width. +//! [`Runs::span`] returns a run's range of item positions. Apply that range to any parallel column +//! with one value per item, such as weights paired with neighbour IDs. The columns can remain +//! separate arrays in memory and on disk while sharing the same run boundaries. #[cfg(test)] mod tests; @@ -44,13 +36,10 @@ use core::{fmt, ops::Range}; use hashql_core::id::{Id, IdSlice, IdVec}; use zerocopy::{LE, U64}; -/// Why two columns are not a valid run structure. -/// -/// Each variant names one broken fencepost rule. The rules are structural, -/// and what the items mean stays the consumer's contract. +/// An invalid boundary in a packed list representation. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum RunsError { - /// The fencepost column is empty, so not even an empty key domain exists. + /// The offset column is empty, lacking the zero offset required even for no runs. Missing, /// The first fencepost is not zero. Anchor, @@ -86,11 +75,13 @@ impl fmt::Display for RunsError { impl core::error::Error for RunsError {} -/// Checks the fencepost law over one post column: anchored at zero, never -/// decreasing, closing at `items`. +/// Checks that the offsets partition exactly `items` elements into runs. +/// +/// # Errors +/// +/// Returns [`RunsError`] for invalid offsets, with the same check order as [`Runs::from_parts`]. fn validate_posts(posts: &[U64], items: u64) -> Result<(), RunsError> { - // `[first, ..]` rather than `[first, .., last]`: the lone anchoring post of an empty key - // domain is a valid column, and it is its own closing post. + // an empty key domain has one offset, serving as both the start and end. let &[first, ..] = posts else { return Err(RunsError::Missing); }; @@ -99,9 +90,7 @@ fn validate_posts(posts: &[U64], items: u64) -> Result<(), RunsError> { return Err(RunsError::Anchor); } - // Strict `>` only: equal neighbouring posts are exactly how an empty run is spelled. The - // window at `index` pairs a post with its successor, so the offending post - the one smaller - // than its predecessor - sits at `index + 1`. + // equal offsets allow empty runs. The decreasing offset is the window's second element. if let Some(index) = posts .array_windows::<2>() .position(|&[lhs, rhs]| lhs.get() > rhs.get()) @@ -117,20 +106,19 @@ fn validate_posts(posts: &[U64], items: u64) -> Result<(), RunsError> { Ok(()) } -/// Items grouped into per-key runs over one shared column. +/// Immutable per-key lists stored in shared offset and item columns. /// -/// `I` is the key domain, dense ids sharing the [`Id`] contract, and `T` is -/// the item element. Key `i` owns the `i`-th run. The structure is immutable -/// after construction, and every run borrows from the one items allocation. +/// `I` numbers the runs from zero under the [`Id`] contract. Every constructor establishes the +/// offset invariants described in the [module documentation](crate::runs). Lookup borrows a +/// contiguous slice without allocating or rescanning the offsets. /// -/// The fencepost column holds one offset per run plus a closing offset. It -/// anchors at zero, never decreases, and closes at the item count. Every -/// constructor establishes or validates these rules, so the accessors index -/// without rechecking them. +/// Keys supplied for lookup must convert losslessly to [`usize`]. Direct lookup also requires `I` +/// to represent the key's successor, which indexes the run's end offset. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct Runs { - /// Fenceposts: one offset per run plus a closing offset equal to - /// `items.len()`. The column anchors at zero and never decreases. + /// Fenceposts: one offset per run plus a closing offset equal to `items.len()`. + /// + /// The column anchors at zero and never decreases. posts: IdVec>, /// Every run's items, back to back in key order. items: Box<[T]>, @@ -140,22 +128,19 @@ impl Runs where I: Id, { - /// Returns the run count: the key domain's size. + /// Returns the number of keys, including keys with empty runs. #[inline] #[must_use] pub(crate) const fn runs(&self) -> usize { self.posts.len() - 1 } - /// Wraps existing fencepost and items columns as a validated structure. + /// Validates and takes ownership of existing offset and item columns. /// /// # Errors /// - /// Returns the first violated rule: [`RunsError::Missing`] when the - /// fencepost column is empty, [`RunsError::Anchor`] when the first - /// fencepost is not zero, [`RunsError::Order`] when a fencepost is - /// smaller than its predecessor, and [`RunsError::Close`] when the last - /// fencepost does not equal the items column's length. + /// Returns [`RunsError`] for invalid offsets, in the order [`Missing`](RunsError::Missing), + /// [`Anchor`](RunsError::Anchor), [`Order`](RunsError::Order), [`Close`](RunsError::Close). pub(crate) fn from_parts(posts: IdVec>, items: Vec) -> Result { validate_posts(posts.as_raw(), items.len() as u64)?; @@ -165,10 +150,9 @@ where }) } - /// Borrows the whole items column, every run back to back in key order. + /// Borrows all items in key order. /// - /// Its length is the total item count, which is also the length every - /// parallel value column matches. + /// Every parallel value column must match this slice's length and item order. #[inline] #[must_use] pub(crate) fn items(&self) -> &[T] { @@ -176,6 +160,11 @@ where } /// Iterates the runs in key order. + /// + /// # Panics + /// + /// Creating or advancing the iterator panics if `I` cannot represent a run index. An empty + /// domain also requires `I` to represent zero. pub(crate) fn iter(&self) -> impl ExactSizeIterator + '_ { self.posts .windows_enumerated() @@ -189,31 +178,31 @@ where }) } - /// Borrows the fencepost and items columns as raw slices. + /// Borrows the stored columns for serialization. /// - /// The fenceposts are usable directly as a file's pointer region and the - /// items as its index or payload region. Reading runs goes through - /// [`run`](Self::run) and [`span`](Self::span) instead. + /// The offsets retain their little-endian representation. Use [`run`](Self::run) or + /// [`span`](Self::span) to look up an individual key. #[must_use] pub(crate) fn as_raw_parts(&self) -> (&IdSlice>, &[T]) { (&self.posts, &self.items) } - /// Counting-sorts `(key, item)` pairs into runs. + /// Groups unordered pairs into runs while preserving each key's item order. + /// + /// Keys cover `0..runs`, including empty runs for keys absent from `pairs`. A clone of the + /// iterator supplies the per-key counts. The original iterator supplies the items and their + /// order within each run. Both iterations must yield the same number of items per key. /// - /// The pairs arrive in any order over a key domain of `runs` keys. A run - /// collects its key's pairs in arrival order, so a stream that ascends - /// within each key yields runs that ascend. Time and memory are linear in - /// the key and pair counts, over one counting pass, one prefix sum, and - /// one placement pass. + /// # Complexity /// - /// The constructor walks the iterator twice through its clone, once to - /// count and once to place. + /// For `n` pairs, construction takes O(`runs` + `n`) time and space. Counting sort uses a + /// counting pass, a prefix sum over keys, and a placement pass over the pairs. /// /// # Panics /// - /// This panics when a pair names a key at or beyond `runs`, and when the - /// cloned iterator does not repeat its sequence. + /// Panics if either iteration names a key outside `0..runs` or the per-key counts differ. The + /// key type must represent every index in `0..=runs` and also `1` for an empty domain. Keys + /// must meet [`Runs`]'s lossless-conversion requirement. pub(crate) fn from_pairs(runs: usize, pairs: impl Iterator + Clone) -> Self where T: Copy, @@ -225,15 +214,14 @@ where posts[key.plus(1)] += 1; } - // The anchor at id 0 stays zero. Every later post accumulates its predecessor. + // prefix sums turn per-key counts into end offsets, retaining zero as the first start. for index in posts.ids().skip(1) { let prev = posts[index.minus(1)]; posts[index] += prev; } - // The first pair's item seeds the whole buffer, and the placement - // overwrites every slot when the two passes agree, which the closing - // assertion checks, so no seeded value survives into the result. + // initialize with an actual item to avoid uninitialized storage. Placement replaces every + // slot when the per-key counts agree, which the final cursor comparison checks. let mut items: Vec = Vec::new(); let mut cursors = posts.prefix(I::from_usize(runs)).to_vec(); for (key, item) in pairs { @@ -262,15 +250,14 @@ where } } - /// Returns run `key`'s index range in the items column. + /// Returns the item positions belonging to `key`. /// - /// The range slices the columns riding beside the structure: a parallel - /// column keeps one value per item, and indexing it with this range - /// yields run `key`'s stretch of it. + /// The range also selects exactly this run's values from any aligned parallel column. /// /// # Panics /// - /// This panics when `key` is not below [`runs`](Self::runs). + /// Panics when `key` is not below [`runs`](Self::runs), or when `I` cannot represent its + /// successor. Keys must meet [`Runs`]'s lossless-conversion requirement. #[inline] #[must_use] pub(crate) fn span(&self, key: I) -> Range { @@ -282,11 +269,11 @@ where start..end } - /// Borrows run `key`: its items, contiguous in the shared column. + /// Borrows the items belonging to `key`. /// /// # Panics /// - /// This panics when `key` is not below [`runs`](Self::runs). + /// Panics under the key conditions of [`Self::span`], including an unrepresentable successor. #[inline] #[must_use] pub(crate) fn run(&self, key: I) -> &[T] { @@ -294,18 +281,20 @@ where } } -/// Runs borrowed from a mapped artifact's fencepost and items regions. +/// Per-key lists borrowed from existing offset and item columns. /// -/// The borrowed counterpart of [`Runs`]: the same fencepost law over columns -/// a read-only file mapping owns, with the fenceposts at their persisted -/// little-endian width. [`RunsView::from_parts`] validates the law where an -/// artifact opens, and [`RunsView::from_parts_unchecked`] re-borrows the same -/// regions afterwards, so an archive that owns its mapping serves runs -/// without storing a self-referential view. +/// The columns have the same representation and lookup requirements as [`Runs`]. Item borrows +/// retain the columns' lifetime and can outlive the view itself. +/// +/// For mapped artifacts, use [`from_parts`](Self::from_parts) to validate the regions when opening +/// the file. [`from_parts_unchecked`](Self::from_parts_unchecked) can reconstruct the view over +/// those unchanged regions for later reads. The mapping owner can then return run slices without +/// storing a self-referential view. #[derive(Debug, Clone, Copy)] pub(crate) struct RunsView<'map, I, T> { - /// Fenceposts: one offset per run plus a closing offset equal to - /// `items.len()`. The column anchors at zero and never decreases. + /// Fenceposts: one offset per run plus a closing offset equal to `items.len()`. + /// + /// The column anchors at zero and never decreases. posts: &'map IdSlice>, /// Every run's items, back to back in key order. items: &'map [T], @@ -315,13 +304,12 @@ impl<'map, I, T> RunsView<'map, I, T> where I: Id, { - /// Wraps mapped fencepost and items columns as a validated view. + /// Validates and borrows existing offset and item columns. /// /// # Errors /// - /// Returns the first violated rule, exactly as [`Runs::from_parts`] - /// reports it: [`RunsError::Missing`], [`RunsError::Anchor`], - /// [`RunsError::Order`], or [`RunsError::Close`]. + /// Returns [`RunsError`] for invalid offsets, with the same check order as + /// [`Runs::from_parts`]. pub(crate) fn from_parts(posts: &'map [U64], items: &'map [T]) -> Result { validate_posts(posts, items.len() as u64)?; @@ -331,12 +319,11 @@ where }) } - /// Re-wraps columns [`Self::from_parts`] validated when the artifact - /// opened. + /// Borrows offset and item columns whose run boundaries already passed validation. /// - /// The caller owns the proof that this exact pair passed validation. The - /// debug assertions catch the realistic misuse - one region's fenceposts - /// paired with another's items - through the anchor and close rules. + /// `posts` and `items` must be the exact pair that passed [`Self::from_parts`]. The validated + /// fenceposts and item count must remain unchanged. + // the fencepost invariant concerns correctness rather than memory safety #[must_use] pub(crate) fn from_parts_unchecked(posts: &'map [U64], items: &'map [T]) -> Self { debug_assert_eq!( @@ -356,21 +343,21 @@ where } } - /// Borrows the whole items column, every run back to back in key order. + /// Borrows all items in key order for the columns' lifetime. #[inline] #[must_use] pub(crate) const fn items(&self) -> &'map [T] { self.items } - /// Borrows run `key`: its items, contiguous in the mapped column. - /// - /// The borrow carries the mapping's lifetime rather than the view's, so a - /// run outlives the view value that served it. + /// Borrows the items belonging to `key` for the columns' lifetime. /// /// # Panics /// - /// This panics when `key` is not below the view's run count. + /// Panics when `key` is not below the view's run count, or when `I` cannot represent its + /// successor. Keys must meet [`Runs`]'s lossless-conversion requirement. Invalid columns + /// supplied through [`Self::from_parts_unchecked`] can also panic during fencepost conversion + /// or slicing. #[inline] #[must_use] pub(crate) fn run(&self, key: I) -> &'map [T] { @@ -383,6 +370,12 @@ where } /// Iterates the runs in key order. + /// + /// # Panics + /// + /// Creating or advancing the iterator panics if `I` cannot represent a run index. An empty + /// domain also requires `I` to represent zero. Invalid columns supplied through + /// [`Self::from_parts_unchecked`] can panic during fencepost conversion or slicing. pub(crate) fn iter(&self) -> impl ExactSizeIterator + '_ { let items = self.items; self.posts @@ -397,11 +390,11 @@ where } } -/// Builds [`Runs`] one whole run at a time, in key order. +/// An append-only builder for per-key lists. /// -/// Each [`push_run`](Self::push_run) call appends one run and returns its key. The fenceposts -/// follow from the pushes alone, so the finished structure satisfies the fencepost rules by -/// construction and [`finish`](Self::finish) validates nothing. +/// Each [`push_run`](Self::push_run) assigns the next key, starting at zero. Push an empty iterator +/// for a key with no items. [`finish`](Self::finish) makes the accumulated lists available as +/// immutable [`Runs`]. #[derive(Debug)] pub(crate) struct RunsBuilder { /// Fenceposts so far: seeded with the zero anchor, one push per run. @@ -414,9 +407,13 @@ impl RunsBuilder where I: Id, { - /// Creates a builder with room for `runs` runs over `items` items. + /// Reserves space for `runs` lists containing `items` items in total. + /// + /// The counts are capacity hints. Pushing beyond either grows the columns. + /// + /// # Panics /// - /// The counts are allocation hints. Pushing beyond either grows the columns. + /// Panics if `I` cannot represent zero. pub(crate) fn with_capacity(runs: usize, items: usize) -> Self { let mut posts = IdVec::with_capacity(runs + 1); posts.push(U64::new(0)); @@ -431,14 +428,14 @@ where /// /// # Panics /// - /// This panics when the finished run count leaves the key domain's encoding. + /// Panics if `I` cannot represent the new run count, which indexes the closing offset. pub(crate) fn push_run(&mut self, run: impl IntoIterator) -> I { self.items.extend(run); - // The pushed fencepost closes the run, so its id sits one past the run's own key. + // the end offset has the next key's index. self.posts.push(U64::new(self.items.len() as u64)).minus(1) } - /// Wraps the columns as the finished structure. + /// Makes the accumulated runs available for indexed lookup. #[must_use] pub(crate) fn finish(self) -> Runs { Runs { diff --git a/libs/@local/graph/atlas/src/runs/tests.rs b/libs/@local/graph/atlas/src/runs/tests.rs index b72cdb662f2..9060a4e0f82 100644 --- a/libs/@local/graph/atlas/src/runs/tests.rs +++ b/libs/@local/graph/atlas/src/runs/tests.rs @@ -4,6 +4,7 @@ use zerocopy::{LE, U64}; use super::{Runs, RunsBuilder, RunsError, RunsView}; use crate::identity::NodeRowId; +/// A node row id from a literal. fn node(row: u64) -> NodeRowId { NodeRowId::new(row) } @@ -13,6 +14,9 @@ fn le_posts(raw: &[u64]) -> IdVec> { IdVec::from_raw(raw.iter().copied().map(U64::new).collect()) } +/// Anchored, non-decreasing fenceposts closing at the item count build valid runs. +/// +/// `run`, `span` and `iter` return the expected slices, empty runs included. #[test] fn from_parts_accepts_a_valid_structure() { let runs = @@ -31,6 +35,7 @@ fn from_parts_accepts_a_valid_structure() { ); } +/// A lone zero fencepost builds an empty structure with no runs and no items. #[test] fn from_parts_accepts_an_empty_domain() { let runs = Runs::::from_parts(le_posts(&[0]), vec![]) @@ -41,6 +46,10 @@ fn from_parts_accepts_an_empty_domain() { assert_eq!(runs.iter().next(), None); } +/// `Runs::from_parts` fails with a distinct variant for each broken fencepost rule. +/// +/// `Missing` covers no fenceposts, `Anchor` a nonzero first post, and the order and closing +/// variants a decreasing post or one not ending at the item count. #[test] fn from_parts_rejects_each_broken_fencepost_rule() { assert_eq!( @@ -65,6 +74,9 @@ fn from_parts_rejects_each_broken_fencepost_rule() { ); } +/// `RunsView::from_parts` wraps mapped columns as the owned structure does. +/// +/// The view serves the same runs and the same iteration. #[test] fn view_wraps_mapped_columns() { let posts = [0_u64, 2, 2, 5].map(U64::::new); @@ -83,6 +95,9 @@ fn view_wraps_mapped_columns() { ); } +/// `RunsView::from_parts` rejects the same fencepost violations as the owned constructor. +/// +/// The `RunsError` variants match. #[test] fn view_from_parts_rejects_each_broken_fencepost_rule() { let items = [0_u32, 0, 0]; @@ -108,12 +123,13 @@ fn view_from_parts_rejects_each_broken_fencepost_rule() { ); } +/// A run borrowed through a view lives as long as the mapped columns, outliving the view value. #[test] fn view_runs_outlive_the_view_value() { let posts = [0_u64, 2, 2, 5].map(U64::::new); let items = [10_u32, 11, 20, 21, 22]; - // `run` borrows for the mapping's lifetime, so the run survives the view that served it. + // `run` borrows for the mapping's lifetime. The run survives the view that served it. let run = { let view = RunsView::::from_parts_unchecked(&posts, &items); view.run(node(2)) @@ -134,6 +150,7 @@ fn from_pairs_groups_pairs_by_key_and_keeps_arrival_order() { assert_eq!(runs.items(), [3, 4, 7, 9]); } +/// `from_pairs` over a zero-key domain and no pairs yields no runs and no items. #[test] fn from_pairs_accepts_an_empty_domain() { let runs = Runs::::from_pairs(0, core::iter::empty()); @@ -142,14 +159,17 @@ fn from_pairs_accepts_an_empty_domain() { assert!(runs.items().is_empty()); } +/// `from_pairs` panics with the documented message when a pair names a key beyond the domain. #[test] #[should_panic(expected = "every pair names a key inside the domain")] fn from_pairs_rejects_a_key_outside_the_domain() { let _runs = Runs::from_pairs(2, [(node(2), 1_u32)].into_iter()); } -/// An iterator whose clone yields one extra pair, breaking the repeatability -/// the counting sort relies on. +/// An iterator whose clone yields one extra pair for key zero. +/// +/// The counting pass and the placement pass therefore disagree on that key's count, which is the +/// agreement the counting sort needs. struct GrowingPairs { remaining: usize, } @@ -171,12 +191,19 @@ impl Clone for GrowingPairs { } } +/// `from_pairs` panics when the two passes disagree on a key's pair count. +/// +/// One pass sees two pairs for key zero where the other sees one. The panic carries the documented +/// message. #[test] #[should_panic(expected = "the placement pass replays the counting pass's pairs")] fn from_pairs_rejects_a_clone_that_repeats_a_different_sequence() { let _runs = Runs::from_pairs(1, GrowingPairs { remaining: 1 }); } +/// `RunsBuilder::push_run` returns successive keys. +/// +/// `finish` builds the same structure as a validated `from_parts`. #[test] fn builder_appends_runs_in_key_order_and_reports_each_key() { let mut builder = RunsBuilder::::with_capacity(3, 3); @@ -190,6 +217,7 @@ fn builder_appends_runs_in_key_order_and_reports_each_key() { assert_eq!(built, validated); } +/// `span` indexes a parallel per-item column to the same items a key's run covers. #[test] fn span_slices_a_parallel_column() { let pairs = [(node(1), 10_u32), (node(0), 20), (node(1), 30)]; diff --git a/libs/@local/graph/atlas/src/salt/adjacency/artifact.rs b/libs/@local/graph/atlas/src/salt/adjacency/artifact.rs index 82f86654cb6..f29d3832f96 100644 --- a/libs/@local/graph/atlas/src/salt/adjacency/artifact.rs +++ b/libs/@local/graph/atlas/src/salt/adjacency/artifact.rs @@ -1,5 +1,3 @@ -//! The writable builder, its matrix file, and the mapped reader that publish an adjacency. - use core::ops::Range; use hashql_core::id::{Id as _, bit_vec::DenseBitSet}; @@ -12,7 +10,7 @@ use crate::{ identity::{EdgeRowId, NodeRowId}, }; -/// An opened sparse matrix file does not hold a valid adjacency. +/// A failure while validating an adjacency's sparse-matrix representation. #[derive(Debug)] pub enum InvalidAdjacencyFile { /// The file fails the published adjacency shape. @@ -85,11 +83,17 @@ enum Width { /// A published adjacency opened over its mapped sparse matrix file. /// -/// Construction checks the list contract once, covering the structure-only element types and -/// compressed structure (fencepost coverage, strictly ascending runs, in-bound indices), paired -/// runs, the domain-bound column dimension, and every edge in exactly one slot per direction. An -/// open adjacency therefore serves only valid runs and consumers re-validate nothing. The regions -/// stay in the page cache under memory pressure and off the heap. +/// For CSR input, construction checks structure-only element types, compressed structure, paired +/// runs, the edge-domain bound, and exactly one slot per edge per direction. The lists borrow the +/// mapping without retaining a heap copy. Pages can be evicted and faulted back under memory +/// pressure. +/// +/// Supply CSR files, as produced by [`super::Adjacency`]. Construction does not check storage order +/// and does not establish the list contract for CSC input. +/// +/// # Panics +/// +/// Opening a CSC file can panic during validation. #[derive(Debug)] pub(crate) struct AdjacencyArchive { file: SprsFile, @@ -99,19 +103,30 @@ pub(crate) struct AdjacencyArchive { } impl AdjacencyArchive { - /// Opens the adjacency over its mapped sparse matrix file. + /// Validates a CSR file for mapped incident-edge lookups. + /// + /// The file must use CSR storage. Validation checks internal list consistency, without + /// comparing against an endpoint column. /// /// # Errors /// - /// Returns an error when the file violates the list contract. + /// Returns [`InvalidAdjacencyFile`] for invalid element types, compressed structure, or list + /// invariants. + /// + /// # Panics + /// + /// A CSC file can panic during run validation. + /// + /// # Complexity + /// + /// Validation takes O(N + E) time and O(E) temporary bits for N nodes and E edges in a CSR + /// file. Successful construction retains only the mapping and scalar metadata. #[tracing::instrument(skip_all)] pub(crate) fn new(file: SprsFile) -> Result { let (width, (nodes, edges)) = match file.header().index() { IndexVariant::U16 => (Width::U16, validate::(&file)?), IndexVariant::U32 => (Width::U32, validate::(&file)?), - // The writer emits unsigned widths only. The signed index - // types fail the element check inside, reported over the - // described types. + // signed widths fail the element check against u64. IndexVariant::U64 | IndexVariant::I16 | IndexVariant::I32 | IndexVariant::I64 => { (Width::U64, validate::(&file)?) } @@ -146,7 +161,7 @@ impl AdjacencyArchive { .expect("construction validated the element types") } - /// Returns the value array at its described width. + /// Borrows the edge-row index array at its stored width. fn values(&self) -> EdgeValues<'_> { let expect = "construction validated the element types"; match self.width { @@ -157,6 +172,11 @@ impl AdjacencyArchive { } /// Returns the run between fenceposts `start` and `end`. + /// + /// # Panics + /// + /// Panics if either fencepost index is outside the column or the selected fenceposts describe a + /// reversed slot range. fn run(&self, start: usize, end: usize) -> EdgeList<'_> { let fenceposts = self.fenceposts(); let from = usize::try_from(fenceposts[start]).expect("slots fit the address space"); @@ -176,16 +196,18 @@ impl AdjacencyArchive { Some(usize::try_from(2 * node.as_u64()).expect("resident node domains fit usize")) } - /// Returns the edge rows leaving `node`, strictly ascending, when the node row is in domain. + /// Returns the strictly ascending edge rows leaving `node`. + /// + /// Returns [`None`] when `node` is outside the node domain. #[must_use] pub(crate) fn outgoing(&self, node: NodeRowId) -> Option> { let posts = self.posts(node)?; Some(self.run(posts, posts + 1)) } - /// Returns the edge rows arriving at `node`. + /// Returns the strictly ascending edge rows arriving at `node`. /// - /// Strictly ascending, when the node row is in domain. + /// Returns [`None`] when `node` is outside the node domain. #[must_use] pub(crate) fn incoming(&self, node: NodeRowId) -> Option> { let posts = self.posts(node)?; @@ -193,15 +215,19 @@ impl AdjacencyArchive { } } -/// Validates the list contract over a mapped file at index type `I`. +/// Validates incident-edge runs in a CSR file at index type `I`. +/// +/// Returns the node and edge counts. The file must use CSR storage. The matrix check establishes +/// compressed structure, and the additional checks require paired runs, an edge-domain column +/// bound, a zero first fencepost, and one slot per edge per direction. +/// +/// # Errors +/// +/// Returns [`InvalidAdjacencyFile`] for invalid matrix or list structure, in check order. /// -/// Returns the node and edge row counts. The matrix view re-checks the compressed structure -/// (fencepost coverage, strictly ascending runs, indices below the column bound). The walk below -/// adds the rules that the format leaves unexpressed. +/// # Panics /// -/// - Runs pair two per node. -/// - The column dimension equals the edge-domain bound. -/// - Each edge occupies exactly one slot per direction. +/// A CSC file can panic when its pointer count or index domain differs from the CSR interpretation. fn validate(file: &SprsFile) -> Result<(u64, u64), InvalidAdjacencyFile> where I: SprsIndex + Into + Copy, @@ -229,12 +255,11 @@ where } let values = file.indices::().expect(expect); - // Each direction gets one bit set. Strict run order rules out - // duplicates within a run and the bit rules them out across runs, so - // 2E valid slots force every edge into exactly one slot of each - // direction. Every index lies below the column bound. That bound - // equals the edge count whenever entries exist, so the bit domain - // covers every walked value. + // A CSR matrix has one compressed run per row, covering every stored entry. For nonempty input + // the checked column bound equals E, giving 2E possible (direction, edge) pairs. The walk + // visits all 2E slots and rejects repeated pairs using one bitset per direction. Therefore + // every edge occupies exactly one slot in each direction. For E = 0 there are no indices to + // access in the empty bitsets. let capacity = usize::try_from(edges).expect("resident edge domains fit usize"); let mut seen = [ DenseBitSet::new_empty(capacity), @@ -245,7 +270,7 @@ where let start = usize::try_from(fenceposts[run]).expect("slots fit the address space"); let end = usize::try_from(fenceposts[run + 1]).expect("slots fit the address space"); - // Runs alternate outgoing (even) and incoming (odd). + // even runs are outgoing, odd runs incoming. let direction = &mut seen[run & 1]; for &value in &values[start..end] { @@ -262,9 +287,9 @@ where Ok((rows >> 1, edges)) } -/// A borrowed edge row id array, at either stored width. +/// A borrowed edge row id array at its stored unsigned width. /// -/// Value-level accessors widen to `u64`, so consumers stay width-agnostic. +/// Accessors widen each id to `u64`. #[derive(Debug, Copy, Clone)] enum EdgeValues<'map> { /// Two-byte edge row ids. @@ -306,7 +331,7 @@ impl EdgeValues<'_> { /// /// # Panics /// - /// This panics when `range` escapes [`len`](Self::len), like a slice. + /// Panics if `range.start > range.end` or `range.end` exceeds [`len`](Self::len). #[inline] #[must_use] const fn slice(&self, range: Range) -> Self { @@ -323,7 +348,7 @@ impl EdgeValues<'_> { } } -/// One node's edge rows, borrowed from the mapped value array. +/// One node's strictly ascending edge rows for a single direction. #[derive(Debug, Copy, Clone)] pub(crate) struct EdgeList<'map> { values: EdgeValues<'map>, diff --git a/libs/@local/graph/atlas/src/salt/adjacency/mod.rs b/libs/@local/graph/atlas/src/salt/adjacency/mod.rs index 83baadeda03..24752589373 100644 --- a/libs/@local/graph/atlas/src/salt/adjacency/mod.rs +++ b/libs/@local/graph/atlas/src/salt/adjacency/mod.rs @@ -1,16 +1,12 @@ -//! An incident-edge adjacency lists the edge rows touching each node row. +//! Incident-edge lookup with separate outgoing and incoming runs. //! -//! [`Adjacency`] is the serving contract's topology artifact. For every node row it records the -//! edge rows leaving it and the edge rows arriving at it as two adjacent runs of one shared entry -//! array. +//! [`Adjacency`] records the edge rows leaving each node and arriving at it as adjacent runs of one +//! shared entry array. Naming edge rows preserves parallel edges between the same node pair. The +//! adjacency depends only on the endpoint column and node domain, allowing attribute columns to +//! change independently. //! -//! Every entry names an edge row rather than a node pair. The same node pair admits more than one -//! edge row, and attributes resolve through edge-row-indexed columns. Naming the edge keeps -//! parallel edges distinct and keeps the artifact stable, and the adjacency never re-publishes when -//! an attribute column changes. -//! -//! A node pair joined both ways draws the incidence picture the file compresses, with edge row `0` -//! the `0 → 1` edge and edge row `1` the `1 → 0` edge: +//! A node pair joined both ways has the following incidence matrix, with edge row `0` the `0 → 1` +//! edge and edge row `1` the `1 → 0` edge: //! //! ```text //! edge 0 edge 1 @@ -23,27 +19,28 @@ //! Each matrix row stores its `x` marks as one ascending run of edge row ids, and each edge column //! holds exactly two marks: one outgoing at its source, one incoming at its target. //! -//! The artifact derives from the endpoint column in one counting pass and publishes as one -//! structure-only [`crate::file::sprs`] matrix: `2N` compressed rows over the fencepost column, -//! edge row ids as the indices, and [`unit`](crate::file::sprs::ValueTag::Unit) values, so no value -//! bytes exist on disk. +//! Construction fills the runs in edge-row order after counting degrees and computing prefix sums. +//! The artifact writes as one structure-only [`crate::file::sprs`] CSR matrix: `2N` compressed rows +//! for `N` nodes, edge row ids as indices, and [`unit`](crate::file::sprs::ValueTag::Unit) values. +//! No value bytes exist on disk. //! -//! [`AdjacencyArchive`] reopens the file over a whole-file mapping and validates the list -//! invariants once, so lookups read from the page cache without holding the lists on the heap. +//! [`AdjacencyArchive`] borrows lists from a whole-file mapping. For CSR input it validates the +//! list invariants at construction, using temporary per-direction bitsets. Lookups borrow the +//! mapped runs without allocating. //! //! # List contract //! -//! - Matrix row `2i` is node row `i`'s outgoing run and row `2i + 1` its incoming run, so one -//! fencepost column serves both directions and the whole incident slice is contiguous for free. +//! - Matrix row `2i` is node row `i`'s outgoing run and row `2i + 1` its incoming run. One +//! fencepost column serves both directions. The incident slice is contiguous. //! - Every edge row occupies exactly one outgoing slot (at its source) and one incoming slot (at -//! its target). A self-loop occupies both slots of its one endpoint, so a consumer merging the -//! directions has to dedupe. +//! its target). A self-loop occupies both slots of its one endpoint. Merging the directions +//! requires deduplication. //! - Within each run the edge row ids are strictly ascending: runs are binary-searchable, and //! filtered merges walk them linearly. //! - Zero-degree nodes hold two empty runs. //! - The column dimension records the edge-domain bound `max(E, 1)`. The shape encoding terminates -//! on zero extents, so an edgeless adjacency records the smallest bound and zero entries, and the -//! edge count reads from the entry count alone. +//! on zero extents. An edgeless adjacency records the smallest bound and zero entries. The edge +//! count reads from the entry count alone. mod artifact; @@ -70,8 +67,13 @@ use crate::{ /// Places edge row `edge` into its source's outgoing and its target's incoming slot. /// -/// `cursors` holds each run's next free slot; a placement advances its run's cursor, so filling in -/// edge-row order lands ascending edge rows ascending in place. +/// `cursors` holds each run's next free slot. Filling slots in edge-row order keeps each run +/// ascending. +/// +/// # Panics +/// +/// Panics if an endpoint's computed run is outside `cursors`, or its next slot is outside `values` +/// or the address space. fn insert_edge( cursors: &mut [u64], values: &mut [I], @@ -93,9 +95,8 @@ fn insert_edge( /// A unit-value array carried as its length alone. /// -/// Sparse-matrix storage wants one value slot per structural entry, and a structure-only matrix's -/// entries are units. A unit occupies no bytes, so the length is the whole value: `n` of them are -/// recoverable from `n`, and holding the count costs what holding the array would have cost. +/// Sparse-matrix storage requires one value slot per structural entry. A unit occupies no bytes, +/// and `n` units are recoverable from the length `n`. #[derive(Debug, Copy, Clone, PartialEq, Eq)] struct UnitSlice { length: usize, @@ -105,9 +106,11 @@ impl Deref for UnitSlice { type Target = [()]; fn deref(&self) -> &[()] { - // SAFETY: `()` is zero-sized, so the slice covers no bytes at any length. The pointer - // to `self` is non-null and trivially aligned for `()`, zero bytes are valid for reads - // at any address, and no element count of a zero-sized type overflows `isize` in bytes. + // SAFETY: a slice of `()` occupies zero bytes for every element count, and `()` has + // alignment one and no initialization bytes. The pointer comes from the shared borrow of + // `self`, is non-null and aligned, and the returned slice borrows no longer than `self`. + // Its byte range is empty and cannot exceed `isize::MAX` or wrap the address space. + // Therefore `from_raw_parts` may construct this shared unit slice. unsafe { core::slice::from_raw_parts(core::ptr::from_ref(self).cast::<()>(), self.length) } } } @@ -128,30 +131,39 @@ enum AdjacencyGraph { /// The incident-edge adjacency of one generation, in writable form. /// -/// Construction orders every run; the fencepost and value columns are exactly the file's pointer -/// and index regions. +/// Construction orders every run. The fencepost and edge-row columns become the file's pointer and +/// index regions. Writing uses the narrowest unsigned index width covering `max(E, 1)`, where `E` +/// is the edge count. +/// +/// Writing a zero-node adjacency returns [`WriteSprsError::ZeroDimension`]. A nonempty node domain +/// with no edges has a file representation. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct Adjacency(AdjacencyGraph); impl Adjacency { /// Builds the adjacency over the endpoint column. /// - /// `endpoints[e]` is edge row `e`'s `[source, target]` node rows; `rows` is the node-row domain - /// they index into. Time and memory are `O(N + E)` over one counting pass, one prefix sum, and - /// one fill in edge-row order, which is what makes every run strictly ascending by - /// construction. + /// `endpoints[e]` is edge row `e`'s `[source, target]` node rows. Every endpoint must lie in + /// the `rows` node domain, and the `2 · rows + 1` fenceposts must fit an addressable + /// allocation. The result has one outgoing and one incoming slot per edge, including + /// self-loops. + /// + /// # Complexity + /// + /// Time and memory are O(N + E), for N node rows and E edges. A counting pass and prefix sum + /// delimit the runs. Filling each run in ascending edge-row order establishes its strict + /// ordering. /// /// # Panics /// - /// This panics when an endpoint lies outside the `rows` domain, which the dataset row contract - /// excludes. + /// Panics if the run-column allocation exceeds the address space or a computed endpoint slot + /// lies outside it. #[must_use] pub(crate) fn build(rows: usize, endpoints: &[[NodeRowId; 2]]) -> Self { let mut fenceposts = vec![0_u64; 2 * rows + 1]; - // Degrees first: slot 2i + 1 counts node i's outgoing edges and - // slot 2i + 2 its incoming, so the prefix sum below turns the - // counts into the run fenceposts directly. + // slot 2i + 1 counts node i's outgoing edges and slot 2i + 2 its incoming edges. Placing + // degrees one slot after each run's start makes the prefix sum produce its fenceposts. for &[source, target] in endpoints { let source = source.as_usize(); let target = target.as_usize(); @@ -162,13 +174,13 @@ impl Adjacency { fenceposts[position] += fenceposts[position - 1]; } - // Fill in edge-row order: each run's cursor starts at its - // fencepost, and ascending edge rows land ascending in place. + // each run's cursor starts at its fencepost. The fill visits edge rows in ascending order + // and appends within each run. let mut cursors = fenceposts[..fenceposts.len() - 1].to_vec(); - // The narrowest covering width shrinks the value column and the - // on-disk index region alike; `bound` stands in for the column - // dimension so an edgeless adjacency keeps a nonzero domain. + // sprs requires the column dimension itself to fit the index type. The narrowest covering + // width shrinks both the resident edge-row column and its file region. A nonzero bound + // preserves the shape of an edgeless adjacency. let bound = endpoints.len().max(1); if u16::try_from(bound).is_ok() { Self(AdjacencyGraph::U16(assemble( @@ -202,16 +214,25 @@ impl Adjacency { AdjacencyGraph::U32(graph) => graph.rows(), AdjacencyGraph::U64(graph) => graph.rows(), }; - // The list contract stores two runs per node, so the halving is exact. + // the list contract stores exactly two runs per node. runs.div_euclid(2) } /// Returns a node's incident-edge degree: its outgoing plus incoming slots. /// - /// A self-loop counts twice, once per direction, matching the slot contract above. Returns - /// [`None`] when the row lies outside the node domain. + /// A self-loop counts twice, once per direction. Returns [`None`] when the platform-sized row + /// index lies outside the node domain. + /// + /// # Warning + /// + /// On targets narrower than 64 bits, row conversion retains only the low `usize::BITS` bits. An + /// out-of-domain row can then alias an in-domain node. #[must_use] pub(crate) fn degree(&self, node: NodeRowId) -> Option { + /// Counts the slots incident to `node` in `graph`. + /// + /// Sums the outgoing and incoming slots, and returns [`None`] when the node's slot pair + /// lies outside `graph`. fn incident(graph: &AdjacencySparseGraph, node: NodeRowId) -> Option where I: SpIndex, @@ -233,10 +254,16 @@ impl Adjacency { } } -/// Fills the value column at index width `I` and assembles the CSR adjacency. +/// Fills the edge-row column at index width `I` and assembles the CSR adjacency. +/// +/// `fenceposts` must be the finished degree prefix sums over the `2N` runs of `endpoints`, starting +/// at zero. `cursors` must start at each run's fencepost. `bound` must equal `max(endpoints.len(), +/// 1)` and fit `I`. The matrix's row dimension is the run count, not the node count. /// -/// `fenceposts` are the finished prefix sums over the `2N` runs; `cursors` start at each run's -/// fencepost. The matrix's row dimension is the run count, not the node count. +/// # Panics +/// +/// Panics if `fenceposts` is empty, an edge row cannot fit `I`, the entry allocation exceeds the +/// address space, or [`insert_edge`] encounters an out-of-range run or slot. fn assemble( bound: usize, fenceposts: Vec, @@ -257,10 +284,14 @@ where let runs = fenceposts.len() - 1; let length = values.len(); - // SAFETY: the counting build establishes the compressed structure: the fenceposts are a - // prefix sum starting at zero and ending at the slot count, one entry past the run count, - // the fill placed ascending edge rows ascending within each run, every value lies below - // `bound`, and the unit storage length equals the value count. + // SAFETY: sprs requires matching entry/value lengths, monotone representable pointers, strictly + // ascending bounded indices per run, and representable dimensions. `build` supplies zero-based + // degree prefix sums ending at `2E`, with `runs + 1` pointers. The successful non-ZST + // allocations bound pointers by `isize::MAX`, and their lengths fit `u64`. Each edge is + // appended once to each endpoint direction in ascending order, including separate self-loop + // slots. The selected index width covers `bound`, every edge index is below it, and `UnitSlice` + // has exactly the index count. Therefore these columns satisfy `new_unchecked`'s + // compressed-structure contract. unsafe { CsMatBase::new_unchecked( sprs::CompressedStorage::CSR, @@ -277,17 +308,6 @@ impl WriteAs for Adjacency {} impl WriteInto for Adjacency { type Error = WriteSprsError; - /// Writes the adjacency as a structure-only sparse matrix file. - /// - /// At the narrowest index width covering the edge count. - /// - /// Returns the SHA-256 of the written bytes, which is the identity the repository records for - /// the published file. - /// - /// # Errors - /// - /// Returns an error when the underlying writer fails, or when the adjacency spans no node rows. - /// The corpus contract places at least one node, and an empty row domain has no on-disk form. fn write_into(&self, write: impl io::Write) -> Result { let mut writer = Writer { accumulator: Sha256::new(), diff --git a/libs/@local/graph/atlas/src/salt/adjacency/tests.rs b/libs/@local/graph/atlas/src/salt/adjacency/tests.rs index 4abe1e4a7ce..4f9cdfd46d7 100644 --- a/libs/@local/graph/atlas/src/salt/adjacency/tests.rs +++ b/libs/@local/graph/atlas/src/salt/adjacency/tests.rs @@ -15,6 +15,11 @@ use crate::{ integrity::{Sha256, Writer}, }; +/// Recreates a per-process scratch directory for the named case. +/// +/// # Panics +/// +/// Panics if the system temporary path is not UTF-8 or directory creation fails. fn scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -37,8 +42,14 @@ const ENDPOINTS: [[NodeRowId; 2]; 4] = [ [NodeRowId::new(3), NodeRowId::new(3)], [NodeRowId::new(0), NodeRowId::new(1)], ]; +/// Node row count of the endpoint fixture. const ROWS: usize = 5; +/// Writes `adjacency` to `name` under `dir` and reopens it as a validated mapped archive. +/// +/// # Panics +/// +/// Panics if file creation, writing, opening, or adjacency validation fails. fn mapped(dir: &Utf8PathBuf, name: &str, adjacency: &Adjacency) -> AdjacencyArchive { let path = dir.join(name); let mut file = fs::File::create(&path).expect("the fixture file should create"); @@ -51,6 +62,11 @@ fn mapped(dir: &Utf8PathBuf, name: &str, adjacency: &Adjacency) -> AdjacencyArch .expect("the fixture adjacency should validate") } +/// Collects an in-domain edge list as row numbers. +/// +/// # Panics +/// +/// Panics if `edges` is [`None`]. fn list(edges: Option>) -> Vec { edges .expect("the queried node row is in domain") @@ -68,22 +84,19 @@ fn build_matches_the_hand_computed_lists() { assert_eq!(mapped.rows(), ROWS as u64); assert_eq!(mapped.edges(), ENDPOINTS.len() as u64); - // Outgoing lists per node row, ascending: the parallel pair leaves - // node 0 as edges 0 and 3. + // the parallel pair leaves node 0 as edges 0 and 3. assert_eq!(list(mapped.outgoing(NodeRowId::new(0))), [0, 3]); assert_eq!(list(mapped.outgoing(NodeRowId::new(1))), [] as [u64; 0]); assert_eq!(list(mapped.outgoing(NodeRowId::new(2))), [1]); assert_eq!(list(mapped.outgoing(NodeRowId::new(3))), [2]); assert_eq!(list(mapped.outgoing(NodeRowId::new(4))), [] as [u64; 0]); - // Incoming lists mirror the targets; the self-loop arrives at its - // own node. + // the self-loop occupies both directions at node 3. assert_eq!(list(mapped.incoming(NodeRowId::new(1))), [0, 3]); assert_eq!(list(mapped.incoming(NodeRowId::new(3))), [1, 2]); assert_eq!(list(mapped.incoming(NodeRowId::new(4))), [] as [u64; 0]); - // Out-of-domain rows answer None. assert!(mapped.outgoing(NodeRowId::new(5)).is_none()); assert!(mapped.incoming(NodeRowId::new(5)).is_none()); } @@ -101,6 +114,14 @@ fn edgeless_corpus_builds_empty_lists() { } /// Writes a hand-built structure-only matrix and opens it as a mapped adjacency. +/// +/// # Errors +/// +/// Returns [`InvalidAdjacencyFile`] if the matrix violates the adjacency list contract. +/// +/// # Panics +/// +/// Panics if the compressed matrix is invalid or file creation, writing, or mapping fails. fn open_structure( dir: &Utf8PathBuf, name: &str, @@ -125,20 +146,16 @@ fn open_structure( fn violated_list_invariants_are_rejected() { let dir = scratch("violations"); - // An odd row dimension pairs no runs. assert_matches!( open_structure(&dir, "odd.sprs", (1, 1), &[0, 0], &[]), Err(InvalidAdjacencyFile::OddRows { rows: 1 }), ); - // An odd entry count holds no two slots per edge. assert_matches!( open_structure(&dir, "slots.sprs", (2, 1), &[0, 1, 1], &[0]), Err(InvalidAdjacencyFile::Slots { entries: 1 }), ); - // A column dimension beyond the edge-domain bound is not the - // canonical artifact. assert_matches!( open_structure(&dir, "bound.sprs", (2, 5), &[0, 1, 2], &[0, 0]), Err(InvalidAdjacencyFile::Bound { @@ -147,14 +164,12 @@ fn violated_list_invariants_are_rejected() { }), ); - // An edge in two slots of one direction: both outgoing runs hold - // edge 0. + // both outgoing runs hold edge 0, although each run alone is valid. assert_matches!( open_structure(&dir, "duplicate.sprs", (4, 1), &[0, 1, 1, 2, 2], &[0, 0]), Err(InvalidAdjacencyFile::Duplicate { edge }) if edge == EdgeRowId::MIN ); - // A valued matrix is not the structure-only artifact. let path = dir.join("valued.sprs"); let mut writer = Writer { accumulator: Sha256::new(), @@ -169,10 +184,10 @@ fn violated_list_invariants_are_rejected() { ); } -/// A fencepost column anchored past zero passes the compressed-row check. +/// A nonzero first fencepost preserves sprs' relative-offset structure. /// -/// The entry count reads relative to the first post - but leaves leading slots no run owns, which -/// the adjacency rejects. +/// Adding the same offset to every post preserves every relative run and the entry count. Adjacency +/// lookups use raw posts as array indices and therefore require the first post to be zero. #[test] #[expect( clippy::little_endian_bytes, @@ -189,10 +204,7 @@ fn shifted_fencepost_column_is_rejected() { .expect("the adjacency should write"); drop(file); - // Shift every fencepost up by one. Monotonicity and the relative - // entry count survive the shift, but the zero anchor does not. The - // posts occupy the page-aligned region behind the header, eight - // bytes each. + // the eight-byte posts begin immediately after the 4096-byte header. let mut bytes = fs::read(&path).expect("the fixture file should read"); let posts = 2 * 2 + 1; for post in 0..posts { @@ -215,8 +227,7 @@ fn shifted_fencepost_column_is_rejected() { #[test] fn build_is_independent_of_the_endpoint_values_within_a_row() { - // A permuted edge order is a different corpus (edge rows are - // positional), but every list still comes out strictly ascending. + // permuting endpoints changes edge row identities. Runs still sort by the new row order. let permuted: [[NodeRowId; 2]; 4] = [ENDPOINTS[3], ENDPOINTS[1], ENDPOINTS[0], ENDPOINTS[2]]; let dir = scratch("permuted"); let adjacency = Adjacency::build(ROWS, &permuted); @@ -227,9 +238,6 @@ fn build_is_independent_of_the_endpoint_values_within_a_row() { assert_eq!(list(mapped.incoming(NodeRowId::new(3))), [1, 3]); } -/// Wide indices read back through the same accessors. -/// -/// A hand-built eight-byte matrix validates and serves runs like the narrow files the writer emits. #[test] fn wide_indices_read_back() { let dir = scratch("wide"); @@ -242,10 +250,6 @@ fn wide_indices_read_back() { assert_eq!(list(mapped.incoming(NodeRowId::new(0))), [0]); } -/// The tests the `miri` nextest profile selects. -/// -/// The test here writes an adjacency into memory and conjures the unit region back from the bytes. -/// The profile selects by module path, so moving a test in or out of this module is the whole edit. mod miri { use crate::{file::WriteInto as _, identity::NodeRowId, salt::adjacency::Adjacency}; diff --git a/libs/@local/graph/atlas/src/salt/embedding/external/mod.rs b/libs/@local/graph/atlas/src/salt/embedding/external/mod.rs index a49fa5996f5..cf16423bb75 100644 --- a/libs/@local/graph/atlas/src/salt/embedding/external/mod.rs +++ b/libs/@local/graph/atlas/src/salt/embedding/external/mod.rs @@ -1,24 +1,20 @@ //! Card embedding through an external provider. //! -//! [`ExternalEmbeddingProvider`] makes any [`EmbeddingGenerator`] usable as a [`CardEmbedder`]. The -//! generator family already pins the vector type, the error type, and the input-order contract, so -//! backends differ only in data. The [`EmbeddingContract`] names the configuration the fingerprint -//! commits to, and the [`RequestLimits`] set the ceilings the proxy packs requests under. One -//! request never exceeds the document ceiling or the summed token ceiling, and token counts are -//! exact `cl100k_base` counts, the encoding of the embedding models this crate targets. The -//! provider additionally gates admission on a byte estimate of the token count (UTF-8 bytes divided -//! by four, measured bit-exact against its rejections), so requests stay under the token ceiling in -//! both accountings. +//! [`ExternalEmbeddingProvider`] adapts an [`EmbeddingGenerator`] to [`CardEmbedder`] with +//! sequential request batches and completed-batch progress. [`EmbeddingContract`] identifies the +//! declared vector configuration, while [`RequestLimits`] controls batch size. //! -//! The request boundary is also the workload's only observable progress, so the proxy carries the -//! run's [`Progress`] observer and reports each request it completes against the workload the -//! caller handed it. Nothing above it observes those boundaries, because a caller hands over the -//! whole workload in one call. +//! For a batch of texts, let tᵢ be the `cl100k_base` token count and bᵢ its UTF-8 byte length. +//! Admission requires both Σtᵢ ≤ L and ⌈Σbᵢ/4⌉ ≤ L, where L is the token limit, together with the +//! document limit. The byte estimate supplements the tokenizer count. These are local sizing rules, +//! and a provider can impose additional restrictions. //! -//! Because the workload arrives whole, the first request is also the first proof that the provider -//! is reachable under the configured credentials, by which point a run has already read the store -//! and assembled its cards. [`ExternalEmbeddingProvider::preflight`] buys that proof up front, for -//! one text. +//! The adapter calls [`Progress::embedding_batch`] after each batch returns the expected number of +//! canonical-width vectors. The report counts completed texts against the whole workload. A +//! generator can retry internally, and batch reports do not count HTTP attempts. +//! +//! [`ExternalEmbeddingProvider::preflight`] checks one short text before committing to a workload. +//! Success establishes that this call returned one canonical-width vector. use core::{error::Error, fmt, iter::Peekable, num::NonZero, ops::ControlFlow}; @@ -39,10 +35,10 @@ mod tests; /// The configuration an [`EmbedderFingerprint`] commits to. /// -/// The fields name everything that determines the vector a text embeds to, as the caller configured -/// the generator. The adapter adds the dimension it enforces. Stating a contract that differs from -/// the generator's actual configuration poisons cross-generation reuse, so construct it beside the -/// generator, from the same values. +/// The fields must describe the generator's actual configuration. The adapter adds +/// [`CANONICAL_DIMENSIONS`] to the fingerprint preimage. Use a model identity that changes when the +/// vector-producing contract changes. A declared model name alone cannot detect provider-side model +/// updates. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct EmbeddingContract<'text> { /// The provider organization, e.g. `openai`. @@ -58,8 +54,9 @@ pub(crate) struct EmbeddingContract<'text> { impl EmbeddingContract<'_> { /// Returns the fingerprint of this contract. /// - /// Every field is length-prefixed in the preimage, so fingerprints distinguish contracts that - /// concatenate to equal bytes. + /// Each text field has a little-endian byte-length prefix. The preimage therefore distinguishes + /// field boundaries even when raw field concatenations are equal. SHA-256 hashes that + /// domain-separated preimage together with the canonical dimension. #[expect( clippy::little_endian_bytes, reason = "the preimage is pinned to canonical little-endian length prefixes on every \ @@ -86,48 +83,48 @@ impl EmbeddingContract<'_> { } } +/// The default maximum of 2,048 texts per batch. const DEFAULT_DOCUMENT_LIMIT: NonZero = const { NonZero::new(2_048).unwrap() }; +/// The default ceiling of 300,000 tokens under each local accounting. const DEFAULT_TOKEN_LIMIT: NonZero = const { NonZero::new(300_000).unwrap() }; -/// The text a preflight request carries. -/// -/// Its content is immaterial, because the only question under test is whether the provider answers -/// at all. The text is one short word, the cheapest request the endpoint accepts. +/// A short input for checking the generator response count and vector width. const PREFLIGHT_TEXT: &str = "preflight"; -/// Ceilings one provider request must stay under. +/// Document and token ceilings for each embedding batch. /// -/// The defaults are the OpenAI embeddings API's published per-request ceilings. +/// Custom limits must keep accumulated token and byte counts, including the next candidate text, +/// representable as `usize`. The default token limit leaves room for these additions. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct RequestLimits { - /// Maximum texts per request. + /// Maximum texts per batch, 2,048 by default. pub documents: NonZero = DEFAULT_DOCUMENT_LIMIT, - /// Maximum summed tokens per request. + /// Maximum batch cost under each token accounting, 300,000 by default. /// - /// Held in both provider accountings: the exact `cl100k_base` count and the admission gate's - /// byte estimate (UTF-8 bytes divided by four). + /// Applies separately to the sum of `cl100k_base` token counts and the total UTF-8 byte length divided by four, rounded up. pub tokens: NonZero = DEFAULT_TOKEN_LIMIT, } -/// The running cost of the request under assembly, in both provider accountings. +/// Accumulated tokenizer and byte counts for the batch under assembly. #[derive(Default)] struct RequestCost { /// Exact `cl100k_base` tokens. tokens: usize, /// UTF-8 bytes. /// - /// The admission gate estimates tokens as bytes divided by four. + /// The admission estimate rounds the batch's total bytes divided by four up to an integer. bytes: usize, } -/// A [`CardEmbedder`] over an external [`EmbeddingGenerator`]. +/// A card embedder with sequential batching and completed-batch progress. /// -/// The proxy sizes the requests. It splits a workload into requests that respect both -/// [`RequestLimits`] ceilings, measuring each text's cost by its exact `cl100k_base` token count. -/// It re-validates returned vectors to the canonical width and hands them back in input order. +/// Under [`RequestLimits`], the workload splits into batches with validated response counts and +/// vector widths. Output retains the order promised by [`EmbeddingGenerator`], and each completed +/// batch reports to the supplied observer. [`super::embed_cards`] checks vector finiteness +/// separately. /// -/// Sizing the requests here makes the proxy the only place a workload's advance is visible, so it -/// reports every completed request to the run's observer. +/// The adapter collects the input text references and all output vectors before returning. A +/// failure returns no partial vector collection, even if earlier batches completed. #[derive(Debug)] pub(crate) struct ExternalEmbeddingProvider { generator: G, @@ -137,7 +134,11 @@ pub(crate) struct ExternalEmbeddingProvider { } impl ExternalEmbeddingProvider { - /// Creates a provider embedding under `contract` within `limits`, reporting to `progress`. + /// Configures batching, reuse identity, and progress for `generator`. + /// + /// `contract` must describe the generator's configuration. The constructor records its + /// fingerprint without changing or inspecting that configuration. `limits` must satisfy + /// [`RequestLimits`]'s arithmetic requirement. #[must_use] pub(crate) fn new( generator: G, @@ -153,26 +154,17 @@ impl ExternalEmbeddingProvider { } } - /// Proves the provider answers, before anything expensive happens. + /// Checks one generator call for a single canonical-width response. /// - /// One request for one short text, through the same generator, contract and canonical - /// validation the workload uses. It settles what a run cannot recover from and otherwise - /// discovers late, namely credentials the provider refuses, an endpoint that does not answer, - /// and a model whose vectors are not the canonical width. - /// - /// The check is unconditional, including for runs that turn out to reuse every card. A run - /// learns whether anything needs embedding only after it reads the store and assembles the - /// cards, which is exactly the work whose cost this exists to avoid paying twice. + /// Use this before expensive input preparation to detect provider refusal or incompatible + /// response shape early. The generator receives one short text and may retry internally. This + /// check bypasses [`RequestLimits`] and emits no batch progress. It checks only response count + /// and width. Success does not guarantee that later requests succeed. /// /// # Errors /// - /// Returns the same [`ExternalEmbeddingError`] a workload would: [`Provider`] when the request - /// fails, [`BatchCount`] when one text does not return one vector, and [`Dimensions`] when the - /// returned vector is not [`CANONICAL_DIMENSIONS`] wide. - /// - /// [`Provider`]: ExternalEmbeddingError::Provider - /// [`BatchCount`]: ExternalEmbeddingError::BatchCount - /// [`Dimensions`]: ExternalEmbeddingError::Dimensions + /// Returns [`ExternalEmbeddingError`] for generator failure, response-count mismatch, or + /// noncanonical width, in that order. pub(crate) async fn preflight(&self) -> Result<(), ExternalEmbeddingError> where G: EmbeddingGenerator, @@ -194,13 +186,18 @@ impl ExternalEmbeddingProvider { Ok(()) } - /// Admits the workload's next text into the request under assembly, or breaks. + /// Admits the next text or finishes the batch with a stop reason. + /// + /// `cost` must describe `batch`, under [`RequestLimits`]'s arithmetic requirement. `index` + /// identifies the next text in the workload. Successful admission consumes that text and + /// returns [`ControlFlow::Continue`], while a break leaves the text unconsumed. /// - /// Breaks with `Ok` when the request is full, either because a further text would cross a - /// ceiling or because the workload holds no more texts. Breaks with the workload-stopping error - /// when the next text cannot embed at all. An empty request admits any text that fits the token - /// ceiling alone, so a break with texts on the iterator always leaves a non-empty request. - /// `index` locates the next text in the workload for error reports. + /// The break contains `Ok(())` at the document ceiling, iterator exhaustion, or when another + /// text would exceed an accumulated cost ceiling. An error reports a reserved token or a text + /// that exceeds the token limit by itself. The batch and cost remain unchanged on a break. + /// + /// A nonempty iterator and empty batch either admit a text or return an error. This ensures + /// that a successful break with remaining texts supplies a nonempty batch. fn admit<'text>( &self, batch: &mut Vec<&'text str>, @@ -262,16 +259,14 @@ impl CardEmbedder &self, texts: impl IntoIterator + Send, ) -> Result>, Self::Error> { - // The provider counts the workload before the first request goes - // out. Every report then states its position against the whole. + // collecting references fixes the denominator for every completed-batch report. let texts: Vec<&str> = texts.into_iter().collect(); let total = texts.len(); let mut iter = texts.into_iter().peekable(); let mut embeddings = Vec::new(); - // One request buffer serves the whole workload, cleared between - // requests. Every text shares the workload lifetime, so reuse - // costs nothing. + // reusing the batch buffer retains its capacity between requests. Its text references + // borrow the workload. let mut batch: Vec<&str> = Vec::new(); let mut offset = 0; @@ -287,8 +282,9 @@ impl CardEmbedder } } - // Texts were on the iterator, so the admission contract - // guarantees a non-empty request here. + // With a nonempty iterator, admission either accepts a text or returns an error before + // a successful break. This loop began with remaining texts and propagated admission + // errors. Therefore the batch is nonempty. let generated = self .generator .create_embeddings(&batch) @@ -316,7 +312,15 @@ impl CardEmbedder } } -/// Converts one provider vector to the canonical width. +/// Copies a canonical-width provider vector into aligned storage. +/// +/// `index` identifies the text for error reporting. Component values copy unchanged, including +/// non-finite values. +/// +/// # Errors +/// +/// Returns [`ExternalEmbeddingError::Dimensions`] when the width differs from +/// [`CANONICAL_DIMENSIONS`]. fn canonical( embedding: Embedding<'static>, index: usize, @@ -332,7 +336,7 @@ fn canonical( Ok(BoxedVecN::new(VecN::from_ref(components))) } -/// An [`ExternalEmbeddingProvider`] workload failed. +/// A batching, generator, or response-shape failure during card embedding. #[derive(Debug)] pub enum ExternalEmbeddingError { /// The generator failed a request. @@ -343,7 +347,9 @@ pub enum ExternalEmbeddingError { Dimensions { index: usize, actual: usize }, /// A text contains a token the encoding reserves for protocol use. ReservedToken { index: usize, token: &'static str }, - /// A single text exceeds the per-request token ceiling in the stricter provider accounting. + /// A single text exceeds the token ceiling under the stricter local accounting. + /// + /// `tokens` is the larger of its tokenizer count and rounded-up byte estimate. OversizedText { index: usize, tokens: usize }, } diff --git a/libs/@local/graph/atlas/src/salt/embedding/external/tests.rs b/libs/@local/graph/atlas/src/salt/embedding/external/tests.rs index 60adaa645bf..b85a149c6ee 100644 --- a/libs/@local/graph/atlas/src/salt/embedding/external/tests.rs +++ b/libs/@local/graph/atlas/src/salt/embedding/external/tests.rs @@ -24,6 +24,7 @@ use crate::{ salt::embedding::CardEmbedder as _, }; +/// Declares the OpenAI model and float encoding named by the fixtures. fn contract() -> EmbeddingContract<'static> { EmbeddingContract { provider: "openai", @@ -33,7 +34,7 @@ fn contract() -> EmbeddingContract<'static> { } } -/// A deterministic vector for one text: component 0 is the text length. +/// Builds a zero vector with the text's byte length in component zero. #[expect( clippy::cast_precision_loss, reason = "fixture texts are a handful of bytes, exactly representable in f32" @@ -44,13 +45,22 @@ fn vector_for(text: &str) -> Vec { components } -/// Generates deterministically via [`vector_for`] and records requests. +/// A deterministic fixture generator with a request log. +/// +/// # Panics +/// +/// Recording or reading requests panics if the fixture mutex is poisoned. #[derive(Default)] struct RecordingGenerator { requests: Mutex>>, } impl RecordingGenerator { + /// Returns the recorded text batches in request order. + /// + /// # Panics + /// + /// Panics if the fixture mutex is poisoned. fn requests(&self) -> Vec> { self.requests .lock() @@ -76,13 +86,22 @@ impl EmbeddingGenerator for RecordingGenerator { } } -/// An observer recording every request the provider reports, shared with its clones. +/// A shared observer log of completed embedding batches. +/// +/// # Panics +/// +/// Recording or reading batches panics if the fixture mutex is poisoned. #[derive(Debug, Default, Clone)] struct RecordingProgress { batches: Arc>>, } impl RecordingProgress { + /// Returns the completed-batch reports in report order. + /// + /// # Panics + /// + /// Panics if the fixture mutex is poisoned. fn batches(&self) -> Vec { self.batches .lock() @@ -92,7 +111,6 @@ impl RecordingProgress { } impl Progress for RecordingProgress { - /// The fixture watches embedding batches, so nothing crosses into owning machinery. type Detached = NoProgress; fn detach(&self) -> NoProgress { @@ -107,7 +125,7 @@ impl Progress for RecordingProgress { } } -/// Returns vectors of a wrong width. +/// A fixture generator returning 512-component vectors. struct NarrowGenerator; impl EmbeddingGenerator for NarrowGenerator { @@ -122,7 +140,7 @@ impl EmbeddingGenerator for NarrowGenerator { } } -/// Fails every request. +/// A fixture generator returning a rate-limit error for every request. struct FailingGenerator; impl EmbeddingGenerator for FailingGenerator { @@ -134,7 +152,7 @@ impl EmbeddingGenerator for FailingGenerator { } } -/// Answers every request with no vectors at all. +/// A fixture generator returning an empty vector collection. struct SilentGenerator; impl EmbeddingGenerator for SilentGenerator { @@ -165,8 +183,7 @@ fn fingerprints_commit_to_every_contract_field() { #[test] fn fingerprints_distinguish_field_boundaries() { - // Both contracts concatenate to the same bytes; the length prefixes - // must keep them apart. + // the raw field concatenations are equal. Length prefixes distinguish their boundaries. let mut left = contract(); left.provider = "ab"; left.endpoint = "c"; @@ -242,9 +259,6 @@ async fn every_completed_request_reports_its_position_in_the_workload() { .await .expect("the fixture generator should embed every text"); - // The provider issues three requests for the five texts. Each report - // counts the texts behind it against the workload the caller handed - // the provider, and the last report closes on the full workload. assert_eq!( progress.batches(), [ @@ -270,17 +284,14 @@ async fn a_failed_request_reports_nothing() { .await .expect_err("the fixture generator fails every request"); - // A report is a completion, so a workload that never completes one - // leaves the operator's counter where it was. assert_eq!(progress.batches(), []); } #[tokio::test] async fn splits_requests_at_the_token_ceiling() { let generator = RecordingGenerator::default(); - // Each fixture word counts one cl100k token in at most four bytes, so - // the exact count is the binding accounting and a ceiling of two - // tokens admits two words per request. + // each word counts one cl100k token in at most four bytes. At limit two, both accountings admit + // two words and the token count excludes a third. let provider = ExternalEmbeddingProvider::new( generator, &contract(), @@ -308,9 +319,7 @@ async fn splits_requests_at_the_token_ceiling() { #[tokio::test] async fn splits_requests_at_the_byte_estimate_ceiling() { - // A word whose byte estimate exceeds its exact count, so the - // provider's admission gate binds the request size rather than the - // tokenizer. + // this word's byte estimate exceeds its tokenizer count. let word = " information"; let tokens = Cl100kTokenizer .count_tokens(word) @@ -322,9 +331,8 @@ async fn splits_requests_at_the_byte_estimate_ceiling() { ); let generator = RecordingGenerator::default(); - // The ceiling is six estimated tokens per request. Two twelve-byte - // words fit (⌈24 / 4⌉ = 6) and a third crosses. The exact count - // alone would admit all five in one request. + // two twelve-byte words meet the limit six: ⌈24 / 4⌉ = 6. A third exceeds it. Each word is one + // token, allowing all five under the tokenizer count alone. let provider = ExternalEmbeddingProvider::new( generator, &contract(), @@ -384,9 +392,8 @@ async fn rejects_a_text_above_the_byte_estimate_ceiling() { NoProgress, ); - // The text carries two exact tokens in twenty-four bytes. The - // tokenizer allows it, and the gate's estimate (⌈24 / 4⌉ = 6) puts - // it above the ceiling. + // the text has two tokens in twenty-four bytes. The byte estimate ⌈24 / 4⌉ = 6 exceeds limit + // three. let result = provider.embed([" information information"]).await; assert_matches!( @@ -467,10 +474,6 @@ async fn a_preflight_spends_one_request_on_one_text() { .expect("the fixture generator should answer the preflight"); assert_eq!(provider.generator.requests(), [[PREFLIGHT_TEXT]]); - // The preflight is not the workload, so it moves no counter and the - // operator's embedding bar still starts at the first card. Today the - // impl block carries no `Progress` bound at all; this assertion is - // what fails if one is ever added for a report from here. assert_eq!(progress.batches(), []); } @@ -523,8 +526,6 @@ async fn a_preflight_refuses_an_answer_of_the_wrong_length() { let result = provider.preflight().await; - // One text is one vector. The provider refuses any other count - // rather than indexing into the response. assert_matches!( result, Err(ExternalEmbeddingError::BatchCount { diff --git a/libs/@local/graph/atlas/src/salt/embedding/mod.rs b/libs/@local/graph/atlas/src/salt/embedding/mod.rs index 3a8cbc00290..2dc75302ab3 100644 --- a/libs/@local/graph/atlas/src/salt/embedding/mod.rs +++ b/libs/@local/graph/atlas/src/salt/embedding/mod.rs @@ -1,25 +1,22 @@ //! Card embedding with cross-generation reuse. //! -//! Every scoped type's relation card embeds once per generation into a canonical 3,072-component -//! vector; the vectors form the card-embedding table, row-aligned with the ontology stream that -//! produced the cards. A card's identity is the SHA-256 of its rendered text. Equal texts embed -//! once within a generation, and a run copies any row whose text the prior generation already -//! carries straight from that generation's table, without touching the provider. +//! [`embed_cards`] produces one 3,072-component vector per [`Card`], in input row order. It groups +//! cards by the SHA-256 of their rendered text, copies matching rows from a compatible prior table, +//! and submits the remaining distinct hashes' texts to a [`CardEmbedder`] in one call. Equal texts +//! share an embedding. Hash equality is the reuse key, without a second text comparison. //! -//! [`embed_cards`] is the entry point. It consumes finished [`Card`]s in ontology row order and -//! deduplicates them by text hash. It satisfies what the prior generation covers and submits the -//! remaining unique texts to a [`CardEmbedder`] in one call. Request sizing against provider -//! ceilings is the embedder's own concern. The run's [`Progress`] observer sees the resolved reuse -//! split and every request the embedder completes, so the paid part of a fit is legible while it -//! runs. The assembled [`CardEmbeddingTable`] serializes into two array files: the `f32[T, 3072]` -//! embedding matrix and the `u8[T, 32]` card-hash column. Card texts are not published, so the hash -//! column is the persisted key that lets the next generation match its freshly rendered cards -//! against these rows. +//! The [`CardEmbeddingTable`] writes an `f32[T, 3072]` embedding matrix and a `u8[T, 32]` card-hash +//! column, where T is the card count. Persisting text hashes instead of card texts lets a later +//! generation match freshly rendered cards against these rows. A [`CardEmbeddingView`] borrows the +//! columns, allowing reuse directly from mapped files. //! -//! A prior generation arrives as a [`CardEmbeddingView`] of borrowed columns, exactly the shape a -//! mapped pair of published files yields, so reuse reads the prior table without materializing it -//! in memory. Reuse is sound only between equal embedding contracts, so every embedder states an -//! [`EmbedderFingerprint`], and a run ignores an entire view whose recorded fingerprint differs. +//! Every embedder declares an [`EmbedderFingerprint`]. A prior view participates only when its +//! fingerprint matches. Interchangeability depends on that declaration accurately identifying the +//! vector-producing contract, including any model revision that affects reuse. +//! +//! The [`Progress`] observer receives the reuse split before embedding starts. Request sizing and +//! batch reporting belong to the embedder. [`external::ExternalEmbeddingProvider`] supplies both +//! for an external generator. use core::{error::Error, fmt}; use std::{collections::HashMap, io}; @@ -43,13 +40,10 @@ mod tests; /// Identity of one complete embedding contract. /// -/// The digest's preimage covers everything that determines the vector a text embeds to: provider, -/// endpoint, model identity, requested dimension, and encoding configuration. Equal fingerprints -/// promise interchangeable vectors for equal texts. -/// -/// A persisted table records the fingerprint that minted it, and a run copies rows out of a prior -/// generation only under a matching fingerprint, so a contract change invalidates every cached row -/// at once. +/// The declared contract must identify the settings that determine vector interchangeability for +/// equal texts, including the provider, endpoint, model revision, dimension, and encoding. Changing +/// the fingerprint invalidates every prior row for reuse. The digest itself does not validate the +/// declaration against the running provider. #[derive( Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, serde::Serialize, serde::Deserialize, )] @@ -58,7 +52,7 @@ mod tests; pub(crate) struct EmbedderFingerprint(Sha256Digest); impl EmbedderFingerprint { - /// Wraps the digest of an embedding-contract preimage. + /// Records a digest identifying the declared embedding contract. #[inline] #[must_use] pub(crate) const fn new(digest: Sha256Digest) -> Self { @@ -68,46 +62,53 @@ impl EmbedderFingerprint { /// A provider turning card texts into canonical embeddings. pub(crate) trait CardEmbedder { + /// The failure [`embed`](Self::embed) reports. type Error; /// Returns the identity of the embedding contract this provider serves. /// - /// See [`EmbedderFingerprint`]. + /// # Implementation Note + /// + /// The fingerprint must satisfy [`EmbedderFingerprint`]'s interchangeability contract. fn fingerprint(&self) -> EmbedderFingerprint; /// Embeds every text, returned in input order. /// - /// One call covers the whole workload: an implementation splits it into as many provider - /// requests as its own ceilings (document counts, token totals) require. + /// One call covers the whole workload. Request sizing and provider-specific limits belong to + /// the implementation. + /// + /// # Implementation Note + /// + /// Return exactly one vector per text, in input order, under the declared fingerprint. A + /// provider may make several requests before returning an error. /// /// # Errors /// - /// Returns a provider-defined error when embedding fails; the caller treats the whole workload - /// as failed. + /// Returns a provider-defined error when embedding fails. An error returns no partial vector + /// collection. fn embed<'text>( &self, texts: impl IntoIterator + Send, ) -> impl Future>, Self::Error>> + Send; } -/// Where the rows of one [`embed_cards`] run came from. +/// The reuse split over distinct card-text hashes. /// -/// The counts describe unique texts: `reused + embedded` is the number of distinct card texts, and -/// rows beyond that count are duplicates resolved without provider or prior-table work. Destined -/// for the generation metadata document. +/// For a completed [`embed_cards`] run, `reused + embedded` equals the number of distinct hashes. +/// Duplicate card rows share those embeddings. The default counts are zero. #[derive(Debug, Copy, Clone, PartialEq, Eq, Default, serde::Serialize, serde::Deserialize)] pub struct CardEmbeddingStats { - /// Unique texts copied from the prior generation's table. + /// Distinct text hashes copied from the prior generation's table. pub reused: usize, - /// Unique texts submitted to the provider. + /// Distinct text hashes requiring provider embeddings. pub embedded: usize, } /// Borrowed card-embedding columns of one generation. /// -/// Row `i` holds the embedding and text hash of the card at ontology row `i`; the -/// ordinal-to-type-id mapping is the type table's. The columns are exactly what the two published -/// array files contain, so a view over mapped files reads a prior generation in place. +/// Row `i` holds a card's embedding and text hash. The columns share positional rows and can borrow +/// the published array files directly. Their producer supplies the fingerprint and hash/vector +/// correspondence. #[derive(Debug, Copy, Clone)] pub(crate) struct CardEmbeddingView<'table> { fingerprint: EmbedderFingerprint, @@ -118,8 +119,9 @@ pub(crate) struct CardEmbeddingView<'table> { impl<'table> CardEmbeddingView<'table> { /// Creates a view over row-aligned columns. /// - /// `rows` is the embedding matrix as its SIMD-aligned rows; the view exists exactly when it - /// holds one row per hash. + /// Returns [`None`] unless there is exactly one vector per hash. The fingerprint and + /// hash/vector correspondence are producer assertions. This constructor checks no component + /// values. #[must_use] pub(crate) const fn new( fingerprint: EmbedderFingerprint, @@ -137,7 +139,7 @@ impl<'table> CardEmbeddingView<'table> { }) } - /// Returns the fingerprint of the contract that produced every row. + /// Returns the declared embedding-contract fingerprint. #[inline] #[must_use] pub(crate) const fn fingerprint(&self) -> EmbedderFingerprint { @@ -161,11 +163,10 @@ impl<'table> CardEmbeddingView<'table> { } } -/// The owned card-embedding table one [`embed_cards`] run assembles. +/// Row-aligned card embeddings and text hashes ready for publication. /// -/// The row semantics are [`CardEmbeddingView`]'s. Owning the columns is what the generation under -/// construction needs before it writes its files. Every read surface is on the -/// [`view`](Self::view). +/// The row semantics are [`CardEmbeddingView`]'s. Construction checks equal column lengths and +/// accepts the supplied component values and fingerprint. #[derive(Debug, Clone, PartialEq)] pub(crate) struct CardEmbeddingTable { fingerprint: EmbedderFingerprint, @@ -218,8 +219,8 @@ impl CardEmbeddingTable { /// Writes the `f32[T, 3072]` embedding matrix as an array file. /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. + /// Components use native byte order, recorded by the array header. Returns the SHA-256 of the + /// written bytes. A zero-row table writes a header-only empty array. /// /// # Errors /// @@ -242,8 +243,7 @@ impl CardEmbeddingTable { /// Writes the `u8[T, 32]` card-hash column as an array file. /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. + /// Returns the SHA-256 of the written bytes. A zero-row table writes a header-only empty array. /// /// # Errors /// @@ -259,14 +259,17 @@ impl CardEmbeddingTable { } } -/// [`embed_cards`] failed to produce a complete table. +/// A provider or vector-validation failure while assembling card embeddings. #[derive(Debug)] pub(crate) enum CardEmbeddingError { /// The provider failed to embed the workload. Embedder(E), /// The provider returned a different number of rows than requested. RowCount { expected: usize, actual: usize }, - /// A returned embedding carries a non-finite component. + /// A newly returned embedding carries a non-finite component. + /// + /// `row` is the first input card with that hash, and `component` is its first non-finite + /// component. NonFinite { row: R, component: usize }, } @@ -295,10 +298,10 @@ impl Error for CardEmbeddingError { } } -/// One distinct card text awaiting an embedding. +/// The first text and all input rows sharing one card-text hash. struct UniqueCard<'card, R> { text: &'card str, - /// Card rows carrying this text, ascending. + /// Card rows carrying this hash, ascending. rows: Vec, } @@ -306,17 +309,28 @@ struct UniqueCard<'card, R> { /// /// `cards` use their own row domain `R` and row `i` of the returned table belongs to `cards[i]`. /// -/// Equal texts embed once. A `prior` view serves rows whose text hash it contains, provided its -/// fingerprint equals the embedder's. The provider sees exactly the texts neither source covers, in -/// one [`embed`](CardEmbedder::embed) call, and sees nothing when those sources cover every row. +/// Equal text hashes share one embedding. A `prior` view supplies matching hashes only when its +/// fingerprint equals the embedder's. Repeated prior hashes select the last row. Reused vectors +/// copy verbatim, without a finiteness check. Supply a prior table with valid vectors and accurate +/// hash/contract metadata. /// -/// `progress` observes the resolved split once and then each request the provider completes, -/// counted against the unique texts the split handed the provider. +/// The provider receives the first text for each remaining hash, in first-occurrence order, through +/// one [`embed`](CardEmbedder::embed) call. It receives no call when every row reuses or the input +/// is empty. Newly returned vectors undergo a finiteness check before the completed table is +/// returned. +/// +/// `progress` receives the split once before the provider call. Batch reports require an embedder +/// configured with its own observer, such as [`external::ExternalEmbeddingProvider`]. /// /// # Errors /// -/// Returns an error when the provider fails, when it changes the row count, or when it returns a -/// vector with a non-finite component. A failed run leaves no partial table. +/// Returns [`CardEmbeddingError`] for provider failure, changed row count, or newly returned +/// non-finite components, in that order. A failed run returns no partial table. Provider work +/// already performed is not rolled back. +/// +/// # Panics +/// +/// Panics if the output matrix's aligned allocation size exceeds `isize::MAX`. pub(crate) async fn embed_cards( embedder: &E, cards: &IdSlice, @@ -373,9 +387,8 @@ pub(crate) async fn embed_cards( embedding: &AlignedVecN, row: R, @@ -427,7 +444,7 @@ fn validate_finite( return Ok(()); } - // Slow/cold path: find the first non-finite component + // the vector-wide check found a non-finite component. Locate its index for the error. let Some(component) = embedding .as_array() .iter() diff --git a/libs/@local/graph/atlas/src/salt/embedding/tests.rs b/libs/@local/graph/atlas/src/salt/embedding/tests.rs index adb7fcbd637..f6d76d6b754 100644 --- a/libs/@local/graph/atlas/src/salt/embedding/tests.rs +++ b/libs/@local/graph/atlas/src/salt/embedding/tests.rs @@ -25,15 +25,16 @@ use crate::{ progress::{NoProgress, Progress}, }; +/// Hashes a fixture contract preimage into an embedder fingerprint. fn fingerprint(preimage: &[u8]) -> EmbedderFingerprint { let mut hasher = Sha256::new(); hasher.update(preimage); EmbedderFingerprint::new(hasher.finalize()) } -/// A deterministic vector for one text. +/// Builds a vector with the text's byte length plus `offset` in component zero. /// -/// Component 0 is the text length plus the embedder's offset, every other component is zero. +/// Every other component is zero. #[expect( clippy::cast_precision_loss, reason = "fixture card texts are a handful of bytes, exactly representable in f32" @@ -44,13 +45,22 @@ fn vector_for(text: &str, offset: f32) -> BoxedVecN { vector } +/// Reads the fixture embedder's text-dependent component at `row`. +/// +/// # Panics +/// +/// Panics if the row is outside the table. fn component_zero(view: CardEmbeddingView<'_>, row: u64) -> f32 { view.embedding(OntologyRowId::new(row)) .expect("the row should be inside the table") .as_array()[0] } -/// Embeds deterministically via [`vector_for`] and records every call. +/// A deterministic fixture embedder with a call log. +/// +/// # Panics +/// +/// Recording or reading calls panics if the fixture mutex is poisoned. struct RecordingEmbedder { fingerprint: EmbedderFingerprint, offset: f32, @@ -58,6 +68,7 @@ struct RecordingEmbedder { } impl RecordingEmbedder { + /// Creates an embedder with the named contract and component-zero offset. fn new(preimage: &[u8], offset: f32) -> Self { Self { fingerprint: fingerprint(preimage), @@ -66,6 +77,11 @@ impl RecordingEmbedder { } } + /// Returns the recorded text batches in call order. + /// + /// # Panics + /// + /// Panics if the fixture mutex is poisoned. fn calls(&self) -> Vec> { self.calls .lock() @@ -99,7 +115,7 @@ impl CardEmbedder for RecordingEmbedder { } } -/// Returns one row fewer than requested. +/// A fixture embedder that omits the first requested row. struct ShortEmbedder; impl CardEmbedder for ShortEmbedder { @@ -122,7 +138,7 @@ impl CardEmbedder for ShortEmbedder { } } -/// Returns a NaN component in every row. +/// A fixture embedder producing NaN in component seven of every row. struct NanEmbedder; impl CardEmbedder for NanEmbedder { @@ -148,13 +164,22 @@ impl CardEmbedder for NanEmbedder { } } -/// An observer recording the splits the card-embedding stage resolves. +/// An observer recording card-embedding reuse splits. +/// +/// # Panics +/// +/// Recording or reading splits panics if the fixture mutex is poisoned. #[derive(Debug, Default)] struct RecordingProgress { splits: Mutex>, } impl RecordingProgress { + /// Returns the observed reuse splits in report order. + /// + /// # Panics + /// + /// Panics if the fixture mutex is poisoned. fn splits(&self) -> Vec { self.splits .lock() @@ -164,7 +189,6 @@ impl RecordingProgress { } impl Progress for RecordingProgress { - /// The fixture watches the reuse split, so nothing crosses into owning machinery. type Detached = NoProgress; fn detach(&self) -> NoProgress { @@ -179,6 +203,7 @@ impl Progress for RecordingProgress { } } +/// Builds one verbatim fixture card per input text, in order. fn cards(texts: &[&str]) -> IdVec { texts .iter() @@ -220,8 +245,8 @@ async fn reuses_prior_rows_under_an_equal_fingerprint() { .await .unwrap_or_else(|error| panic!("the fixture embedder is infallible: {error}")); - // The second embedder serves the same contract but would produce - // shifted vectors, so a row equal to the prior vector proves reuse. + // the fixture deliberately reuses the fingerprint with a different offset to distinguish a + // copied vector from a fresh one. let embedder = RecordingEmbedder::new(b"contract", 100.0); let (table, stats) = embed_cards( &embedder, @@ -262,9 +287,6 @@ async fn the_resolved_split_reaches_the_observer_before_the_provider_does() { .await .unwrap_or_else(|error| panic!("the fixture embedder is infallible: {error}")); - // The observer learns the whole split (what the prior covers and - // what the run asks the provider for) as one report, and the - // published evidence says the same thing. assert_eq!( progress.splits(), [CardEmbeddingStats { @@ -298,8 +320,6 @@ async fn a_wholly_reused_workload_still_reports_its_split() { .await .unwrap_or_else(|error| panic!("the fixture embedder is infallible: {error}")); - // Nothing goes to the provider, and the operator still learns why - // the stage costs nothing. assert_eq!(embedder.calls(), [] as [Vec; 0]); assert_eq!( progress.splits(), @@ -437,7 +457,6 @@ fn view_exists_exactly_for_row_aligned_columns() { assert_eq!(view.hashes.len(), 2); assert!(view.embedding(OntologyRowId::new(2)).is_none()); - // The count clause, violated from either side. assert!( CardEmbeddingView::new(fingerprint(b"contract"), &hashes[..1], rows).is_none(), "an extra row must not form a view", @@ -448,12 +467,14 @@ fn view_exists_exactly_for_row_aligned_columns() { ); } +/// Computes the SHA-256 digest used to key a card's text. fn text_digest(text: &str) -> Sha256Digest { let mut hasher = Sha256::new(); hasher.update(text.as_bytes()); hasher.finalize() } +/// Embeds `alpha`, `beta`, `alpha` with the recording fixture provider. async fn three_row_table() -> super::CardEmbeddingTable { let embedder = RecordingEmbedder::new(b"contract", 0.0); let (table, _) = embed_cards( @@ -492,8 +513,8 @@ async fn writes_the_embedding_matrix_as_an_array_file() { "the file must satisfy the format's length equation", ); - // Row i starts at the header boundary plus i full rows; component 0 - // carries the fixture's per-text value. + // row i starts at the header boundary plus i full rows. Component zero carries the fixture's + // text-dependent value. for (row, expected) in [(0_usize, 5.0_f32), (1, 4.0), (2, 5.0)] { let offset = PAGE_BYTES + row * CANONICAL_DIMENSIONS * size_of::(); let component = f32::from_le_bytes( diff --git a/libs/@local/graph/atlas/src/salt/file/mod.rs b/libs/@local/graph/atlas/src/salt/file/mod.rs index adc5d2d3c96..106fdca5845 100644 --- a/libs/@local/graph/atlas/src/salt/file/mod.rs +++ b/libs/@local/graph/atlas/src/salt/file/mod.rs @@ -1,3 +1,8 @@ +//! Mapped point and vector matrices with typed row domains. +//! +//! [`PointFile`] admits two-component points, and [`VectorFile`] admits SIMD-aligned vectors of a +//! fixed width. [`FinitePointFile`] retains a finiteness check across repeated point access. + mod point; mod vector; diff --git a/libs/@local/graph/atlas/src/salt/file/point.rs b/libs/@local/graph/atlas/src/salt/file/point.rs index ae9b6d55dd5..5ac2807199f 100644 --- a/libs/@local/graph/atlas/src/salt/file/point.rs +++ b/libs/@local/graph/atlas/src/salt/file/point.rs @@ -1,5 +1,3 @@ -//! Mapped representation matrices with their row domain in the handle type. - use core::{error::Error, fmt, marker::PhantomData, ops::Deref, ptr::NonNull}; use std::path::Path; @@ -10,12 +8,12 @@ use crate::{ math::{FinitePointField, NonFinitePoint, Vec2}, }; -/// A coordinate or representation matrix failed to open as `f32` rows of the expected shape. +/// A failure to open an array as two-component points. #[derive(Debug)] pub(crate) enum OpenPointError { /// The underlying array file failed to open. Open(OpenArrayError), - /// The array does not hold `f32` rows of the expected shape. + /// The array is not native `f32[T, 2]` or an empty native-`f32` array. InvalidArray, } @@ -45,20 +43,19 @@ impl Error for OpenPointError { } } -/// A mapped matrix of aligned `f32` rows, addressed by the row domain `I`. +/// A mapped two-component point column with row domain `I`. +/// +/// The row type distinguishes, for example, corpus and distinct-row point columns at typed +/// interfaces. The file itself supplies shape and component bytes, not a row-domain identifier. +/// Choose `I` to match the artifact's provenance. /// -/// The handle owns the mapping and is the matrix: it dereferences to the typed row slice, with -/// the row domain traveling in the type, so a corpus-row matrix and a distinct-row matrix are -/// different types a call cannot confuse. Where a theorem identifies two domains, the -/// identification lives with the theorem's owner - the quotient's `training()` reborrows the corpus -/// under the distinct domain instead of retyping the handle. +/// Construction accepts finite and non-finite components. [`Self::finite`] checks every point and +/// retains the result in a [`FinitePointFile`]. The mapped file must remain immutable under +/// [`crate::file::region::PageMap`]'s file contract. pub(crate) struct PointFile { - /// The mapped row slice, validated at construction. - /// - /// The pointee lives inside the mapping owned by `_file`, whose address is stable under moves - /// of this handle, so the pointer stays valid for exactly as long as the handle lives. + /// The validated row slice within `_file`'s mapping, whose address survives handle moves. rows: NonNull<[Vec2]>, - /// The mapping. Held for its lifetime alone: every read goes through `rows`. + /// Owns the mapping and its shared advisory lock for the lifetime of `rows`. _file: ArrayFile, _marker: PhantomData, } @@ -67,12 +64,13 @@ impl PointFile where I: Id, { - /// Validates an open array file as a matrix of aligned `f32` rows of width `N`. + /// Validates an array as a two-component point column. + /// + /// Accepts native `f32[T, 2]` or a header-only empty native-`f32` array. /// /// # Errors /// - /// Returns [`OpenPointError::InvalidArray`] when the array does not hold aligned `f32` rows - /// of width `N`. + /// Returns [`OpenPointError::InvalidArray`] for another element type or shape. pub(crate) fn new(file: ArrayFile) -> Result { let rows = file.points().ok_or(OpenPointError::InvalidArray)?; let rows = NonNull::from(rows); @@ -84,19 +82,25 @@ where }) } - /// Maps the file at `path` as a matrix of aligned `f32` rows of width `N`. + /// Maps `path` as a two-component point column. /// /// # Errors /// - /// Returns [`OpenPointError::Open`] when the file does not open as an array, and - /// [`OpenPointError::InvalidArray`] when the array does not hold aligned `f32` rows of - /// width `N`. + /// Returns [`OpenPointError`] for array open failures or a type/shape mismatch. pub(crate) fn open(path: impl AsRef) -> Result { ArrayFile::open(path) .map_err(OpenPointError::from) .and_then(Self::new) } + /// Validates every coordinate and retains the mapping as a [`FinitePointFile`]. + /// + /// Validation takes O(T) time for T points. Later access reuses the check without rescanning or + /// copying the points. + /// + /// # Errors + /// + /// Returns the smallest row whose point has a NaN or infinite component. pub(crate) fn finite(self) -> Result, NonFinitePoint> { let points = &*self; let _field = FinitePointField::new(points)?; @@ -105,11 +109,14 @@ where } } -// SAFETY: the mapping is read-only for the handle's whole life, `rows` points into memory owned -// by `_file` within the same value, and no interior mutability exists, so moving the handle or -// sharing it across threads leaves every read valid. +// SAFETY: `Mmap` is Send and its mapped address survives moves. `rows` points to the validated +// slice owned through `_file`, which keeps the mapping and lock alive. The marker stores no `I` +// value. Therefore transferring the handle preserves the row pointer's validity under PageMap's +// immutable-file contract. unsafe impl Send for PointFile {} -// SAFETY: shared access only ever reads the immutable mapping. See the `Send` proof above. +// SAFETY: `Mmap` is Sync under its immutable-file contract. This handle exposes only shared point +// borrows and never remaps or mutates the file. It stores no `I` value to share. Therefore shared +// access introduces no mutable alias or data race. unsafe impl Sync for PointFile {} impl Deref for PointFile @@ -119,14 +126,20 @@ where type Target = IdSlice; fn deref(&self) -> &Self::Target { - // SAFETY: `rows` was derived from the mapping owned by `self._file` at construction, the - // mapping is immutable and lives as long as `self`, and the returned borrow is tied to - // `&self`, so the pointee is valid and unaliased by writes for the borrow's life. + // SAFETY: dereferencing a slice pointer requires a live, aligned, initialized range and + // shared access for its borrow. `new` saved exactly the slice returned by + // `ArrayFile::points`, and `_file` retains its mapping without moving the mapped address. + // PageMap's immutable-file contract preserves those bytes, and the result borrows no longer + // than `self`. Therefore the stored pointer may be read as this shared slice. let rows = unsafe { &*self.rows.as_ptr() }; IdSlice::from_raw(rows) } } +/// A mapped point column with finite coordinates. +/// +/// [`PointFile::finite`] establishes finiteness once. Shared access preserves it without +/// rescanning, under the mapped file's immutability contract. pub(crate) struct FinitePointFile { inner: PointFile, } @@ -140,13 +153,12 @@ where fn deref(&self) -> &Self::Target { let inner = &raw const *self.inner; - // `new_unchecked` would re-run its debug assert over every point on each deref, so the - // cast is taken directly. - // SAFETY: `FinitePointField` is `repr(transparent)` over `IdSlice`, so the - // cast preserves the address and the slice metadata. Its finiteness invariant was proven - // by `PointFile::finite` over these same rows at this handle's construction, and the - // mapping is read-only for the handle's whole life, so the proof cannot rot. The borrow - // is derived from `&self`, so the pointee outlives it. + // the direct cast avoids `new_unchecked`'s debug-only coordinate scan on every access. + // SAFETY: `FinitePointField` is transparent over `IdSlice`, preserving layout + // and slice metadata. `PointFile::finite` checked these same rows, and the owned immutable + // mapping retains both their storage and finiteness. The result shares `self`'s borrow + // lifetime. Therefore the cast preserves the reference's validity and the finite-field + // invariant. unsafe { &*(inner as *const FinitePointField) } } } diff --git a/libs/@local/graph/atlas/src/salt/file/vector.rs b/libs/@local/graph/atlas/src/salt/file/vector.rs index ea849fc4bd3..a741c5c444d 100644 --- a/libs/@local/graph/atlas/src/salt/file/vector.rs +++ b/libs/@local/graph/atlas/src/salt/file/vector.rs @@ -1,5 +1,3 @@ -//! Mapped representation matrices with their row domain in the handle type. - use core::{error::Error, fmt, marker::PhantomData, ops::Deref, ptr::NonNull}; use std::path::Path; @@ -10,12 +8,12 @@ use crate::{ math::AlignedVecN, }; -/// A coordinate or representation matrix failed to open as `f32` rows of the expected shape. +/// A failure to open an array as SIMD-aligned fixed-width vectors. #[derive(Debug)] pub(crate) enum OpenVectorError { /// The underlying array file failed to open. Open(OpenArrayError), - /// The array does not hold `f32` rows of the expected shape. + /// The element type, shape, or row stride cannot supply the requested aligned vectors. InvalidArray, } @@ -45,20 +43,19 @@ impl Error for OpenVectorError { } } -/// A mapped matrix of aligned `f32` rows, addressed by the row domain `I`. +/// A mapped fixed-width vector column with row domain `I`. +/// +/// The row type distinguishes, for example, corpus and distinct-row columns at typed interfaces. +/// The file supplies shape and component bytes, not a row-domain identifier. Choose `I` to match +/// the artifact's provenance. /// -/// The handle owns the mapping and is the matrix: it dereferences to the typed row slice, with -/// the row domain traveling in the type, so a corpus-row matrix and a distinct-row matrix are -/// different types a call cannot confuse. Where a theorem identifies two domains, the -/// identification lives with the theorem's owner - the quotient's `training()` reborrows the corpus -/// under the distinct domain instead of retyping the handle. +/// Every row satisfies [`AlignedVecN`]'s alignment invariant. `N` must be nonzero, and its byte +/// stride must preserve that alignment. Construction accepts non-finite and unnormalized vectors. +/// The mapped file must remain immutable under [`crate::file::region::PageMap`]'s file contract. pub(crate) struct VectorFile { - /// The mapped row slice, validated at construction. - /// - /// The pointee lives inside the mapping owned by `_file`, whose address is stable under moves - /// of this handle, so the pointer stays valid for exactly as long as the handle lives. + /// The validated row slice within `_file`'s mapping, whose address survives handle moves. rows: NonNull<[AlignedVecN]>, - /// The mapping. Held for its lifetime alone: every read goes through `rows`. + /// Owns the mapping and its shared advisory lock for the lifetime of `rows`. _file: ArrayFile, _marker: PhantomData, } @@ -67,12 +64,15 @@ impl VectorFile where I: Id, { - /// Validates an open array file as a matrix of aligned `f32` rows of width `N`. + /// Validates an array as SIMD-aligned rows of width `N`. + /// + /// Accepts native `f32[T, N]` or an empty native-`f32` array. `N` must satisfy the type's + /// nonzero-width and alignment requirements, including for empty input. /// /// # Errors /// - /// Returns [`OpenVectorError::InvalidArray`] when the array does not hold aligned `f32` rows - /// of width `N`. + /// Returns [`OpenVectorError::InvalidArray`] for an incompatible element type, shape, or row + /// stride. pub(crate) fn new(file: ArrayFile) -> Result { let rows: &[AlignedVecN] = file.vectors().ok_or(OpenVectorError::InvalidArray)?; let rows = NonNull::from(rows); @@ -84,13 +84,12 @@ where }) } - /// Maps the file at `path` as a matrix of aligned `f32` rows of width `N`. + /// Maps `path` as SIMD-aligned vector rows of width `N`. /// /// # Errors /// - /// Returns [`OpenVectorError::Open`] when the file does not open as an array, and - /// [`OpenVectorError::InvalidArray`] when the array does not hold aligned `f32` rows of - /// width `N`. + /// Returns [`OpenVectorError`] for array open failures or an incompatible element type, shape, + /// or row stride. pub(crate) fn open(path: impl AsRef) -> Result { ArrayFile::open(path) .map_err(OpenVectorError::from) @@ -98,11 +97,14 @@ where } } -// SAFETY: the mapping is read-only for the handle's whole life, `rows` points into memory owned -// by `_file` within the same value, and no interior mutability exists, so moving the handle or -// sharing it across threads leaves every read valid. +// SAFETY: `Mmap` is Send and its mapped address survives moves. `rows` points to the validated +// aligned slice owned through `_file`, which keeps the mapping and lock alive. The marker stores no +// `I` value. Therefore transferring the handle preserves the row pointer and alignment under +// PageMap's immutable-file contract. unsafe impl Send for VectorFile {} -// SAFETY: shared access only ever reads the immutable mapping. See the `Send` proof above. +// SAFETY: `Mmap` is Sync under its immutable-file contract. This handle exposes only shared vector +// borrows and never remaps or mutates the file. It stores no `I` value to share. Therefore shared +// access introduces no mutable alias or data race. unsafe impl Sync for VectorFile {} const impl Deref for VectorFile @@ -112,9 +114,12 @@ where type Target = IdSlice>; fn deref(&self) -> &Self::Target { - // SAFETY: `rows` was derived from the mapping owned by `self._file` at construction, the - // mapping is immutable and lives as long as `self`, and the returned borrow is tied to - // `&self`, so the pointee is valid and unaliased by writes for the borrow's life. + // SAFETY: dereferencing this slice pointer requires shared access to a live initialized + // range with every row at its promised SIMD alignment. `new` saved exactly the slice + // returned by `ArrayFile::vectors`, and `_file` retains its mapping without moving the + // mapped address. PageMap's immutable-file contract preserves those bytes, and the result + // borrows no longer than `self`. Therefore dereferencing the stored pointer yields a valid + // shared aligned slice. let rows = unsafe { &*self.rows.as_ptr() }; IdSlice::from_raw(rows) } diff --git a/libs/@local/graph/atlas/src/salt/fit/annotations/mod.rs b/libs/@local/graph/atlas/src/salt/fit/annotations/mod.rs index fd1acce1c97..404da5b788e 100644 --- a/libs/@local/graph/atlas/src/salt/fit/annotations/mod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/annotations/mod.rs @@ -2,9 +2,9 @@ //! //! A fit receives an annotation corpus beside the dataset rather than deriving one from it, the //! same input category as the reviewed verdicts. [`SuppliedAnnotations`] runs the document's whole -//! wire contract at construction through the annotation reader and keeps the exact wire bytes, so -//! the staged artifact is byte-identical to the supplied file and the digest computed here is the -//! supplied file's identity. +//! wire contract at construction through the annotation reader and keeps the exact wire bytes. +//! The staged artifact is therefore byte-identical to the supplied file, and the digest computed +//! here is the supplied file's identity. //! //! The fit consumes the document through the training-set assembly: the classifier stage fits the //! relation-policy model from the assembled corpus and evaluates it on the corpus's holdout cards. @@ -53,15 +53,16 @@ impl Error for SupplyError { /// One validated annotation-corpus document with its exact wire bytes. /// -/// A value of this type is admissible by existence. Construction validated the document, so a fit -/// holding one stages the bytes verbatim and binds the digest without any further check. Admission -/// rejects a document that would fail it before the fit spends anything. +/// A value of this type is admissible by existence. Construction validated the document, and a +/// fit holding one therefore stages the bytes verbatim and binds the digest without any further +/// check. Admission rejects a document that would fail it before the fit spends anything. #[derive(Debug, Clone, PartialEq)] pub(crate) struct SuppliedAnnotations { - /// The exact wire bytes, kept beside their parse: staging writes these verbatim, so the - /// published digest binds precisely what admission validated. A re-serialization of - /// [`document`](Self::document) would bind different bytes; a staging-time re-read of the - /// source would bind unvalidated ones. + /// The exact wire bytes, kept beside their parse. + /// + /// Staging writes these verbatim, and the published digest therefore binds precisely what + /// admission validated. A re-serialization of [`document`](Self::document) can differ from + /// these bytes, and a staging-time re-read of the source would bind bytes admission never saw. bytes: Box<[u8]>, document: AnnotationCorpus, hash: Sha256Digest, diff --git a/libs/@local/graph/atlas/src/salt/fit/annotations/tests.rs b/libs/@local/graph/atlas/src/salt/fit/annotations/tests.rs index c6790a28e65..0d32e3217bd 100644 --- a/libs/@local/graph/atlas/src/salt/fit/annotations/tests.rs +++ b/libs/@local/graph/atlas/src/salt/fit/annotations/tests.rs @@ -10,7 +10,9 @@ use crate::{ salt::policy::annotation::InvalidAnnotationCorpus, }; +/// The record hash the fixture card and vote carry. const DIGEST: &str = "2a9934acae8bf210b6a3428e553b1bcc0e220a4de113940782cd573da1ea4f4b"; +/// The versioned HASH type URL of the fixture card. const EMPLOYED_BY: &str = "https://hash.ai/@h/types/entity-type/employed-by/v/1"; /// Composes a minimal contract-conforming document: one hash card carrying one geometry vote. @@ -70,6 +72,7 @@ fn document() -> String { .to_string() } +/// A fresh per-process scratch directory named `name` under the system temp dir. fn scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -82,6 +85,10 @@ fn scratch(name: &str) -> Utf8PathBuf { dir } +/// Keeps the document bytes verbatim in `from_bytes` and hashes exactly those bytes. +/// +/// `SuppliedAnnotations::from_bytes` keeps the document bytes verbatim and identifies them by the +/// SHA-256 of those wire bytes. #[test] fn construction_preserves_bytes_and_binds_their_digest() { let document = document(); @@ -112,6 +119,10 @@ fn contract_violation_is_rejected_at_supply() { ); } +/// Reads a written document verbatim and maps a missing file and malformed JSON to their variants. +/// +/// `open` reads a written document verbatim, reports a missing file as `Io`, and malformed JSON as +/// `Invalid`. #[test] fn open_reads_the_file_and_reports_both_failure_shapes() { let dir = scratch("open"); diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/classifier.rs b/libs/@local/graph/atlas/src/salt/fit/compute/classifier.rs index 38a02de1ff3..8a0c8624445 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/classifier.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/classifier.rs @@ -30,10 +30,10 @@ use crate::{ }, }; -/// The classifier stage failed and acquired no deployable model. +/// A classifier-stage failure, leaving the run without a deployable model. /// -/// One variant per way the stage refuses, so a classifier failure attributes to this stage by -/// construction. +/// One variant per way the stage refuses. A classifier failure therefore attributes to this +/// stage by construction. #[derive(Debug)] pub(crate) enum ClassifierError { /// The assembled corpus violates the classifier's training-set contract. @@ -79,10 +79,14 @@ impl Error for ClassifierError { /// The staged annotation artifacts of one in-run classifier fit. /// -/// The corpus document beside the embedding table it assembled to. +/// The corpus document beside the embedding table it assembled to and that table's card text +/// hashes. pub(super) struct AnnotationArtifacts { + /// The corpus document, staged verbatim. pub corpus: Binding, + /// The card embedding matrix, row-aligned with the assembled corpus. pub embeddings: Binding, + /// The card text hashes, one SHA-256 per assembled row. pub hashes: Binding, } @@ -108,10 +112,8 @@ impl AcquiredClassifier { /// /// # Errors /// - /// Returns [`ClassifierError::Training`] when the assembled corpus violates the classifier's - /// training-set contract, [`ClassifierError::Fit`] when the fit fails, - /// [`ClassifierError::Holdout`] when a holdout prediction overflows, and an I/O error when a - /// staged annotation artifact does not write. + /// A supplied model cannot fail. Fitting one returns a [`ClassifierError`], reached in the + /// order [`fit`](Self::fit) documents. pub(super) fn acquire( context: &Context, plan: &ClassifierPlan, @@ -132,6 +134,12 @@ impl AcquiredClassifier { } /// Fits the classifier from the assembled corpus and evaluates it on the holdout cards. + /// + /// # Errors + /// + /// Returns a [`ClassifierError`]. The body writes the two staged annotation artifacts (the + /// embedding table, then the hashes) before it checks the training-set contract and runs the + /// fit, and it predicts the holdout cards last. #[tracing::instrument(name = "classifier-fit", skip_all)] fn fit( context: &Context, @@ -151,7 +159,7 @@ impl AcquiredClassifier { corpus.table().write_hashes_into(writer) })?; - // The trained rows lead the embedding table; the holdout rows + // The trained rows lead the embedding table, and the holdout rows // after them are evaluation material. The pin claims the corpus's // card-row domain over the table's domain-neutral rows, and the // trained prefix keeps that domain. diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/coordinates.rs b/libs/@local/graph/atlas/src/salt/fit/compute/coordinates.rs index 49c8b584600..4818804f8f3 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/coordinates.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/coordinates.rs @@ -56,17 +56,16 @@ impl Error for OpenCoordinatesError { /// The staged coordinate column, mapped, proven finite, and addressed by the corpus row domain. /// -/// The value is the fit's coordinates: it owns the mapping and dereferences to the proven point -/// field, while the repository binding it holds is what the seal publishes. Finiteness is proven -/// once at the open, so every consumer downstream reads a [`FinitePointField`] and re-proves -/// nothing. +/// The value is the fit's coordinates: it owns the mapping, while the repository binding it holds +/// is what the seal publishes. Finiteness is proven once at the open, and every consumer +/// downstream reads a [`FinitePointField`] and re-proves nothing. pub(super) struct Coordinates { /// The staged column's typed repository binding. pub binding: Binding, /// The mapped point slice, validated as finite at construction. /// /// The pointee lives inside the mapping owned by `_file`, whose address is stable under moves - /// of this handle, so the pointer stays valid for exactly as long as the handle lives. + /// of this handle. The pointer therefore stays valid for exactly as long as the handle lives. points: NonNull<[Vec2]>, /// The mapping. Held for its lifetime alone: every read goes through `points`. _file: ArrayFile, @@ -75,8 +74,8 @@ pub(super) struct Coordinates { impl Coordinates { /// Maps the staged coordinate column back under its typed binding and proves it finite. /// - /// The binding is the one the staging boundary returned for the column's write, so the - /// mapped view carries the same identity the seal publishes. + /// The binding is the one the staging boundary returned for the column's write, and the + /// mapped view therefore carries the same identity the seal publishes. /// /// # Errors /// @@ -103,8 +102,8 @@ impl Coordinates { } // SAFETY: the mapping is read-only for the handle's whole life, `points` points into memory -// owned by `_file` within the same value, and no interior mutability exists, so moving the handle -// or sharing it across threads leaves every read valid. +// owned by `_file` within the same value, and no interior mutability exists. Moving the handle or +// sharing it across threads therefore leaves every read valid. unsafe impl Send for Coordinates {} // SAFETY: shared access only ever reads the immutable mapping. See the `Send` proof above. unsafe impl Sync for Coordinates {} @@ -115,7 +114,7 @@ impl Deref for Coordinates { fn deref(&self) -> &Self::Target { // SAFETY: `points` was derived from the mapping owned by `self._file` at construction, the // mapping is immutable and lives as long as `self`, and the returned borrow is tied to - // `&self`, so the pointee is valid and unaliased by writes for the borrow's life. + // `&self`. The pointee is therefore valid and unaliased by writes for the borrow's life. let points = unsafe { &*self.points.as_ptr() }; // Finiteness was proven over these exact bytes at the open, and the mapping is immutable. FinitePointField::new_unchecked(IdSlice::from_raw(points)) diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/error.rs b/libs/@local/graph/atlas/src/salt/fit/compute/error.rs index 1ae1147678e..ed78b777579 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/error.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/error.rs @@ -1,14 +1,14 @@ //! The compute run's failure surface. //! //! [`ComputeError`] holds one variant per stage: each stage owns a closed error enum beside its -//! stage type. The run carries the stage's enum here whole, so a failure attributes to its -//! stage by construction. The enum carries no dataset or provider type, so it is -//! `Send + 'static` by construction and crosses the rayon offload. The trunk keeps its own -//! legs. The corpus matrix and the identity table map in at its boundary. The quotient's distinct -//! matrix materializes into scratch, and the trunk's own staged writes and the seal complete -//! its variants. A stage never reads its own staged bytes back mid-run, so no map-back variant -//! exists for the values that flow as owned containers. The placement's deliberate re-reads of -//! persisted artifacts live in its own error. +//! stage type. The run carries the stage's enum here whole, and a failure therefore attributes +//! to its stage by construction. The enum carries no dataset or provider type, which makes it +//! `Send + 'static` by construction, and it crosses the rayon offload. The trunk's own failures +//! have their own variants: the corpus matrix and the identity table map in at its boundary, the +//! quotient's distinct matrix materializes into scratch, and the trunk's own staged writes and +//! the seal complete the set. The values that flow between stages as owned containers have no +//! map-back variant here. The placement is the one stage that maps its own staged artifacts back, +//! and those failures live in its own error. use core::{error::Error, fmt}; use std::io; @@ -28,10 +28,12 @@ use crate::{ salt::file::OpenVectorError, }; -/// A compute-side stage failed and published nothing. +/// A compute-side failure, attributed to the stage that produced it. /// -/// Every variant is dataset- and provider-free, so the whole enum is `Send + 'static` and -/// crosses the rayon offload boundary. +/// Every variant is dataset- and provider-free. The whole enum is therefore `Send + 'static` and +/// crosses the rayon offload boundary. Every variant but [`Seal`](Self::Seal) and +/// [`Offload`](Self::Offload) arises before the seal's rename and leaves nothing published. Those +/// two can also arise after the rename, with the generation directory already visible. #[derive(Debug)] pub(crate) enum ComputeError { /// The staged representation matrix failed to map in at the run's boundary. diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/landmark.rs b/libs/@local/graph/atlas/src/salt/fit/compute/landmark.rs index 2fe1d6453f7..afca7fb08b9 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/landmark.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/landmark.rs @@ -37,8 +37,8 @@ use crate::{ /// The landmark stage failed and staged no skeleton. /// -/// One variant per way the stage refuses, so a landmark failure attributes to this stage by -/// construction. The published failure surface speaks corpus rows. +/// One variant per way the stage refuses. A landmark failure therefore attributes to this stage +/// by construction. The published failure surface names corpus rows. #[derive(Debug)] pub(crate) enum LandmarkError { /// The landmark selection rejected its input. @@ -135,8 +135,9 @@ impl PriorMarks { /// /// # Errors /// - /// Returns a [`PriorError`] when a prior artifact does not map or the prior skeleton names a - /// row beyond the prior identity table. + /// Returns a [`PriorError`] when a prior artifact does not map, does not hold a valid + /// skeleton or identity table, or the prior skeleton names a row beyond the prior identity + /// table. #[tracing::instrument(name = "prior-translation", skip_all)] pub(super) fn translated( prior: &Generation, @@ -229,19 +230,20 @@ impl<'fit> LandmarkSurvey<'fit> { /// Selects, assigns, contracts, and lays out the landmark skeleton, and stages it. /// - /// Candidates are uniform over the distinct rows; the prior marks name the rows competing + /// Candidates are uniform over the distinct rows, and the prior marks name the rows competing /// for the retained share. The skeleton builds over the distinct representation rows and /// publishes over the corpus row domain: selected rows name their first corpus rows, and /// every corpus row takes its representative's landmark. It returns owned beside its typed - /// binding, so the placement stage reads the value this call built rather than the staged + /// binding, and the placement stage reads the value this call built rather than the staged /// bytes. /// /// # Errors /// /// Returns [`LandmarkError::Selection`] when the landmark selection rejects its input, - /// [`LandmarkError::Index`] when the assignment's search backend fails, - /// [`LandmarkError::Quotient`] when the graph contraction rejects its input, - /// [`LandmarkError::Layout`] when the layout rejects its input, and an I/O error when the + /// [`LandmarkError::Index`] when the assignment's search backend fails to construct, + /// [`LandmarkError::Assignment`] when the assignment fails, [`LandmarkError::Quotient`] when + /// the graph contraction rejects its input, [`LandmarkError::Layout`] when the layout rejects + /// its input, and an I/O error when the assignment scratch directory does not create or the /// staged skeleton does not write. #[tracing::instrument(name = "landmark-survey", skip_all)] pub(super) fn run( @@ -251,7 +253,7 @@ impl<'fit> LandmarkSurvey<'fit> { LandmarkError, > { let training = self.quotient.training(); - // The skeleton builds over distinct rows; the published failure surface speaks corpus + // The skeleton builds over distinct rows, and the published failure surface names corpus // rows. let corpus = |row: DistinctRowId| self.quotient.representative(row); @@ -313,9 +315,9 @@ impl<'fit> LandmarkSurvey<'fit> { )?; drop(contracted); - // Publication crosses back to the corpus row domain; the - // first-row map ascends strictly, so the selection's order and - // the assignment's ordinal vocabulary carry over unchanged. + // Publication crosses back to the corpus row domain. The first-row map ascends strictly, + // and the selection's order and the assignment's ordinal vocabulary therefore carry + // over unchanged. let skeleton = LandmarkSkeleton::new( selection.map_rows(|row| self.quotient.representative(row)), assignment.reindex(self.quotient.classes().iter().copied()), diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/lod.rs b/libs/@local/graph/atlas/src/salt/fit/compute/lod.rs index 5720f8bc3bd..138984b36b3 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/lod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/lod.rs @@ -28,11 +28,15 @@ use crate::{ /// The delivery stage failed and staged no column. /// -/// One variant per way the stage refuses, so a delivery failure attributes to this stage by -/// construction. +/// One variant per way the stage refuses. A delivery failure therefore attributes to this stage +/// by construction. #[derive(Debug)] pub(crate) enum DeliveryError { - /// The corpus exceeds the `u32` wire position encoding. + /// The rank inputs refused their columns. + /// + /// The corpus exceeds the `u32` wire position encoding, or the importance, priority and + /// identity columns disagree on length. The second case is a pipeline defect: all three + /// columns derive from one corpus. WireEncoding { rows: u64 }, /// The level-of-detail derivation rejected its input. Lod(LodError), @@ -97,9 +101,9 @@ impl Error for DeliveryError { /// The level-of-detail stage, bound to the values it derives from. /// -/// [`run`](Self::run) derives the delivery structure and [`Delivery::stage`] persists it, so the -/// computation and its artifacts separate: the stage consumes proven values, and only the staging -/// step touches the generation. +/// [`run`](Self::run) derives the delivery structure and [`Delivery::stage`] persists it. The +/// computation and its artifacts therefore separate: the stage consumes proven values, and only +/// the staging step touches the generation. pub(super) struct LevelOfDetail<'fit, I> { /// The canonical coordinates, proven finite at their readback. coordinates: &'fit FinitePointField, @@ -156,12 +160,10 @@ where RankingConfig::IncidentDegree => DegreeImportance::new(self.adjacency).derive(rows), }; - // The priority column is the rank inputs' product-override - // lane: a product-side boost (a pinned or promoted entity) - // will feed it the day one exists. Until then every row - // carries the neutral 0 and the column stays present, so the - // rank contract and the wire shape do not change when the - // signal arrives. + // The priority column is the rank inputs' product-override lane for a product-side boost (a + // pinned or promoted entity). No such signal exists, and every row carries the neutral 0. + // The column is present regardless, and the rank contract and the wire shape therefore do + // not depend on whether a boost signal exists. let priority = IdVec::from_domain(0.0_f32, &importance); let inputs = RankInputs::new(&importance, &priority, self.ids.ids()).ok_or_else(|| { DeliveryError::WireEncoding { @@ -265,20 +267,32 @@ impl Delivery { /// The staged level-of-detail files of one fit, each binding typed by its artifact. pub(super) struct LodArtifacts { + /// The Morton code column in base delivery order. pub morton: Binding, + /// The quadtree topology. pub quad: Binding, + /// The type postings over the base delivery order. pub postings: Binding, + /// The wire coordinate column in base delivery order. pub wire_coordinates: Binding, + /// Each base position's importance rank. pub rank_of_position: Binding, + /// Each rank's base position. pub position_of_rank: Binding, + /// Each node row's base position. pub position_of_row: Binding, + /// Each base position's node row. pub row_of_position: Binding, } /// The staged delivery structure, pairing the typed artifact set with every evidence section. pub(super) struct StagedDelivery { + /// The staged files, each binding typed by its artifact. pub files: LodArtifacts, + /// The delivery-order readings. pub evidence: LodMeasurements, + /// The quadtree readings. pub quad: QuadMeasurements, + /// The postings readings. pub postings: PostingsMeasurements, } diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/mod.rs b/libs/@local/graph/atlas/src/salt/fit/compute/mod.rs index 5f2c316ec57..787286a402d 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/mod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/mod.rs @@ -1,19 +1,42 @@ -//! The compute side of one fit, covering every stage after ingest. +//! Generation construction from ingested dataset artifacts. //! -//! [`Compute::run`] executes on the rayon pool, so the tokio runtime thread stays free while the -//! CPU-heavy stages - the neighbour link, the landmark layout, the level-of-detail sort - do -//! their work. Nothing here touches the dataset or the embedding provider: every input is a -//! staged file or a value carried across the boundary in [`Compute`], and every failure is a -//! [`ComputeError`]. +//! [`Compute::run`] builds the geometry and relation policy, then seals the staged files into a +//! published generation. Run it on a Rayon worker to keep its synchronous CPU and file work off +//! async executor threads. Its inputs consist of owned values and staged files, never a live +//! dataset or embedding provider. //! -//! Data flows through the run as owned values, and artifacts are its rims. The staged ingest -//! files map in once at the top - the corpus matrix, the identity table, the endpoint and card -//! columns - and every stage after that consumes the values the stages before it built: the -//! quotient carries both row-domain matrices, the admitted neighbour table feeds the semantic -//! smoothing, the skeleton and the trainer indexes feed the placement. A staging write returns -//! the repository binding the seal publishes, and nothing reads its own staged bytes back -//! mid-run. The deliberate exceptions live in the placement's measurement pass, which re-reads -//! persisted artifacts exactly because the published bytes are what its readings certify. +//! # Stage dependencies +//! +//! Placement combines semantic structure with relation constraints. [`classifier`] and [`policy`] +//! determine how to interpret relation types, and [`relation`] builds their attraction and +//! protection indexes. [`neighbours`] constructs and smooths the k-NN table. [`landmark`] derives a +//! layout from that semantic graph, retaining eligible landmarks from the prior generation. +//! +//! [`projector`] produces coordinates using either the trained model or the landmark baseline. +//! [`lod`] then derives the delivery order, spatial index and type postings from the coordinates +//! and corpus topology. +//! +//! # Row domains +//! +//! [`Quotient`] groups byte-identical representations into distinct rows for training. Training +//! indexes and the semantic graph use that distinct-row domain. The published neighbour and +//! semantic artifacts cover corpus rows addressed by [`NodeRowId`]. A stage's in-memory training +//! value can therefore differ from the corpus-domain artifact it stages. [`Staged`] keeps the value +//! and artifact binding separate. +//! +//! # Working data and staged files +//! +//! Representations and node identities remain mapped from the start of the run. Policy and relation +//! preparation open the card and endpoint columns, and verdict resolution opens the ontology +//! identities. +//! +//! The run retains computed products for dependent stages after writing their artifacts. Peak +//! memory includes these retained products alongside the current stage's working storage. +//! +//! Placement reopens its staged coordinates through [`coordinates::Coordinates::open`] to check +//! finiteness before deriving the delivery structure. Ladder measurements also read the staged +//! coordinate and attraction columns to measure the bytes that will publish. Temporary matrices and +//! ladder frames use [`Context::scratch`], separate from the generation's staged artifacts. #[cfg(test)] pub(super) use self::projector::error::ProjectorError; @@ -63,9 +86,7 @@ mod projector; mod quotient; mod relation; -/// The relation-policy classifier supply. -/// -/// Resolved on the async side and carried across the thread boundary. +/// A supplied relation classifier or the training material needed to fit one. pub(super) enum ClassifierPlan { /// Use a fitted model supplied to the run. Use { @@ -76,7 +97,7 @@ pub(super) enum ClassifierPlan { }, /// Fit a model from the assembled annotation corpus. Fit { - /// The assembled training and holdout material, boxed to keep the variants near one size. + /// The assembled training examples and holdout material. corpus: Box, /// The SHA-256 of the corpus document's bytes. source: Sha256Digest, @@ -85,42 +106,42 @@ pub(super) enum ClassifierPlan { }, } -/// The places and settings of one compute run. +/// Shared configuration and storage for one generation under construction. /// -/// Every stage reads the same staged generation, the same scratch directory, the same -/// configuration, and the same device. The run owns them for its whole life and consumes the -/// staged generation at the seal. +/// Stages write publishable artifacts into `staging` and temporary working files into `scratch`. +/// Sealing consumes the staged generation. Dropping the context attempts to remove remaining +/// staging and scratch files. pub(super) struct Context { - /// The staged generation every stage writes into and the seal consumes. + /// Artifacts awaiting publication as one generation. pub staging: StagedGeneration, - /// The scratch directory for artifacts that live and die with the run. + /// Temporary working files to remove when the run ends. pub scratch: ScratchDirectory, - /// The fit's configuration, echoed into the metadata. + /// Stage settings, also recorded in the generation metadata. pub config: FitConfig, - /// The device every tensor stage runs on. + /// The device for tensor computation. pub device: PhysicalDevice, } -/// One stage's product, pairing the owned value with its typed binding and its evidence. +/// A stage result with its artifact binding and recorded evidence. /// -/// The value flows to the stages downstream, while the binding and the evidence flow to the -/// seal, so everything one stage produced travels as one typed unit until the trunk routes its -/// parts. The value and the binding need not describe the same bytes: under a real quotient the -/// neighbour and semantic stages publish the corpus-domain table while the distinct-domain twin -/// flows on as the value the trainer consumes. +/// `value` supports further computation. `binding` identifies the staged artifact, and `evidence` +/// records the stage's measurements for the repository metadata. +/// +/// The value need not be a decoded copy of the artifact. For example, semantic smoothing returns a +/// distinct-row training graph while staging the corpus-domain graph when [`Quotient`] groups +/// duplicate representations. pub(super) struct Staged { - /// The owned value the next stages consume. + /// The in-memory result available to later stages. pub value: V, /// The staged file's typed repository binding. pub binding: Binding, - /// The stage's measurement, echoed into the metadata document. + /// The stage's evidence for the metadata document. pub evidence: E, } -/// One fit's compute run, holding the owned inputs the async side hands across the thread -/// boundary. +/// Owned inputs for completing a fit after dataset ingestion. pub(super) struct Compute { - /// The run's places and settings. + /// Storage, configuration and device shared by the stages. pub context: Context, /// A fitted model, or the assembled corpus to fit one from. pub classifier: ClassifierPlan, @@ -130,21 +151,34 @@ pub(super) struct Compute { pub verdicts: Option, /// The generation seeding reuse, when the fit received one. pub prior: Option, - /// The staged stream artifacts and drain facts of the ingest. + /// Ingested artifact bindings, row-aligned type columns and input measurements. pub ingested: Ingested, } impl Compute { - /// Runs every compute stage over the staged ingest artifacts and seals the generation. + /// Builds the remaining artifacts and publishes the completed generation. + /// + /// `I` and `O` must be the ingested dataset's node and ontology identity types. Node identities + /// support prior-landmark translation and ranking tiebreaks. Ontology identities resolve the + /// supplied verdicts against the staged ontology table. /// - /// `I` is the dataset's node id type: the identity artifacts open under it for - /// prior-landmark translation and the ranking tiebreak. `O` is the dataset's ontology id - /// type, under which the supplied verdicts resolve. + /// Success returns a durable [`PublishedGeneration`]. Publication does not activate it for + /// serving. /// /// # Errors /// - /// Returns the failing stage's [`ComputeError`]. The staging and scratch directories remove - /// themselves on the early return, so a failed run publishes nothing. + /// Returns [`ComputeError`] for a failed stage, artifact operation or seal. Every error before + /// the seal's rename leaves nothing published. A [`ComputeError::Seal`] from opening or syncing + /// the root after the rename leaves the generation directory visible. + /// + /// On an early return, the staging and scratch directories attempt cleanup and log any cleanup + /// failures. + /// + /// # Panics + /// + /// Propagates panics from compute stages, including the input and scratch-file conditions of + /// [`Quotient::build`] and [`PlacementPass::run`]. A `progress` callback can also panic, + /// including the seal-completion callback after publication. #[expect( clippy::too_many_lines, reason = "the run is the fit's one straight line, and splitting it would scatter the data \ @@ -165,9 +199,8 @@ impl Compute { ingested, } = self; - // The run maps in two boundary artifacts. The corpus matrix's file is the data's home - // for the whole run, and the identity table feeds the prior translation and the ranking - // tiebreak. + // retain the corpus mapping while later stages borrow its representations. Node identities + // translate prior landmarks and break ranking ties. let corpus: VectorFile = VectorFile::open(context.staging.path_of(&artifact::Representations::NAME)) .map_err(ComputeError::OpenRepresentations)?; @@ -219,8 +252,8 @@ impl Compute { .run()?; progress.stage_completed(Stage::Landmarks); - // Built ahead of placement: the paired-movement draw derives its salt from these exact - // values, and the seal serializes the same ones. + // construct these before placement: paired-movement sampling derives its salt from the same + // snapshot and configuration that the metadata records. let dataset = ingested.origin; let snapshot = ingested.snapshot(); let reproducibility = ingested.reproducibility(context.config.clone(), prior.as_ref()); @@ -255,8 +288,8 @@ impl Compute { .stage(&context.staging)?; progress.stage_completed(Stage::Lod); - // Each typed binding and evidence value enters the repository exactly once at the seal, - // and the sealed document is the generation's identity. + // the repository records both artifact digests and fit evidence. Its serialized bytes + // determine the generation ID. let (annotation_corpus, annotation_embeddings, annotation_hashes) = acquired .annotation .map_or((None, None, None), |annotation| { diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/neighbours.rs b/libs/@local/graph/atlas/src/salt/fit/compute/neighbours.rs index a3206482e4c..27ca1e0c2fa 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/neighbours.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/neighbours.rs @@ -1,5 +1,7 @@ -//! The neighbour stage constructs, admits, and publishes the k-NN table, and smooths it into -//! the semantic graphs. +//! The neighbour stage over the k-NN table and the semantic graphs smoothed from it. +//! +//! The neighbour stage constructs, admits, and publishes the k-NN table, and smooths it into the +//! semantic graphs. use core::{error::Error, fmt}; use std::io; @@ -29,8 +31,8 @@ use crate::{ /// The neighbour stage failed and staged no table. /// -/// One variant per way the stage refuses, so a neighbour failure attributes to this stage by -/// construction. The published failure surface speaks corpus rows. +/// One variant per way the stage refuses. A neighbour failure therefore attributes to this stage +/// by construction. The published failure surface names corpus rows. #[derive(Debug)] pub(crate) enum NeighbourError { /// The search backend failed. @@ -124,12 +126,12 @@ impl<'fit> NeighbourAdmission<'fit> { /// Constructs and admits the neighbour lists, then publishes the corpus table. /// - /// Exact recall admits the lists. One construction runs at the wider of the spot check's - /// depth and the stored width, so the admitted lists and the training table are the same - /// lists. The published table covers the corpus row domain: under a real quotient every row - /// takes its representative's list, and under the identity the admitted table is already the - /// corpus's own. The admitted table speaks distinct rows, and the published failure surface - /// speaks corpus rows. + /// A recall spot check against exact rankings admits the lists. One construction runs at the + /// wider of the spot check's depth and the stored width, and the admitted lists and the + /// training table are therefore the same lists. The published table covers the corpus row + /// domain: under a real quotient every row takes its representative's list, and under the + /// identity the admitted table is already the corpus's own. The admitted table names distinct + /// rows, and the published failure surface names corpus rows. /// /// # Errors /// @@ -145,7 +147,7 @@ impl<'fit> NeighbourAdmission<'fit> { P: Progress + Sync, { let training = self.quotient.training(); - // Construction speaks distinct rows. The published failure surface speaks corpus rows. + // Construction names distinct rows. The published failure surface names corpus rows. let corpus = |row: DistinctRowId| self.quotient.representative(row); let width = self @@ -187,9 +189,8 @@ impl<'fit> NeighbourAdmission<'fit> { }) .map_err(|error| error.map_rows(corpus, |fault| fault.map_rows(corpus)))?; - // The report comes before the admission decision. The run measured a construction - // the floor rejects, and that measurement is what an operator is - // watching for. + // reporting before the admission decision lets the operator observe the recall measurement + // even when the floor rejects the construction. progress.knn_recall(&recall); match recall.admission() { @@ -234,11 +235,11 @@ impl<'fit> NeighbourAdmission<'fit> { } } -/// The corpus-domain expansion, waiting for the smoothing that spends it. +/// The corpus-domain expansion, waiting for the smoothing that consumes it. /// -/// Under a real quotient the carrier holds the expanded corpus-domain table; under the identity +/// Under a real quotient the carrier holds the expanded corpus-domain table. Under the identity /// the admitted table is already the corpus's own and the carrier is empty. The smoothing -/// consumes the carrier by value, so a second smoothing does not compile. +/// consumes the carrier by value, and a second smoothing therefore does not compile. pub(super) struct Expansion(Option>); /// The corpus's admitted neighbourhood structure. @@ -248,7 +249,7 @@ pub(super) struct Expansion(Option>); pub(super) struct Neighbourhood { /// The admitted distinct-domain table, the trainer's. pub admitted: Knn, - /// The passed recall spot check, echoed into the metadata. + /// The recall spot check the table was admitted under, echoed into the metadata. pub recall: RecallSpotCheck, /// The published table's typed binding. pub binding: Binding, @@ -259,9 +260,9 @@ impl Neighbourhood { /// /// The published graph weighs the corpus-domain table, and the trainer's graph weighs the /// distinct table by the same kernel. Under the identity quotient the two domains are the - /// same rows in the same order, so one graph serves both: it stages as the published - /// artifact and returns as the training graph. The expansion arrives by value and does not - /// survive the call. + /// same rows in the same order, and one graph therefore serves both: it stages as the + /// published artifact and returns as the training graph. The expansion is consumed by value + /// and does not survive the call. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/policy.rs b/libs/@local/graph/atlas/src/salt/fit/compute/policy.rs index 2282d572283..63b4b5ab098 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/policy.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/policy.rs @@ -24,7 +24,7 @@ use crate::{ /// The policy stage failed and staged no table. /// -/// One variant per way the stage refuses, so a policy failure attributes to this stage by +/// One variant per way the stage refuses. A policy failure therefore attributes to this stage by /// construction. #[derive(Debug)] pub(crate) enum PolicyError { @@ -108,10 +108,10 @@ impl<'fit> PolicyResolution<'fit> { /// Classifies every relation type's card and resolves the policy table. /// /// The relation universe is the distinct ontology rows the edge stream carried. Each indexes - /// the staged card table, which is row-aligned with the type table. Every card exists, so - /// every relation classifies. The resolved table returns owned and certified beside its - /// staged binding, so the relation stage consumes the certified value rather than the staged - /// bytes. + /// the staged card table, which is row-aligned with the type table. Every card exists, and + /// every relation therefore classifies. The resolved table returns owned and certified beside + /// its staged binding, and the relation stage consumes the certified value rather than the + /// staged bytes. /// /// # Errors /// @@ -123,8 +123,8 @@ impl<'fit> PolicyResolution<'fit> { pub(super) fn run( self, ) -> Result, PolicyError> { - // The staged card table is row-aligned with the type table, so its rows index by - // ontology row: the handle's id domain makes that claim once. + // The staged card table is row-aligned with the type table, and its rows therefore index + // by ontology row: the handle's id domain makes that claim once. let cards: VectorFile = VectorFile::open( self.context .staging diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/projector/error.rs b/libs/@local/graph/atlas/src/salt/fit/compute/projector/error.rs index 8c1d0146e8b..d686454f814 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/projector/error.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/projector/error.rs @@ -1,10 +1,10 @@ //! The placement stage's failure surface. //! //! [`ProjectorError`] holds every failure the placement stage produces, and nothing any other -//! stage can reach: the trunk widens it once at the stage boundary, so a signature naming this -//! type states exactly which failures its caller can observe. Like the trunk's error it carries -//! no dataset or provider type, so it is `Send + 'static` by construction and crosses the rayon -//! offload. +//! stage can reach. The trunk widens it once at the stage boundary, and a signature naming this +//! type therefore states exactly which failures its caller can observe. Like the trunk's error it +//! carries no dataset or provider type, which makes it `Send + 'static` by construction, and it +//! crosses the rayon offload. use core::{error::Error, fmt}; use std::io; @@ -24,7 +24,7 @@ use crate::{ }, }; -/// The placement stage failed to bind, train, measure, or publish. +/// A placement-stage failure, in binding, training, measurement or publication. /// /// Every variant aborts the fit: a generation whose coordinates the configured placement could /// not produce publishes nothing. @@ -44,12 +44,15 @@ pub(crate) enum ProjectorError { Ladder(LadderError), /// The ladder rejects the configured canonical condition. Canonical(CanonicalError), - /// A persisted coordinate column failed to map back for its measurement. + /// A staged coordinate column failed to map back as a finite point field. + /// + /// Every placement plan reopens the column it staged, and the ladder pass reopens it again for + /// its persisted-loss measurement. OpenCoordinates(OpenCoordinatesError), /// The canonical step's aligned frame has a non-finite point. /// - /// The alignment onto the baseline basis runs in `f32` and can overflow, so the aligned - /// frame is proven at its creation before it publishes, and the persisted coordinate + /// The alignment onto the baseline basis runs in `f32` and can overflow. The aligned frame + /// is therefore proven at its creation before it publishes, and the persisted coordinate /// column re-proves the same frame at its own readback. NonFiniteAligned { /// The first offending row. @@ -59,7 +62,7 @@ pub(crate) enum ProjectorError { OpenAttraction(OpenAttractionError), /// The paired-movement salt preimage failed to serialize. /// - /// The preimage is a strict subset of the metadata document, so the seal would refuse the + /// The preimage is a strict subset of the metadata document, and the seal would refuse the /// same generation. SaltPreimage(EncodeError), /// A placement artifact failed to write or persist. diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/projector/inputs.rs b/libs/@local/graph/atlas/src/salt/fit/compute/projector/inputs.rs index c7ad540b2aa..5c2e26473dc 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/projector/inputs.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/projector/inputs.rs @@ -36,8 +36,8 @@ pub(crate) struct PlacementInputs<'fit> { pub resolution: &'fit VerdictResolution, /// The metadata document's `snapshot` section, the value the seal serializes. /// - /// With [`Self::reproducibility`] it forms the paired-movement salt preimage, so the - /// readout's draw replays from the published document's input sections alone. + /// With [`Self::reproducibility`] it forms the paired-movement salt preimage. The readout's + /// draw therefore replays from the published document's input sections alone. pub snapshot: &'fit Snapshot, /// The metadata document's `reproducibility` section, the value the seal serializes. pub reproducibility: &'fit Reproducibility, @@ -48,8 +48,8 @@ pub(crate) struct PlacementInputs<'fit> { /// The trainer's distinct-row view of the corpus. /// /// Training and the ladder's loss measurements run over the quotient and the values built on it, -/// where byte-identical rows are one point. Publication evaluates the full corpus, and identical -/// representations project identically, so the two domains describe one field. +/// where byte-identical rows are one point. Publication evaluates the full corpus. Identical +/// representations project identically, and the two domains therefore describe one field. pub(crate) struct DistinctInputs<'fit> { /// The corpus-to-distinct row quotient, carrying both row domains' matrices. pub quotient: &'fit Quotient<'fit, PROJECTOR_DIMENSIONS>, @@ -63,8 +63,8 @@ pub(crate) struct DistinctInputs<'fit> { /// The training-domain views the publish half reads. /// -/// The quotient, the neighbour table, and the attraction index carry the ladder's per-level loss -/// measurements over the distinct rows. The metadata document's input sections ride beside them +/// The quotient, the neighbour table, and the attraction index carry the ladder's per-step loss +/// measurements over the distinct rows. The metadata document's input sections accompany them /// as the paired-movement salt preimage. pub(crate) struct PublishInputs<'fit> { /// The corpus-to-distinct row quotient. @@ -75,8 +75,8 @@ pub(crate) struct PublishInputs<'fit> { pub attraction: &'fit AttractionIndex, /// The metadata document's `snapshot` section, the value the seal serializes. /// - /// With [`Self::reproducibility`] it forms the paired-movement salt preimage, so the - /// readout's draw replays from the published document's input sections alone. + /// With [`Self::reproducibility`] it forms the paired-movement salt preimage. The readout's + /// draw therefore replays from the published document's input sections alone. pub snapshot: &'fit Snapshot, /// The metadata document's `reproducibility` section, the value the seal serializes. pub reproducibility: &'fit Reproducibility, @@ -94,9 +94,9 @@ pub(crate) struct VerdictResolution { impl VerdictResolution { /// Resolves the supplied verdicts against the staged ontology identity column. /// - /// Typed by the dataset's own ontology id, and addressed by the binding the ingest minted - /// for the column, so the resolution reads exactly the staged entry the seal publishes. A - /// run without supplied verdicts resolves to the empty resolution. + /// The column is typed by the dataset's own ontology id and addressed through the binding the + /// ingest created for it. The resolution therefore reads exactly the staged entry the seal + /// publishes. A run without supplied verdicts resolves to the empty resolution. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/projector/mod.rs b/libs/@local/graph/atlas/src/salt/fit/compute/projector/mod.rs index 5ec95d17cb1..fad11535ff5 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/projector/mod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/projector/mod.rs @@ -1,15 +1,17 @@ //! The placement stage runs either the trained projector or the landmark baseline. //! -//! The stage is the one owner of a fit's coordinates. Under the baseline placement every row takes -//! its assigned landmark's layout coordinate. Under the projector placement the stage trains the -//! conditioned model over the staged artifacts and stages its checkpoint. It projects the whole -//! corpus at every ladder level and measures the ladder. It publishes the canonical level's field -//! aligned into the baseline frame. The metadata records which placement ran and, for a trained -//! one, the training and ladder measurements. +//! The stage owns a fit's coordinates. Under the baseline placement every row takes its assigned +//! landmark's layout coordinate. Under the projector placement the stage trains the conditioned +//! model over the staged artifacts and stages its checkpoint. When the trained boundary supplies a +//! relation energy, the stage projects the whole corpus at every ladder step and publishes the +//! canonical step's field aligned into the baseline frame. Otherwise it publishes the +//! zero-condition field without alignment or ladder measurements. The metadata records which +//! placement ran and the available training and ladder evidence. //! -//! Ladder frames are transient: each projects into the run's scratch directory and maps back for -//! measurement, so the stage's owned working set stays one frame regardless of the schedule -//! length. Only the canonical aligned column publishes - version 1 publishes one variant. +//! Writing each ladder frame to scratch avoids retaining an owned corpus frame for every condition. +//! The pass retains all step mappings and loss readouts through measurement. Each ladder step also +//! allocates a corpus frame and a gathered distinct-row frame. Publication includes only one +//! coordinate column. use burn::module::AutodiffModule as _; use hashql_core::id::IdVec; @@ -63,7 +65,7 @@ pub(super) struct Placement { pub checkpoint: Option>, /// Which placement ran. pub kind: metadata::Placement, - /// The training and ladder measurements of a trained placement. + /// The evidence of a trained placement, with ladder measurements when relation energy exists. pub evidence: Option, } @@ -72,7 +74,7 @@ pub(super) struct Placement { enum Plan<'fit> { /// Every row takes its assigned landmark's layout coordinate. Baseline, - /// Train the conditioned projector and publish the canonical level's aligned field. + /// Train the conditioned projector and publish its coordinate column. Projector { /// The validated projector configuration. options: &'fit ProjectorOptions, @@ -83,9 +85,9 @@ enum Plan<'fit> { /// The placement process of one fit. /// -/// Construction resolves the configured plan and owns every configuration refusal, so a -/// contradictory configuration refuses before any placement span opens. [`run`](Self::run) -/// executes the resolved plan and stages the [`Placement`] coordinates. +/// Construction resolves the configured plan and checks its representation width, canonical +/// membership and affinity curve before any placement span opens. [`run`](Self::run) executes the +/// resolved plan and stages the [`Placement`] coordinates. pub(super) struct PlacementPass<'fit> { /// The stage's staging, scratch, configuration, and device. context: &'fit Context, @@ -98,12 +100,12 @@ pub(super) struct PlacementPass<'fit> { impl<'fit> PlacementPass<'fit> { /// Resolves the configured placement plan. /// - /// Owns every configuration refusal. Each fires before the first placement span opens; a - /// run that could not start never reaches a span. + /// Checks representation width, canonical membership and the affinity curve before the first + /// placement span opens. /// /// # Errors /// - /// Returns a [`ProjectorError`] when the projector configuration refuses. + /// Returns [`ProjectorError`] when the placement-plan configuration is invalid. pub(super) fn new( context: &'fit Context, inputs: &'fit PlacementInputs<'fit>, @@ -121,9 +123,8 @@ impl<'fit> PlacementPass<'fit> { return Err(ProjectorError::RepresentationWidth { configured }); } - // The canonical level's membership in the schedule is decidable from the options alone, - // so a contradictory configuration refuses here rather than after training runs the - // schedule and every level projects. + // canonical membership depends only on the options. Checking it before training avoids + // running the schedule and projecting every step for an invalid selection. options.ladder.canonical_index()?; let Some(affinity) = AffinityEnergy::new(context.config.curve, options.affinity_offset) @@ -142,14 +143,19 @@ impl<'fit> PlacementPass<'fit> { /// Places the corpus under the resolved plan, staging the canonical coordinates. /// - /// The trained placement is the run's only long loop with a per-iteration reading, so it - /// reports every training step to `progress`, while the baseline places in one pass and - /// reports nothing but its stage completion. + /// The trained placement reports every training step to `progress`. The baseline places in one + /// pass without per-row progress reports. /// /// # Errors /// - /// Returns an error of the training loop translated onto corpus rows when training fails, - /// and an I/O error when a staged output does not write or map back. + /// Returns [`ProjectorError`] for training, checkpoint encoding, projection, ladder measurement + /// or staged-artifact failures, translating training errors onto corpus rows. + /// + /// # Panics + /// + /// Panics on unrepresentable coefficient normalization, non-finite relation-loss readouts, or + /// failed scratch-frame readback as finite `f32` pairs. Inconsistent row domains can also panic + /// when accessed. pub(super) fn run(self, progress: &P) -> Result { match self.plan { Plan::Baseline => self.baseline(), @@ -162,6 +168,10 @@ impl<'fit> PlacementPass<'fit> { /// Every corpus row takes its assigned landmark's layout coordinate as the baseline /// placement, gathered into one owned column and staged as the `f32[N, 2]` coordinate /// artifact. + /// + /// # Errors + /// + /// Returns [`ProjectorError`] when staging or opening the coordinate column fails. #[tracing::instrument(name = "baseline-placement", skip_all)] fn baseline(self) -> Result { let skeleton = self.inputs.skeleton; @@ -187,9 +197,14 @@ impl<'fit> PlacementPass<'fit> { }) } - /// Trains the projector and publishes the canonical level's aligned field. + /// Trains the projector and stages its checkpoint and coordinate column. + /// + /// [`Self::publish`] selects the aligned canonical field or the zero-condition field from the + /// trained boundary evidence. + /// + /// # Errors /// - /// The checkpoint stages beside it. + /// Returns [`ProjectorError`] when training or publication fails. #[tracing::instrument(name = "projector-placement", skip_all)] fn projector( self, @@ -204,9 +219,9 @@ impl<'fit> PlacementPass<'fit> { let corpus = distinct.quotient.corpus(); let training = distinct.quotient.training(); - // Every corpus row is a knowledge entity: the dataset streams entities, and no other role - // projects yet. Each domain's uniform column is born in its own row domain, so neither view - // relabels the other's rows. + // every corpus row is a knowledge entity: the dataset streams entities. Each uniform role + // column uses its corresponding row domain, preserving the distinction between corpus and + // training rows. let corpus_roles = IdVec::from_elem(NodeRole::KnowledgeEntity, corpus.len()); let training_roles = IdVec::from_elem(NodeRole::KnowledgeEntity, training.len()); let landmarks = SupportAnchor::at_landmarks( @@ -224,9 +239,10 @@ impl<'fit> PlacementPass<'fit> { roles: &training_roles, }; - // A vacuous placement withholds the relation evidence. The trainer sees no force at all, so - // no radius freezes and the trainer demands no reviewed verdicts, while the published - // relation artifacts stay real for serving. + // a vacuous placement supplies no attraction force to the trainer. It requires no + // reviewed-Proximal verdict and freezes no relation radius. Semantic, protection and + // support inputs remain available, and the published relation artifacts still describe the + // corpus. let vacuous = AttractionIndex::vacuous(); let attraction = if options.vacuous { tracing::info!("vacuous attraction select. no attraction term will be used"); @@ -243,21 +259,21 @@ impl<'fit> PlacementPass<'fit> { knn: distinct.knn.view(), columns: trainer_columns, landmarks: &landmarks, - // No stage supplies temporal anchors, so the pool is empty. + // no fit stage supplies temporal anchors. The empty pool disables temporal support. anchors: &[], verdicts: &self.inputs.resolution.resolved, - // The released configuration trains no target objective: neither the declared - // constants nor the draws exist here. + // this placement trains without a target objective. target: None, }; - // The configured coefficients are corpus-free bases. The semantic and ordinary bases divide - // by the total semantic edge weight, the hard-negative base by the row count, and the - // support bases by the pools the trainer receives, so each base weighs the same objective - // share on every corpus. The relation base is already mass-free. The masses are the - // training domain's - the distinct rows and their graph - matching the objective the - // trainer optimizes. A weightless graph passes the bases through - the trainer rejects it - // as evidence-free immediately after. + // the configured coefficients are corpus-free bases. The semantic and ordinary bases divide + // by the total semantic edge weight, the hard-negative base by the row count, and each + // support base by its own pool size. This normalization removes each objective family's + // mass from its configured weight. The relation base is already mass-free. The masses + // belong to the training domain: the distinct rows and their graph, matching the objective + // the trainer optimizes. With positive semantic mass, an empty support pool receives a zero + // coefficient. A weightless graph passes every base through for the trainer to reject as + // evidence-free. let coefficients = options.coefficients.normalized( distinct.semantic.view().total_weight(), training.len(), @@ -288,16 +304,22 @@ impl<'fit> PlacementPass<'fit> { /// Stages everything the projector placement publishes. /// - /// The publish half of the placement reads a model and its frozen boundary evidence. It - /// stages the checkpoint and the canonical coordinate column beside the placement's evidence. A - /// boundary whose radius composes a relation energy opens the ladder, and the canonical level's - /// aligned field publishes ([`LadderPass::measure_conditions`]). + /// A boundary whose radius composes a relation energy enables + /// [`LadderPass::measure_conditions`], which stages the canonical level's aligned field. + /// Otherwise the coordinate column contains the zero-condition field without alignment. The + /// checkpoint and the available evidence accompany either field. + /// + /// # Errors + /// + /// Returns [`ProjectorError`] for checkpoint encoding, projection, ladder measurement or + /// staged-artifact failures. fn publish( &self, options: &ProjectorOptions, model: Model, columns: NodeColumns<'_, NodeRowId>, ) -> Result { + // retain the inference model before consuming the training model to record its checkpoint. let projector = model.projector.valid(); let checkpoint = self.checkpoint(model.projector)?; @@ -369,8 +391,11 @@ impl<'fit> PlacementPass<'fit> { /// Runs the training loop under its own span. /// - /// Training speaks distinct rows, and its errors translate through the quotient onto - /// corpus rows. + /// Training uses distinct rows, and its errors translate through the quotient onto corpus rows. + /// + /// # Errors + /// + /// Returns [`ProjectorError`] when the training loop fails. #[tracing::instrument(name = "projector-training", skip_all)] fn train( &self, @@ -398,8 +423,8 @@ impl<'fit> PlacementPass<'fit> { let model = match outcome { train::FitOutcome::Trained(model) => model, - // This stage constructs no target inputs, and only a declared target objective can - // refuse. + // only a declared target objective can refuse. This stage supplies no target inputs. + // Therefore this outcome is unreachable. train::FitOutcome::TargetRefused(refusal) => { unreachable!("no target objective is declared, yet training refused: {refusal}") } diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/projector/report.rs b/libs/@local/graph/atlas/src/salt/fit/compute/projector/report.rs index 2430d678b8b..ba361039409 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/projector/report.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/projector/report.rs @@ -1,5 +1,7 @@ -//! Measurements over the staged placement: the ladder readings, loss regressions, and -//! paired-movement evidence. +//! Measurements over the staged placement. +//! +//! They are the ladder readings and loss regressions, with the paired-movement evidence beside +//! them. use core::num::NonZero; use std::{fs::File, io::Write as _}; @@ -71,9 +73,18 @@ impl<'fit> LadderPass<'fit> { /// Projects, measures, and publishes the condition ladder, returning its evidence. /// - /// Every step projects into the scratch directory and maps back. The canonical step's field - /// aligns into the baseline frame and stages as the coordinate column, and the relation loss - /// re-measures over the persisted bytes. + /// Every step projects once. Its relation loss measures over the owned frame, and the frame + /// then persists into the scratch directory and maps back for the alignment fits. The + /// canonical step's mapped field aligns into the baseline frame and stages as the coordinate + /// column, and the relation loss re-measures over the persisted column. + /// + /// # Errors + /// + /// Returns [`ProjectorError`] when the scratch directory or a step file fails to write, a + /// projection or local-scale pass fails, the ladder measurement or the canonical selection + /// refuses, the aligned frame has a non-finite point, staging fails, the persisted column + /// fails to map back, or the paired-movement readout fails to open the staged index or to + /// serialize its salt preimage. #[tracing::instrument(skip_all)] pub(super) fn measure_conditions( &self, @@ -93,8 +104,8 @@ impl<'fit> LadderPass<'fit> { let frame = refresh::forward(model, columns, eta, options.forward_rows, self.device)?; // The loss population is the training domain: the full frame gathers at the quotient's - // first rows - identical representations project identically, so the gather is the - // distinct rows' own frame. + // first rows. Identical representations project identically, and the gather is + // therefore the distinct rows' own frame. let distinct_frame = inputs.quotient.training_frame(&frame); let scales = refresh::scales(&distinct_frame, &inputs.knn, eta) .map_err(|error| error.map_rows(|row| inputs.quotient.representative(row)))?; @@ -189,8 +200,14 @@ impl<'fit> LadderPass<'fit> { /// Re-measures the relation loss over the persisted aligned column. /// - /// The narrowing to `f32` and the alignment application are inside the measurement, ahead of - /// the same distinct gather the step losses used, so the reading guards both. + /// The column carries the alignment's narrowed `f32` coefficients applied row by row. The + /// reading maps the column back, gathers the distinct rows as the step losses did, and + /// measures the result. It therefore covers the alignment and the narrowing together. + /// + /// # Errors + /// + /// Returns [`ProjectorError`] when the column fails to map back, holds a non-finite point, or + /// its local-scale pass fails. fn measure_persisted_loss( &self, inputs: &PublishInputs<'_>, @@ -219,25 +236,27 @@ impl<'fit> LadderPass<'fit> { /// Measures the paired-movement readout over the staged attraction index. /// - /// The index maps back from its staged bytes rather than riding in from the stage that built - /// it, so the readout replays from exactly what the published generation carries. + /// The readout opens the index from its staged bytes rather than taking the in-memory index + /// of the stage that built it. It therefore replays from exactly what the published + /// generation carries. /// - /// The readings run over the ladder's aligned frames: `zero` is the baseline step's field - /// and `canonical` the published step's field in the baseline basis. [`paired::measure`] - /// runs the whole readout, and every readout resolution is an evidence body, so the - /// generation publishes around a vacuous or failed reading. + /// The readings run over two frames in the baseline basis: `zero` is the baseline step's + /// field and `canonical` the published step's aligned field. [`paired::measure`] runs the + /// whole readout. Every readout resolution is an evidence body, and the generation publishes + /// around a vacuous or failed reading. /// /// # Errors /// /// - [`ProjectorError::SaltPreimage`] when the salt preimage does not serialize. The preimage - /// is a strict subset of the metadata document, so the seal shares the failure. + /// is a strict subset of the metadata document, and the seal would refuse the same + /// generation. /// - [`ProjectorError::OpenAttraction`] when the staged attraction index does not map back. /// /// # Panics /// /// This panics when the staged index and the ladder frames disagree on the corpus row count. - /// One fit stages both over one corpus, so the disagreement is a pipeline defect rather than - /// a data condition, and no persisted refusal names it. + /// One fit stages both over one corpus. The disagreement is therefore a pipeline defect + /// rather than a data condition, and no persisted refusal names it. #[expect( clippy::panic_in_result_fn, reason = "the Result carries fit-level failures; a row-count contradiction between two \ @@ -271,9 +290,11 @@ impl<'fit> LadderPass<'fit> { /// One step's frame, persisted in the ladder's scratch directory. /// -/// The handle is the step's persisted-bytes contract: the alignment fits and the loss readings -/// map the written file back rather than reading the owned frame, so every downstream reading -/// measures exactly what the scratch file carries. +/// The handle is the step's persisted-bytes contract. The alignment fits, the canonical field's +/// alignment and the paired-movement baseline map the written file back rather than reading the +/// owned frame, and each of those readings therefore measures exactly what the scratch file +/// carries. The step's relation loss is the one reading taken from the owned frame, ahead of the +/// write. struct StepFrame(FinitePointFile); impl StepFrame @@ -281,6 +302,17 @@ where N: Id, { /// Persists one step's frame as a scratch array file. + /// + /// # Errors + /// + /// Returns [`ProjectorError::Io`] when creating, writing, flushing or syncing the file fails. + /// + /// # Panics + /// + /// This panics when the flushed file does not open, or opens as something other than a finite + /// `f32` point file. The write that just completed is the file's only author: a shape or + /// finiteness mismatch is a mapping defect rather than a data condition, while the open itself + /// can fail on the scratch file system after a complete write. fn make( ladder: &Utf8Path, index: usize, @@ -310,6 +342,7 @@ where Ok(Self(file)) } + /// Returns the frame's coordinates as a finite point field. fn field(&self) -> &FinitePointField { &self.0 } @@ -345,10 +378,15 @@ pub(super) struct LossSeries<'schedule> { impl<'schedule> LossSeries<'schedule> { /// Pairs the schedule's conditions with their measured losses. /// + /// The constructor checks the two lengths and nothing else: it accepts empty slices and a + /// first condition other than zero. [`baseline`](Self::baseline) reads the first loss as the + /// zero-condition reading, and the caller supplies a nonempty schedule whose first condition + /// is zero. + /// /// # Panics /// - /// This panics when the condition and loss counts disagree: one fit measures one loss per - /// condition, so a mismatch is a pipeline defect rather than a data condition. + /// This panics when the condition and loss counts disagree. One fit measures one loss per + /// condition, and a mismatch is therefore a pipeline defect rather than a data condition. pub(super) fn new(conditions: &'schedule [NonNegative], losses: Vec) -> Self { assert_eq!( conditions.len(), @@ -411,10 +449,16 @@ impl<'schedule> LossSeries<'schedule> { } } - /// The zero-condition raw loss. + /// Returns the first loss as the zero-condition raw loss. + /// + /// The reading is the zero-condition one when the series came from a ladder schedule, whose + /// first step is bit-exactly `0.0` by construction + /// ([`Conditions::new`](crate::salt::ladder::Conditions::new)). The constructor does not + /// check that. + /// + /// # Panics /// - /// The schedule's first step is bit-exactly `0.0` by construction, so the first loss is the - /// zero-condition reading. + /// This panics on an empty series. const fn baseline(&self) -> DNonNegative { self.losses[0] } @@ -429,8 +473,9 @@ impl<'schedule> LossSeries<'schedule> { return; } - // In domain with no check: the guard proves the difference non-negative, and the - // difference of two finite values is finite. + // The difference of two non-negative finite values has magnitude at most the larger value. + // The guard establishes `persisted` ≥ `baseline`, making this difference non-negative. + // Therefore the difference is finite and in domain without another check. let delta = DNonNegative::new_unchecked((persisted - baseline).get()); let relative = baseline .positive() @@ -464,13 +509,13 @@ pub(super) struct RelationLossReadout { /// The capped trained estimand. /// /// Each group's share enters scaled by `min(cap, n) / n` and folds in its own accumulation - /// chain, so the reading is the exact expectation of the trainer's capped-sampling batch + /// chain. The reading is the exact expectation of the trainer's capped-sampling batch /// estimator. pub capped_total: DNonNegative, /// Each group's own accumulated share, in the index's group order (ascending by relation). /// - /// The shares carry their own accumulation chains, so their sum matches the uncapped total - /// to rounding rather than bit-exactly. The uncapped total's own chain is the persisted + /// The shares carry their own accumulation chains. Their sum therefore matches the uncapped + /// total to rounding rather than bit-exactly. The uncapped total's own chain is the persisted /// contract. pub per_type: Vec<(OntologyRowId, DNonNegative)>, } @@ -482,8 +527,8 @@ impl RelationLossReadout { /// distance, accumulated in double precision - one accumulator for the corpus, one per group, /// and one for the capped estimand, all in the same walk. The capped accumulator scales each /// group's finished share by `min(cap, n) / n`, the probability that one of the group's `n` - /// edges enters the trainer's per-type draw, so the reading is the exact expectation of the - /// capped-sampling batch estimator the trainer optimizes. + /// edges enters the trainer's per-type draw. The reading is therefore the exact expectation + /// of the capped-sampling batch estimator the trainer optimizes. /// /// The per-instance formula is the batch relation term's with the estimator scale at one, and /// the twin lives at [`relation_term`](crate::salt::projector::loss::relation_term). @@ -500,8 +545,8 @@ impl RelationLossReadout { let (frame, scales) = (frame.coordinates(), frame.scales()); let epsilon = energy.epsilon(); - // The mixture readings are raw and can carry an f32 overflow, so every accumulation chain - // runs as a derivation and makes its one claim at the readout's construction. + // The mixture returns unclaimed derivations. Every accumulation chain runs as a derivation + // too and makes its one claim at the readout's construction. let mut uncapped_total = Derivation::::ZERO; let mut capped_total = Derivation::::ZERO; let mut per_type = Vec::with_capacity(index.groups().len()); @@ -532,8 +577,8 @@ impl RelationLossReadout { clippy::cast_precision_loss, reason = "group sizes and the cap stay far below f64's exact-integer range" )] - // `min(cap, n) / n` with `n ≥ 1`: the index never stores an empty group, so the - // quotient is finite and in `(0, 1]`. + // `min(cap, n) / n` with `n ≥ 1`. The index never stores an empty group, and the + // quotient is therefore finite and in `(0, 1]`. let clip = DNonNegative::new_unchecked(cap.get().min(edges.len()) as f64 / edges.len() as f64); capped_total = share.mul_add(clip, capped_total); @@ -543,8 +588,9 @@ impl RelationLossReadout { )); } - // The folds run over validated weights and clips in (0, 1], so a non-finite finish marks - // a defect upstream of this readout. + // The folds run over validated f32-born weights and clips in (0, 1], and their double-width + // products lie far inside the f64 range. A non-finite finish therefore marks a defect + // upstream of this readout. Self { uncapped_total: uncapped_total .finish() diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/projector/tests.rs b/libs/@local/graph/atlas/src/salt/fit/compute/projector/tests.rs index 99fcd93e7c4..9e51d66178e 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/projector/tests.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/projector/tests.rs @@ -74,15 +74,18 @@ use crate::{ /// Corpus rows of the publish fixture. /// -/// Row 3 carries row 0's representation and row 5 carries row 2's, so the quotient collapses -/// six corpus rows onto four distinct rows. +/// Row 3 carries row 0's representation and row 5 carries row 2's: the quotient collapses six +/// corpus rows onto four distinct rows. const ROWS: usize = 6; +/// Distinct row count of the publish fixture after the quotient. const DISTINCT: usize = 4; +/// Component count of the publish fixture's corpus storage. const CORPUS_CAPACITY: usize = ROWS * PROJECTOR_DIMENSIONS; /// The reviewed relation type of the attraction fixture. const RELATION: u64 = 7; +/// A fresh per-process scratch directory named `name` under the system temp dir. fn scratch_dir(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -202,8 +205,8 @@ fn corpus_indexes() -> RelationIndexes { /// Stages the corpus-domain attraction file, as the relation stage leaves it. /// -/// The paired-movement readout replays over the published index, so the ladder-walking -/// publish reads this file back. +/// The paired-movement readout replays over the published index: the ladder-walking publish +/// reads this file back. fn stage_attraction(staging: &StagedGeneration) { let relations = corpus_indexes(); staging @@ -213,10 +216,10 @@ fn stage_attraction(staging: &StagedGeneration) { .expect("the attraction index should stage"); } -/// The skinny projector fixture. +/// Builds the skinny projector fixture's options. /// /// The representation width stays the pipeline's contract while the hidden architecture -/// shrinks, so a forward pass costs a fraction of the `ratified()` model's. +/// shrinks: a forward pass costs a fraction of the `ratified()` model's. fn skinny_options() -> ProjectorOptions { let mut options = ProjectorOptions::ratified(); options.architecture = Architecture { @@ -246,15 +249,16 @@ fn skinny_options() -> ProjectorOptions { options } -/// The widest duplicate-cluster spread the staged column may show, in units in the last -/// place. +/// The widest duplicate-cluster spread the staged column may show. +/// +/// The spread counts units in the last place of the stored coordinates. /// /// Byte-identical representations project to one coordinate mathematically. The column, /// however, computes in `forward_rows`-bounded slices that put rows 5 and 2 into dispatches /// of different shapes, whose kernels the GPU backend's autotune selects independently. /// Under a loaded device the selections can disagree, and two reduction /// orders for one value differ in the last bit (observed at exactly one ulp under the full -/// parallel suite). Within one dispatch shape the selection is cached per process, so the +/// parallel suite). Within one dispatch shape the selection is cached per process, and the /// projection-identity assertions stay bit-exact. The tolerance keeps headroom over the /// observed single-ulp motion while still refusing value-scale divergence: a duplicate /// placed anywhere else in the plane is millions of ulps away. @@ -269,8 +273,10 @@ fn ulp_key(value: f32) -> i64 { } } -/// Reads the staged canonical column and asserts each duplicate cluster shares one -/// coordinate, up to [`DUPLICATE_ULPS`]. +/// Reads the staged canonical column and asserts its duplicate clusters coincide. +/// +/// Reads the staged canonical column and asserts each duplicate cluster shares one coordinate, up +/// to [`DUPLICATE_ULPS`]. fn staged_column(staging: &StagedGeneration) -> Vec { let column = ArrayFile::open(staging.path_of(&artifact::Coordinates::NAME)) .expect("the column should map"); @@ -292,8 +298,10 @@ fn staged_column(staging: &StagedGeneration) -> Vec { placed.to_vec() } -/// Asserts the staged column is the staged checkpoint's canonical-step projection under the -/// recorded alignment, bit for bit: checkpoint, evidence, and column describe one field. +/// Asserts the staged column equals the checkpoint's aligned canonical projection bit for bit. +/// +/// The column is the staged checkpoint's canonical-step projection under the recorded alignment: +/// checkpoint, evidence, and column describe one field. #[track_caller] fn assert_column_is_aligned_projection( staging: &StagedGeneration, @@ -347,8 +355,7 @@ fn tick_fractions() -> [RefreshFraction; 2] { /// Asserts every step's per-type shares add up to its total within accumulation rounding. /// -/// The shares run their own chains, so the agreement is a relative tolerance, not bit -/// equality. +/// The shares run their own chains: the agreement is a relative tolerance, not bit equality. #[track_caller] fn assert_per_type_additivity(ladder: &LadderEvidence) { for step in &ladder.steps { @@ -359,9 +366,9 @@ fn assert_per_type_additivity(ladder: &LadderEvidence) { (sum - total).abs() <= 1e-12 * total.max(1.0), "per-type shares {sum} should add up to the step total {total}", ); - // The fixture's one group holds fewer edges than the ratified cap, so its clip is - // one and the capped estimand echoes the total bit-exactly: with a single group the - // share's chain is the total's, and a fused multiply by one onto zero is exact. + // The fixture's one group holds fewer edges than the ratified cap: its clip is one, and + // the capped estimand echoes the total bit-exactly, because with a single group the + // share's chain is the total's and a fused multiply by one onto zero is exact. assert_eq!( step.capped_relation_loss, Some(step.relation_loss), @@ -430,6 +437,7 @@ fn loss_regression_final_odd_transition() { assert_eq!(regression.relative, Some(d_positive!(1.0))); } +/// A rise from a zero loss reports its delta with no relative rise. #[test] fn loss_regression_zero_predecessor() { let conditions = [non_negative!(0.0), non_negative!(1.0)]; @@ -648,6 +656,7 @@ fn reproducibility() -> Reproducibility { } } +/// Builds the publish fixture configuration. fn fit_config() -> FitConfig { FitConfig { seed: 11, @@ -752,9 +761,8 @@ fn publish_vacuous_baseline() { "a vacuous boundary persists no calibration body, absent rather than zero" ); - // The staged column is the model's own zero-step projection, bit - // for bit, and byte-identical representations share one - // coordinate up to the last-bit motion `staged_column` prices. + // The persisted coordinates match the fresh zero-step projection bit for bit. Separately, + // `staged_column` checks duplicate representations within `DUPLICATE_ULPS`. let placed = staged_column(&context.staging); let projected = refresh::forward( &model.valid(), @@ -776,6 +784,11 @@ fn publish_vacuous_baseline() { ); } +/// Publishes a two-step proximal run and checks its recorded placement artifacts. +/// +/// Publishing a two-step run under proximal force records a measured boundary with its calibration +/// body, a two-step ladder with the canonical step last at `1.0` and the baseline at the identity +/// alignment, per-type additivity, and a column equal to the aligned canonical projection. #[test] #[expect( clippy::significant_drop_tightening, diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/quotient.rs b/libs/@local/graph/atlas/src/salt/fit/compute/quotient.rs index 7becc10ffcb..b5dd4ea6c83 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/quotient.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/quotient.rs @@ -42,7 +42,7 @@ hashql_core::id::newtype! { /// The byte-exact quotient of a representation matrix, holding both row domains of one fit. /// -/// The quotient owns the two row maps and keeps the corpus matrix it was built over; where +/// The quotient owns the two row maps and keeps the corpus matrix it was built over. Where /// copies exist it also owns the materialized distinct matrix. It is therefore the one value that /// answers every domain question of a fit: [`corpus`](Self::corpus) is the publication domain, /// [`training`](Self::training) the training domain, and [`class_of`](Self::class_of) and @@ -65,9 +65,9 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { /// Quotients the corpus by byte equality of its rows and materializes the training matrix. /// /// Every corpus row maps to a distinct row, and every distinct row names its first corpus - /// row. Distinct rows ascend with their first occurrence, so gathering the first rows from - /// the corpus preserves stream order. Row equality is raw byte equality of the representation - /// vectors: the key distinguishes every representable bit pattern. + /// row. Distinct rows ascend with their first occurrence, and gathering the first rows from + /// the corpus therefore preserves stream order. Row equality is raw byte equality of the + /// representation vectors: the key distinguishes every representable bit pattern. /// /// The distinct matrix materializes under `scratch` exactly when copies exist, gathering each /// distinct row's bytes from its first corpus occurrence. An identity quotient writes nothing, @@ -80,10 +80,11 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { /// # Panics /// /// A corpus row count exceeding the store's `u32` row domain panics here: every published - /// column and the neighbour table's packed columns address rows as `u32`, so a wider corpus - /// refuses before any stage spends time on it. The call also panics when the matrix it just - /// wrote does not map back as aligned `f32` rows, which is a defect of the writer rather - /// than of the input. + /// column and the neighbour table's packed columns address rows as `u32`, and a wider corpus + /// therefore refuses before any stage spends time on it. The call also panics when the matrix + /// it just wrote does not open and map back as aligned `f32` rows. A shape mismatch there is a + /// defect of the writer, while the open can also fail on the scratch file system after a + /// complete write. #[tracing::instrument(name = "quotient-build", skip_all)] pub(super) fn build( corpus: &'corpus IdSlice>, @@ -134,17 +135,17 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { }) } - /// Number of corpus rows. + /// Returns the number of corpus rows. pub(super) const fn len(&self) -> usize { self.classes.len() } - /// Number of distinct representation rows. + /// Returns the number of distinct representation rows. pub(super) const fn distinct_len(&self) -> usize { self.representatives.len() } - /// Whether every corpus row is already distinct. + /// Returns whether every corpus row is already distinct. /// /// An identity quotient lets a fit skip the distinct detour entirely: the corpus and the /// distinct domain are the same rows in the same order. @@ -152,16 +153,16 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { self.representatives.len() == self.classes.len() } - /// The corpus matrix the quotient was built over: the publication row domain. + /// Returns the corpus matrix the quotient was built over, the publication row domain. pub(super) const fn corpus(&self) -> &'corpus IdSlice> { self.corpus } - /// The training matrix of every distinct representation row, in first-occurrence order. + /// Returns the training matrix of every distinct representation row, in first-occurrence order. /// /// Where copies exist this is the materialized distinct matrix. Under the identity quotient - /// it is the corpus matrix reborrowed under the distinct key, because the two domains are - /// then the same rows in the same order, so no second mapping and no copy exists. + /// it is the corpus matrix reborrowed under the distinct key: the two domains are then the + /// same rows in the same order, and no second mapping and no copy exists. pub(super) const fn training(&self) -> &IdSlice> { match &self.materialized { Some(matrix) => matrix, @@ -179,20 +180,20 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { self.representatives[distinct] } - /// The distinct class of every corpus row, in corpus order. + /// Returns the distinct class of every corpus row, in corpus order. pub(super) const fn classes(&self) -> &IdSlice { &self.classes } - /// The representative of every distinct class, strictly ascending. + /// Returns the representative of every distinct class, strictly ascending. pub(super) const fn representatives(&self) -> &IdSlice { &self.representatives } /// Gathers a corpus frame at the representatives: the training domain's own frame. /// - /// Identical representations project identically, so the gathered frame is the distinct - /// rows' own coordinates rather than a sample of them. + /// Identical representations project identically, and the gathered frame is therefore the + /// distinct rows' own coordinates rather than a sample of them. pub(super) fn training_frame( &self, frame: &FinitePointField, @@ -203,12 +204,13 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { /// Expands a distinct-domain neighbour table onto the corpus row domain. /// /// Every corpus row takes its representative's neighbour list, each neighbour named by its own - /// first corpus row. The gather preserves every table invariant: `representatives` ascends - /// strictly, so row entries stay strictly ascending, and no entry names its own row - a + /// first corpus row. The gather preserves every table invariant. `representatives` ascends + /// strictly, and row entries therefore stay strictly ascending. No entry names its own row: a /// distinct list excludes its own index, and expanded entries name first rows only. /// - /// Under the identity quotient the two domains are the same rows in the same order, so the - /// table is already the corpus's own and no expansion materializes: the call returns `None`. + /// Under the identity quotient the two domains are the same rows in the same order. The + /// table is then already the corpus's own and no expansion materializes: the call returns + /// `None`. pub(super) fn expand_neighbours( &self, table: &KnnView<'_, DistinctRowId>, @@ -267,7 +269,8 @@ impl<'corpus, const N: usize> Quotient<'corpus, N> { /// render one asserted link as many edge rows, and the trainer weighs the assertion once /// rather than per copy. Instances whose endpoints collapse onto one distinct row pass /// through, and the index build drops and counts them as self-references. The collapse - /// orders totally before deduplicating, so the result is a function of the instance set. + /// orders totally before deduplicating, and the result is therefore a function of the + /// instance set. pub(super) fn collapse_instances( &self, instances: &[RelationInstance], @@ -340,6 +343,11 @@ mod tests { } impl Matrix { + /// Copies `rows` into the head of the aligned storage. + /// + /// # Panics + /// + /// Panics when `rows` exceeds the fixture capacity. fn new(rows: &[[f32; WIDTH]]) -> Self { let mut storage = BoxedVecN::zero(); let (chunks, _) = storage.as_array_mut().as_chunks_mut::(); @@ -353,6 +361,7 @@ mod tests { } } + /// The resident rows as an aligned row-indexed slice. fn view(&self) -> &hashql_core::id::IdSlice> { hashql_core::id::IdSlice::from_raw( AlignedVecN::from_slice(&self.storage.as_array()[..self.rows * WIDTH]) @@ -361,6 +370,7 @@ mod tests { } } + /// A row whose every component is `fill`. fn row(fill: f32) -> [f32; WIDTH] { [fill; WIDTH] } @@ -395,6 +405,10 @@ mod tests { ) } + /// Builds the identity quotient from a matrix of distinct rows. + /// + /// A matrix of distinct rows builds the identity quotient: the training matrix reborrows the + /// corpus, the maps are identities and no neighbour expansion materializes. #[test] fn identity_quotient() { let matrix = Matrix::new(&[row(1.0), row(2.0), row(3.0)]); @@ -413,7 +427,7 @@ mod tests { assert_eq!(classes, [0, 1, 2]); assert_eq!(representatives, [0, 1, 2]); - // Under the identity quotient the distinct table is already the corpus's own, so no + // Under the identity quotient the distinct table is already the corpus's own, and no // expansion materializes. let table_matrix = KnnMatrix::try_new( (3, 3), @@ -454,10 +468,11 @@ mod tests { } } + /// `0.0` and `-0.0` fall into different classes, since the quotient keys on bytes. #[test] fn byte_exact_equality() { - // 0.0 and -0.0 compare equal as floats and differ as bytes; the - // quotient keys on bytes. + // 0.0 and -0.0 compare equal as floats and differ as bytes, and + // the quotient keys on bytes. let mut negative = row(0.0); negative[0] = -0.0; let matrix = Matrix::new(&[row(0.0), negative, row(0.0)]); @@ -471,6 +486,7 @@ mod tests { assert_eq!(representatives, [0, 1]); } + /// Representatives are the first rows of their classes, in ascending order. #[test] fn representatives_ascending() { let matrix = Matrix::new(&[row(5.0), row(5.0), row(4.0), row(4.0), row(6.0)]); @@ -484,6 +500,10 @@ mod tests { assert!(ascending, "distinct rows follow first occurrence order"); } + /// Expands a distinct-domain kNN table onto every copy of a representative. + /// + /// Expanding a distinct-domain kNN table gives every copy its representative's list, with + /// neighbours naming first rows. #[test] fn expand_names_representatives() { let matrix = Matrix::new(&[row(1.0), row(2.0), row(1.0), row(3.0), row(2.0)]); @@ -520,6 +540,11 @@ mod tests { assert_eq!(lists, [(1, 0.5), (0, 0.5), (1, 0.5), (1, 0.25), (0, 0.5)],); } + /// Collapses instances onto the distinct domain and keeps the strongest reading per triple. + /// + /// Collapsing instances onto the distinct domain keeps the strongest reading per + /// `(relation, source, target)`, breaking equal confidence by the lower edge row, and keeps + /// distinct relations apart. #[test] fn collapse_strongest_per_triple() { let matrix = Matrix::new(&[row(1.0), row(2.0), row(1.0), row(3.0), row(2.0)]); @@ -575,6 +600,7 @@ mod tests { ); } + /// The training matrix's row `i` is byte-equal to the corpus row `representatives[i]`. #[test] fn materialize_gathers_representatives() { let matrix = Matrix::new(&[row(1.0), row(2.0), row(1.0), row(3.0)]); diff --git a/libs/@local/graph/atlas/src/salt/fit/compute/relation.rs b/libs/@local/graph/atlas/src/salt/fit/compute/relation.rs index b16dcddec2f..433e0dbf86c 100644 --- a/libs/@local/graph/atlas/src/salt/fit/compute/relation.rs +++ b/libs/@local/graph/atlas/src/salt/fit/compute/relation.rs @@ -26,8 +26,8 @@ use crate::{ /// The adjacency stage failed and staged nothing. /// -/// One variant per way the stage refuses, so an adjacency failure attributes to this stage by -/// construction. +/// One variant per way the stage refuses. An adjacency failure therefore attributes to this +/// stage by construction. #[derive(Debug)] pub(crate) enum AdjacencyError { /// The staged endpoint column failed to map in. @@ -64,8 +64,8 @@ impl Error for AdjacencyError { /// The relation stage failed and staged no index. /// -/// One variant per way the stage refuses, so a relation failure attributes to this stage by -/// construction. +/// One variant per way the stage refuses. A relation failure therefore attributes to this stage +/// by construction. #[derive(Debug)] pub(crate) enum RelationError { /// An index build rejected its instances. @@ -133,13 +133,19 @@ impl<'fit> AdjacencyDerivation<'fit> { /// Derives the incident-edge adjacency from the staged endpoint column and stages it. /// - /// Returns the owned adjacency beside its typed binding, so the level-of-detail stage reads + /// Returns the owned adjacency beside its typed binding, and the level-of-detail stage reads /// degrees from the value this call built rather than from the staged bytes. /// /// # Errors /// /// Returns [`AdjacencyError::OpenEndpoints`] when the staged endpoint column does not map, /// and [`AdjacencyError::Write`] when the staged adjacency does not write. + /// + /// # Panics + /// + /// This panics when the mapped endpoint column does not hold little-endian `u64` pairs. The + /// ingest sealed the column in that shape, and a mismatch is a defect of the writer rather + /// than of the input. #[tracing::instrument(name = "adjacency-derivation", skip_all)] pub(super) fn run(self) -> Result, AdjacencyError> { let endpoints = @@ -205,16 +211,17 @@ impl<'fit> RelationAssembly<'fit> { /// Assembles the spooled relation instances against the resolved policy table. /// - /// Builds the corpus-domain relation indexes and stages their published artifacts, keeping - /// their measurements, so the edge-scale corpus pair is spent inside this stage. It then - /// rebuilds the pair over the distinct row domain for the placement stage: endpoints - /// quotient-mapped, duplicate readings collapsed, degrees and protection evidence re-derived - /// by the same build over the collapsed set. The trainer's indexes return owned. + /// Builds the corpus-domain relation indexes, stages their published artifacts and keeps + /// their measurements. The edge-scale corpus pair is therefore consumed inside this stage. + /// The stage then rebuilds the pair over the distinct row domain for the placement stage: + /// endpoints quotient-mapped, duplicate readings collapsed, degrees and protection evidence + /// re-derived by the same build over the collapsed set. The trainer's indexes return owned. /// /// # Errors /// - /// Returns [`RelationError::Build`] when either index build rejects the instances, and an - /// I/O error when the instance spool does not map or a staged artifact does not write. + /// Returns [`RelationError::Build`] when either index build rejects the instances, + /// [`RelationError::Write`] when the protection index does not stage, and an I/O error when + /// the instance spool does not map or the attraction index does not stage. #[tracing::instrument(name = "relation-assembly", skip_all)] pub(super) fn run( self, @@ -222,9 +229,9 @@ impl<'fit> RelationAssembly<'fit> { let policies = Policies::from(self.policies); let mapped = self.spool.map()?; - // The spool maps read-only encoded records and the index build sorts its instance slice - // in place, so the readings decode into owned storage once; the mapping unmaps before - // the sorts run. + // The spool maps read-only encoded records, and the index build sorts its instance slice + // in place. The readings therefore decode into owned storage once, and the mapping + // unmaps before the sorts run. let mut instances: Vec<_> = mapped .records() .iter() @@ -249,9 +256,8 @@ impl<'fit> RelationAssembly<'fit> { )?; drop(collapsed); - // The histogram and the clamp count are drain facts the build - // cannot see; they join the build measurements here on their way - // to the manifest. + // The multiplicity histogram is a drain fact the build cannot + // see. It joins the build measurements here for the manifest. corpus.measurements.multi_typed_edges = self.multi_typed.to_vec(); let attraction = self @@ -288,10 +294,11 @@ impl<'fit> RelationAssembly<'fit> { } } -/// The relation stage's published artifacts, pairing the staged corpus bindings with their -/// measurements. +/// The relation stage's published artifacts. +/// +/// They pair the staged corpus bindings with their measurements. /// -/// The corpus-domain indexes are spent once staged. The manifest keeps their measurements, and +/// The corpus-domain indexes are dropped once staged. The manifest keeps their measurements, and /// the placement's paired-movement readout replays from the staged bytes on purpose. pub(super) struct RelationArtifacts { /// The staged attraction index's typed binding. diff --git a/libs/@local/graph/atlas/src/salt/fit/error.rs b/libs/@local/graph/atlas/src/salt/fit/error.rs index 3677953490d..f6e1315c33d 100644 --- a/libs/@local/graph/atlas/src/salt/fit/error.rs +++ b/libs/@local/graph/atlas/src/salt/fit/error.rs @@ -96,10 +96,13 @@ impl Error for PriorError { } } -/// One fit failed and published nothing. +/// A failure of one fit, from either side of its thread boundary. /// /// `D` is the dataset's error and `E` the embedding provider's. Both arise only during ingest. -/// Every compute-side failure arrives as [`FitError::Compute`]. +/// Every compute-side failure is wrapped as [`FitError::Compute`]. Every ingest-side variant +/// arises before the seal and leaves nothing published. A [`ComputeError::Seal`] raised after the +/// seal's rename and a [`ComputeError::Offload`] raised by the progress observer's seal report +/// return with the generation directory already visible. #[derive(Debug)] pub(crate) enum FitError { /// The dataset failed to deliver a stream item. diff --git a/libs/@local/graph/atlas/src/salt/fit/ingest.rs b/libs/@local/graph/atlas/src/salt/fit/ingest.rs index ba4c73c4095..39f329b985a 100644 --- a/libs/@local/graph/atlas/src/salt/fit/ingest.rs +++ b/libs/@local/graph/atlas/src/salt/fit/ingest.rs @@ -1,11 +1,10 @@ //! The ingest side of one fit: everything that reads the dataset. //! -//! [`Ingest::run`] drains the dataset streams - nodes, edges, ontology, cards, and the display -//! streams -//! beside them - into their staged artifacts -//! and resident columns and certifies the representation contract, so everything the compute side -//! needs afterwards lives in staged files and the returned [`Ingested`] value. The dataset and the -//! embedding provider are never touched again after this module returns. +//! [`Ingest::run`] drains the dataset streams - nodes, edges, ontology, cards, and the auxiliary +//! payload streams (labels and icons) beside them - into their staged artifacts and resident +//! columns, and certifies the representation contract. Everything the compute side needs +//! afterwards therefore lives in staged files and the returned [`Ingested`] value. The dataset +//! and the embedding provider are never touched again after this module returns. use alloc::collections::BTreeSet; use core::{borrow::Borrow, pin::pin}; @@ -75,8 +74,9 @@ struct EdgeArtifacts { endpoints: Binding, /// The spooled `(edge, relation)` readings. instances: InstanceSpool, - /// The edge multiplicity histogram, whose entry `i` counts edges carrying `i + 1` relation - /// readings. + /// The edge multiplicity histogram. + /// + /// Entry `i` counts edges carrying `i + 1` relation readings. multi_typed: Vec, } @@ -125,8 +125,9 @@ pub(super) struct Ingested { pub edge_endpoints: Binding, /// The spooled `(edge, relation)` readings the relation stage consumes. pub instances: InstanceSpool, - /// The edge multiplicity histogram, whose entry `i` counts edges carrying `i + 1` relation - /// readings. + /// The edge multiplicity histogram. + /// + /// Entry `i` counts edges carrying `i + 1` relation readings. pub multi_typed: Vec, /// The staged card-embedding artifacts. pub cards: CardArtifacts, @@ -145,11 +146,12 @@ impl Ingested { } } - /// The metadata document's `reproducibility` section: the configuration and provenance the - /// fit ran under. + /// The metadata document's `reproducibility` section. + /// + /// It records the configuration and provenance the fit ran under. /// - /// With [`snapshot`](Self::snapshot) it forms the paired-movement salt preimage, so the - /// readout's draw replays from the published document's input sections alone. + /// With [`snapshot`](Self::snapshot) it forms the paired-movement salt preimage. The + /// readout's draw therefore replays from the published document's input sections alone. pub(super) const fn reproducibility( &self, config: FitConfig, @@ -180,9 +182,18 @@ where /// Drains the dataset into the staged stream artifacts. /// /// The stages run in the dataset's documented ingest order (nodes, edges, ontology, then the - /// card render over the same type table) and the ingest certifies the representation - /// contract before the card stream touches the embedding provider, so a defective corpus - /// never spends provider budget. + /// card render over the same type table). Before the card stream touches the embedding + /// provider, the norm spot check certifies the representation contract on a row sample, and + /// its refusal stops the ingest before the generation's cards embed. A defective row outside + /// that sample passes unseen, while a supplied annotation corpus has already embedded before + /// the ingest begins. + /// + /// # Errors + /// + /// Returns the failing stage's [`FitError`] for a dataset stream failure, a streamed staged + /// write, a representation matrix that does not map back or whose norm spot check refuses or + /// fails, a card stream or embedding failure, or a prior generation whose card files do not + /// serve reuse. pub(super) async fn run( self, embedder: &E, @@ -273,6 +284,11 @@ where /// /// The matrix digest streams over the finished file because the writer seals its header by /// seeking. The identity writer is forward-only and digests inline. + /// + /// # Errors + /// + /// Returns [`FitError::Dataset`] when the node stream or its auxiliary payload fails, and + /// [`FitError::Io`] when a staged write, the flush, or the digest fails. async fn stage_representations(&self) -> Result> { let mut writer = BufWriter::new(self.staging.create(&artifact::Representations::NAME)?); let columns = prepare::write_node_representations(self.dataset, &mut writer) @@ -313,6 +329,18 @@ where /// Certifies the source contract on the freshly staged representation rows. /// /// Returns the passing evidence. + /// + /// # Errors + /// + /// Returns [`FitError::OpenRepresentations`] when the staged matrix does not map back, + /// [`FitError::NormCheck`] when the spot check's sampling settings are unusable, and + /// [`FitError::RepresentationDefects`] when the sampled rows violate the source contract. + /// + /// # Panics + /// + /// This panics when the mapped matrix does not hold `f32` rows of the projector width. The + /// node drain just sealed it in that shape, and a mismatch is a defect of the writer rather + /// than of the input. fn certify_representations( &self, config: &FitConfig, @@ -351,6 +379,16 @@ where /// /// The endpoint digest streams over the finished file because the array writer seals its /// header by seeking. The identity writer is forward-only and digests inline. + /// + /// # Errors + /// + /// Returns [`FitError::Dataset`] when the edge stream or its auxiliary payload fails, and + /// [`FitError::Io`] when the spool, a staged write, the flush, or the digest fails. + /// + /// # Panics + /// + /// This panics when an edge carries more direct types than `u32` counts. The dataset's direct + /// type lists are deduplicated ontology rows, far below that bound. async fn stage_edges(&self) -> Result> { let mut ids = IdentityTable::new(); let mut relations = BTreeSet::new(); @@ -427,6 +465,10 @@ where /// /// The column is type-scale and crosses to the compute side by value: the postings build /// restates it as the published type graph's parent regions. + /// + /// # Errors + /// + /// Returns [`FitError::Dataset`] when the ontology stream fails. async fn collect_type_parents( &self, ) -> Result>, FitError> { @@ -441,11 +483,19 @@ where /// Renders every card and stages the card-embedding columns from the embedded unique texts. /// - /// Both columns land beside the ontology identity table collected from the same stream. + /// Both columns are staged beside the ontology identity table collected from the same stream. /// /// A prior generation's card files map back as the reuse table: texts whose hash the reuse /// table lists keep their rows without touching the provider. Reuse is fingerprint-guarded - /// inside [`embed_cards`], so a changed embedding contract re-embeds everything. + /// inside [`embed_cards`], and a changed embedding contract therefore re-embeds everything. + /// + /// # Errors + /// + /// Returns [`FitError::Cards`] when the card render stream fails, [`FitError::Dataset`] when + /// the ontology auxiliary payload fails, [`FitError::Io`] when a staged write fails, + /// [`FitError::Embedding`] when the provider fails to produce the table, and a [`PriorError`] + /// (wrapped as [`FitError::Compute`]) when the prior card files do not map or are not + /// row-aligned digest and embedding columns. async fn embed_card_table( &self, embedder: &E, diff --git a/libs/@local/graph/atlas/src/salt/fit/mod.rs b/libs/@local/graph/atlas/src/salt/fit/mod.rs index de0b7daf737..bf31e0bd44c 100644 --- a/libs/@local/graph/atlas/src/salt/fit/mod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/mod.rs @@ -9,16 +9,25 @@ //! //! The last dataset touch splits the pipeline. [`ingest`] runs on the async runtime and drains the //! dataset's streams and the embedding provider into staged files. [`compute`] runs on the rayon -//! pool behind [`offload`], so the CPU-heavy stages never occupy a tokio runtime thread. A stage -//! panic surfaces as [`compute::ComputeError::Panicked`] instead of poisoning the executor. +//! pool behind [`offload`], and the CPU-heavy stages never occupy a tokio runtime thread. A stage +//! panic surfaces as `compute::ComputeError::Offload` instead of poisoning the executor. //! //! # Memory discipline //! -//! Every corpus-scale stage output goes to its staged file and maps back before the next stage -//! reads it. Owned `N`-scale values are construction-transient and drop at stage exit, so the -//! pipeline's peak residency is one stage's working set rather than the sum. The mapped views stay -//! cheap because their pages are freshly written and evictable under pressure. Config-bounded -//! `M`-scale values (the landmark selection, the quotient graph) stay resident within the run. +//! Ingest streams the corpus-scale inputs into staged files, and the compute stages map those +//! files in where they consume them. The corpus matrix and the identity table open at the compute +//! boundary, and the card, endpoint and ontology identity columns open inside the stages that +//! read them. The mapped pages are freshly written and evictable under pressure. Every compute +//! stage returns its own product as an owned value beside the staged file's binding +//! ([`compute::Staged`]), and the run retains each value until it returns. The representation +//! quotient's row maps, the adjacency, the trainer's relation indexes, the admitted neighbour +//! table, the semantic graph and the skeleton are all resident while the placement runs, and the +//! adjacency and the ingest's resident type columns feed the level-of-detail stage after it. Peak +//! residency is the retained products plus the running stage's own working storage, not one +//! stage's working set alone. Corpus-scale intermediates a stage writes and maps rather than +//! retains live in the scratch directory: the quotient's distinct matrix and the placement's +//! ladder frames. Config-bounded `M`-scale values (the landmark selection and its quotient graph) +//! stay resident within the run. //! //! # Seeds //! @@ -164,7 +173,7 @@ const _: () = assert!(LandmarkSupport::new(LandmarkSupport::default().weight()). /// /// The model, its training run, and the condition ladder that publishes the canonical field. /// -/// Each field is a validated value. The struct is plain wiring. [`ratified`](Self::ratified) is the +/// Each field is a validated value. The struct is plain wiring. [`live`](Self::live) is the /// stamped live configuration and the placement default. /// /// The semantic affinity energy composes at stage entry from the fit's low-dimensional kernel and @@ -183,8 +192,8 @@ pub(crate) struct ProjectorOptions { pub plan: BatchPlan, /// The logarithm offset of the semantic affinity energy. /// - /// It bounds the near-coincidence repulsion derivative, so it is a force ceiling, not a - /// numerical crumb. + /// It bounds the near-coincidence repulsion derivative: it is a force ceiling rather than a + /// numerical guard alone. pub affinity_offset: Positive, /// The support-term constants shared by anchors and landmarks. pub support: SupportOptions, @@ -199,7 +208,7 @@ pub(crate) struct ProjectorOptions { /// /// The placement stage normalizes them at assembly. The semantic and ordinary bases divide by /// the corpus's total semantic edge weight, the hard-negative base by the row count, and the - /// support bases by their pool sizes, so a configured base weighs the same objective share on + /// support bases by their pool sizes, and a configured base weighs the same objective share on /// every corpus. The relation base passes through unchanged, because its estimator is already /// mass-free. pub coefficients: Coefficients, @@ -225,7 +234,7 @@ pub(crate) struct ProjectorOptions { } impl ProjectorOptions { - /// Returns the ratified live configuration. + /// Returns the live configuration. /// /// Every value stamped for production training, schedule included. /// @@ -319,7 +328,7 @@ pub(crate) enum PlacementOptions { /// How one fit constructs its k-NN lists. /// -/// The search-backend wrapper is the default; NN-Descent derives the lists directly, with no +/// The search-backend wrapper is the default. NN-Descent derives the lists directly, with no /// search structure. Either construction answers to the same recall spot check, and neither /// outlives the fit: the wrapper's index lives in the fit's scratch directory, which removes /// itself when the run ends. @@ -414,7 +423,7 @@ impl Stage { /// Derives one stage's generator from the fit seed and the stage's pinned name. /// -/// The full 32-byte digest seeds the generator, so a derived stream keeps the derivation's whole +/// The full 32-byte digest seeds the generator, and a derived stream keeps the derivation's whole /// entropy. pub(crate) fn stage_rng(seed: u64, stage: Stage) -> Xoshiro256PlusPlus { let mut hasher = Sha256::new(); @@ -518,9 +527,9 @@ pub(crate) struct Supplies<'fit> { /// Runs one fit over the dataset and publishes the generation. /// /// The stages run in the dataset's documented ingest order (nodes, edges, ontology) with every -/// artifact staged in place, so the returned generation is complete, durable, and verifiable -/// against its metadata document. Activation stays with the caller. Publishing a generation and -/// serving it are separate decisions. +/// artifact staged in place. The returned generation is therefore complete, durable, and +/// verifiable against its metadata document. Activation stays with the caller. Publishing a +/// generation and serving it are separate decisions. /// /// The `classifier` input resolves to a fitted model either way ([`ClassifierInput`]). A supplied /// artifact passes through unchanged. The run instead stages an annotation corpus verbatim, @@ -530,8 +539,10 @@ pub(crate) struct Supplies<'fit> { /// /// The `verdicts` are a supplied input in the policy-override category. A validated /// reviewed-verdicts document ([`SuppliedVerdicts`]) stages verbatim as the generation's -/// `reviewed_verdicts` role for the trainer's phase boundary to consume. The fit itself never acts -/// on it. A fit run without one publishes with the role absent. The manifest records the absence. +/// `reviewed_verdicts` role. The fit derives nothing from it before the placement stage, where it +/// resolves the verdicts against the staged ontology identity column into the corpus row domain +/// and hands the resolution to the trainer's phase boundary. The staged bytes stay the supplied +/// file's. A fit run without one publishes with the role absent. The manifest records the absence. /// /// A `prior` generation seeds reuse. Card texts whose hash its card table lists keep their /// embeddings without touching the provider (under a matching embedder fingerprint). Its landmarks @@ -544,8 +555,10 @@ pub(crate) struct Supplies<'fit> { /// [`FitError::Cards`], [`FitError::Embedding`]) or a supplied annotation corpus fails to assemble /// into the classifier's training set ([`FitError::Assembly`]). A streamed ingest write can also /// fail ([`FitError::Io`]), and any compute stage rejecting its input, failing an admission check, -/// or unable to write, map, or publish answers [`FitError::Compute`]. The run publishes nothing on -/// any error. +/// or unable to write, map, or publish answers [`FitError::Compute`]. Every error before the +/// seal's rename leaves nothing published. A seal error after the rename, or a progress observer +/// panicking after the seal, returns an error although the generation directory is already +/// visible. #[expect( clippy::significant_drop_tightening, reason = "the staging and scratch directories move into the compute closure whole; nothing \ @@ -572,9 +585,9 @@ where let staging = root.stage()?; let scratch = root.scratch()?; - // The supplied verdicts stage before any derivation: construction - // already validated the document, so nothing after this write can - // reject it, and the staged bytes are the supplied file verbatim. + // The supplied verdicts stage before any derivation. Construction + // already validated the document, and nothing after this write can + // therefore reject it. The staged bytes are the supplied file verbatim. let reviewed_verdicts = match verdicts { Some(supplied) => { let file = staging.stage_with(artifact::ReviewedVerdicts, |writer| { @@ -651,7 +664,7 @@ where ingested, }; - // The compute half leaves this stack for the rayon pool, so it takes the observer's detached + // The compute half leaves this stack for the rayon pool, and it takes the observer's detached // half rather than a borrow the spawn cannot hold. let detached = progress.detach(); let published = diff --git a/libs/@local/graph/atlas/src/salt/fit/prepare/identity.rs b/libs/@local/graph/atlas/src/salt/fit/prepare/identity.rs index e0361b02511..3a9b080598f 100644 --- a/libs/@local/graph/atlas/src/salt/fit/prepare/identity.rs +++ b/libs/@local/graph/atlas/src/salt/fit/prepare/identity.rs @@ -10,14 +10,14 @@ //! opaque to the pipeline, ordered by its bytes since source identifiers carry no other order. //! The payload is the row's display bytes, a legend for node and edge rows and an icon for //! ontology rows, its type's empty value when the row displays nothing. The file format is -//! [`file::identity`](crate::file::identity)'s, and the row domain a file covers travels in -//! its header, so a file reopens only under the row type that wrote it. +//! [`file::identity`](crate::file::identity)'s, and its header records the row domain. A file +//! reopens only under the row type that wrote it. //! //! This module owns the table's domain invariants. The index holds exactly one entry per row and //! every entry agrees with the id column, which makes the index and the column two views of one //! bijection, every span lies inside the payload region, and every span's bytes cast as the id -//! type's payload. One `O(N)` pass validates all of that on open, so a lookup afterwards never -//! reports a malformed file. +//! type's payload. One `O(N)` pass validates all of that on open. A lookup afterwards never reports +//! a malformed file. //! //! [`Dataset::NodeId`]: crate::dataset::Dataset::NodeId //! [`Dataset::EdgeId`]: crate::dataset::Dataset::EdgeId @@ -168,11 +168,8 @@ impl Error for InvalidIdentityFile {} /// A written identity table reopened as its mapped lookup surface. /// -/// Construction validates the domain invariants in one pass, so the lookups skip validation -/// afterwards: [`id`](Self::id) indexes the id column, [`row_of`](Self::row_of) resolves one -/// index lookup, and [`payload_of`](Self::payload_of) slices the payload region through the span -/// table. The table translates between one id domain and one row domain: `K` is the source id -/// type and `R` the row identity its lookups answer. +/// Construction validates the identity bijection and payload spans. The source identity type is +/// `K`, and `R` is the row domain. #[derive(Debug)] pub(crate) struct IdentityTableArchive { file: IdentityFile, @@ -224,7 +221,7 @@ where } // One entry per row plus column agreement makes the index a bijection: the map's keys - // are pairwise distinct, so two entries agreeing with one row's column bytes would be + // are pairwise distinct. Two entries agreeing with one row's column bytes would be // one key twice. let mut entries = index.stream(); while let Some((id, row)) = entries.next() { @@ -290,9 +287,6 @@ where self.file.index().get(id.as_bytes()).map(R::from_u64) } - /// Returns the display payload of `row`, or [`None`] beyond the domain. - /// - /// A row without a display value returns its payload type's empty value. #[expect( clippy::cast_possible_truncation, reason = "`Self::new` bounded every span by the payload region, whose length is a `usize`" @@ -305,8 +299,8 @@ where let bytes = &self.file.payload()[offset..offset + length]; // SAFETY: `Self::new` cast every span's bytes as `K::Payload` and rejected the file - // otherwise, and the mapped file is immutable under the `crate::file` publish contract, - // so the bytes validated there are the bytes sliced here. + // otherwise, and the mapped file is immutable under the `crate::file` publish contract. + // Therefore the bytes validated there are the bytes sliced here. Some(unsafe { ::try_ref_from_bytes(bytes).unwrap_unchecked() }) } @@ -321,6 +315,7 @@ where #[cfg(test)] mod tests { + use core::assert_matches; use std::{fs, path::PathBuf}; @@ -357,7 +352,7 @@ mod tests { OwnedLegend::new(OntologyRowId::new(0), Label::new(text)) } - // Ids in row order; ascending id-byte order is rows 2, 0, 1. + /// Ids in row order whose little-endian bytes sort as rows 2, 0, 1. #[expect( clippy::little_endian_bytes, reason = "the fixture pins ids whose little-endian bytes sort unlike their values" @@ -387,6 +382,11 @@ mod tests { (path, digest) } + /// Round-trips every fixture row through `key_of`, `row_of` and `payload_of_row`. + /// + /// The written fixture's digest hashes its bytes, every row round-trips through `key_of` and + /// `row_of` with misses answering `None`, and `payload_of_row` slices the interned region + /// including the empty label. #[test] fn written_table_reopens_with_all_three_translations() { let (path, digest) = written_fixture("roundtrip.idnt"); @@ -427,6 +427,7 @@ mod tests { assert_eq!(table.payload_of(NodeRowId::new(3)), None); } + /// An empty table writes, reopens with zero rows, and answers `None` for any lookup. #[test] fn empty_table_round_trips() { let table = IdentityTable::::new(); @@ -446,6 +447,7 @@ mod tests { assert_eq!(table.row_of(MemoryOntologyId::new(0)), None); } + /// Writing a table in which two rows carry one key panics with the documented message. #[test] #[should_panic(expected = "two rows carry one key")] fn table_refuses_duplicate_ids_at_write() { @@ -457,6 +459,10 @@ mod tests { let _result = table.write_into(core::iter::repeat_n(empty.as_ref(), 2), &mut Vec::new()); } + /// Looks up six hundred ids whose byte order differs from their value order in both directions. + /// + /// Six hundred ids whose byte order differs from their value order write and look up correctly + /// in both directions, with a miss beyond the domain answering `None`. #[test] fn lookups_hold_at_six_hundred_rows() { // Little-endian bytes of 0..600 sort unlike the values, so the write path's ordering @@ -490,6 +496,7 @@ mod tests { assert_eq!(table.row_of(MemoryNodeId::new(600)), None); } + /// Opening a node table as an ontology table fails with `Domain` naming both kinds. #[test] fn archive_refuses_a_foreign_row_domain() { let (path, _digest) = written_fixture("foreign-domain.idnt"); @@ -505,6 +512,7 @@ mod tests { ); } + /// Opening a `u64` keyed table under a UUID key type fails with `KeyKind` naming both kinds. #[test] fn archive_refuses_a_foreign_id_type() { let (path, _digest) = written_fixture("foreign-id.idnt"); @@ -520,6 +528,10 @@ mod tests { ); } + /// The open fails with `ColumnDisagreement` when a byte of the id column moves. + /// + /// Moving a byte of the id column under an intact index fails the open with + /// `ColumnDisagreement` naming the row. #[test] fn archive_refuses_a_disagreeing_id_column() { let (path, _digest) = written_fixture("disagreeing-column.idnt"); @@ -538,11 +550,15 @@ mod tests { ); } + /// The open fails with `SpanOutOfBounds` when a span overreaches the payload region. + /// + /// A span whose length overreaches the payload region fails the open with `SpanOutOfBounds` + /// naming the row. #[test] fn archive_refuses_a_span_beyond_the_payload() { let (path, _digest) = written_fixture("overreaching-span.idnt"); - // Three eight-byte ids pad to one region unit each for column and index, so the span + // Three eight-byte ids pad to one region unit each for column and index. The span // table starts at 12288. Row 0's length field is its second eight-byte word. let mut bytes = fs::read(&path).expect("the scratch file reads back"); bytes[12296..12304].fill(0xFF); @@ -556,13 +572,14 @@ mod tests { ); } + /// A label byte no UTF-8 sequence contains fails the typed open with `Payload` naming the row. #[test] fn archive_refuses_a_payload_that_is_not_utf8() { let (path, _digest) = written_fixture("invalid-payload.idnt"); // The payload region starts at 0x4000 and row 0's span selects its first twelve bytes: - // the eight-byte representative, then the label. No UTF-8 sequence contains 0xFF, so - // corrupting the label's first byte makes the typed cast refuse. + // the eight-byte representative, then the label. No UTF-8 sequence contains 0xFF. + // Corrupting the label's first byte makes the typed cast refuse. let mut bytes = fs::read(&path).expect("the scratch file reads back"); bytes[0x4000 + 8] = 0xFF; fs::write(&path, &bytes).expect("the scratch file is writable"); @@ -575,6 +592,10 @@ mod tests { ); } + /// The typed open fails with `IndexSize` when the index holds fewer entries than rows. + /// + /// A hand-built file whose index holds fewer entries than its rows fails the typed open with + /// `IndexSize` carrying both counts. #[test] fn archive_refuses_an_index_missing_a_row() { // Hand-crafted geometry the writer refuses to produce: two rows whose index carries one diff --git a/libs/@local/graph/atlas/src/salt/fit/prepare/instance.rs b/libs/@local/graph/atlas/src/salt/fit/prepare/instance.rs index 3a0172dad0c..d9ca189e52f 100644 --- a/libs/@local/graph/atlas/src/salt/fit/prepare/instance.rs +++ b/libs/@local/graph/atlas/src/salt/fit/prepare/instance.rs @@ -1,9 +1,9 @@ //! The relation-instance spool, the edge drain's working artifact. //! -//! The drain consumes the edge stream exactly once, before the policy table exists, so it spools -//! every `(edge, relation)` reading into a scratch file and the relation stage maps that file back -//! once the policy table resolves. The spool is transient by design. It lives in the run's -//! [`ScratchDirectory`], and only the run that wrote it consumes it. It never publishes. +//! The drain consumes the edge stream exactly once, before the policy table exists. It therefore +//! spools every `(edge, relation)` reading into a scratch file, and the relation stage maps that +//! file back once the policy table resolves. The spool is transient by design. It lives in the +//! run's [`ScratchDirectory`], and only the run that wrote it consumes it. It never publishes. #![expect(clippy::empty_enums, reason = "zerocopy uses them in the derive")] use core::fmt; @@ -26,10 +26,11 @@ use crate::{ /// One spooled `(edge, relation)` reading. /// /// Each confidence stores its value beside a presence bit (bit 0 link, bit 1 source, bit 2 target, -/// the attraction file's score vocabulary), so the absent-score distinction survives the spool. -// The confidence fields carry their domains in their types, so the mapping's parse refuses an -// out-of-domain value. The row ids, the presence bits and the multiplicity are unconstrained -// primitive encodings. +/// the attraction file's score vocabulary), and the absent-score distinction therefore survives +/// the spool. +// The confidence fields carry their domains in their types, and the mapping's parse therefore +// refuses an out-of-domain value. The row ids, the presence bits and the multiplicity are +// unconstrained primitive encodings. #[derive( Copy, Clone, @@ -52,8 +53,11 @@ pub(crate) struct InstanceRecord { } impl InstanceRecord { + /// Presence bit of the link confidence in `scored`. const LINK: u32 = 1; + /// Presence bit of the source confidence in `scored`. const SOURCE: u32 = 1 << 1; + /// Presence bit of the target confidence in `scored`. const TARGET: u32 = 1 << 2; /// Encodes one reading of `edge` under `relation`. @@ -209,7 +213,7 @@ impl InstanceSpool { /// # Panics /// /// This panics when the file length disagrees with the pushed count. One run writes and - /// consumes the spool, so a mismatch is a program bug rather than a data error. + /// consumes the spool, and a mismatch is therefore a program bug rather than a data error. #[tracing::instrument(skip_all)] pub(crate) fn map(&self) -> io::Result { if self.count == 0 { @@ -240,8 +244,14 @@ pub(crate) struct MappedInstances { impl MappedInstances { /// Views the spooled readings, in drain order. /// - /// The parse validates each confidence's domain as the bytes are read, so a reading never - /// decodes to a value its type refuses. + /// The parse validates each confidence's domain as the bytes are read, and a reading therefore + /// never decodes to a value its type refuses. + /// + /// # Panics + /// + /// This panics when the mapped bytes do not parse as records with every confidence in its + /// domain. The run wrote the spool from typed values, and a mismatch is therefore a program + /// bug rather than a data error. #[must_use] pub(crate) fn records(&self) -> &[InstanceRecord] { let Some(map) = &self.map else { diff --git a/libs/@local/graph/atlas/src/salt/fit/prepare/mod.rs b/libs/@local/graph/atlas/src/salt/fit/prepare/mod.rs index f43c5f34c93..7a0e1e23b1b 100644 --- a/libs/@local/graph/atlas/src/salt/fit/prepare/mod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/prepare/mod.rs @@ -1,7 +1,7 @@ //! Prepare representations: persist the working artifacts of one generation. //! -//! The stage consumes dataset streams once and writes the artifacts every later stage reads, so -//! downstream stages address rows in mapped files instead of re-consuming the source. +//! The stage consumes dataset streams once and writes the artifacts every later stage reads. +//! Downstream stages therefore address rows in mapped files instead of re-consuming the source. //! [`write_node_representations`] covers the node half: the `f32[N, 512]` representation matrix, //! row-aligned with the node stream, with the node ids and direct types collected into //! [`NodeColumns`] in the same pass, and [`norm::spot_check`] certifies the written rows' source @@ -72,12 +72,12 @@ pub(crate) struct NodeColumns { /// /// Collects the node ids and direct types in the same pass. /// -/// Row `i` of the written matrix is the embedding of node row `i`, so the matrix is row-aligned -/// with every artifact keyed by [`NodeRowId`], and entry `i` of the -/// returned columns is that row's source id and type set. The publish step computes the finished -/// file's repository digest. +/// Row `i` of the written matrix is the embedding of node row `i`. The matrix is therefore +/// row-aligned with every artifact keyed by [`NodeRowId`], and entry `i` of the returned columns +/// is that row's source id and type set. The publish step computes the finished file's repository +/// digest. /// -/// Every node issues one write; wrap a raw [`File`](std::fs::File) in a +/// Every node issues one write, and a raw [`File`](std::fs::File) belongs inside a /// [`BufWriter`](io::BufWriter). /// /// # Errors diff --git a/libs/@local/graph/atlas/src/salt/fit/prepare/norm.rs b/libs/@local/graph/atlas/src/salt/fit/prepare/norm.rs index 0750a481f4e..170c95b5623 100644 --- a/libs/@local/graph/atlas/src/salt/fit/prepare/norm.rs +++ b/libs/@local/graph/atlas/src/salt/fit/prepare/norm.rs @@ -8,9 +8,10 @@ //! contract. One defective sampled row refutes the contract, and the evidence lists every defective //! sampled row with its diagnosis. //! -//! The check reads the rows a mapped `f32[N, 512]` artifact yields, so it faults only the sampled -//! pages. The check visits sampled rows in ascending order, which keeps a cold mapping's faults -//! forward. The sample is small and each row is one kernel pass, so the check runs serially. +//! The check reads the rows a mapped `f32[N, 512]` artifact yields, and it therefore faults only +//! the sampled pages. The check visits sampled rows in ascending order, which keeps a cold +//! mapping's faults forward. The sample is small and each row is one kernel pass, and the check +//! therefore runs serially. use core::{error::Error, fmt}; @@ -30,14 +31,19 @@ use crate::{ // prefix's share of the parent vector's unit energy (1/6 for energy spread evenly over 3072 // components), thousands of tolerances from one. The sampling budget matches the crate's other // acceptance checks: 688 rows certify a 1% defect rate at 99.9% confidence when all pass. +/// The default admitted deviation of a row's squared norm from one. const DEFAULT_TOLERANCE: DPositive = d_positive!(1e-4); +/// The default defect rate the sample size certifies. const DEFAULT_DEFECT_RATE: OpenUnitFraction = open_unit_fraction!(0.01); +/// The default confidence level of the certification. const DEFAULT_CONFIDENCE: OpenUnitFraction = open_unit_fraction!(0.999); // The magnitudes the tolerance sits between: the squared-norm perturbation of narrowing one row to // f32, and the deviation of the nearest real failure mode (a prefix that kept 1/6 of the parent // vector's unit energy). +/// The squared-norm perturbation of narrowing one `f64`-normalized row to `f32`. const NARROWING_PERTURBATION: DPositive = d_positive!(1e-6); +/// The squared-norm deviation of a prefix that kept one sixth of the parent vector's unit energy. const UNRENORMALIZED_DEVIATION: DPositive = d_positive!(1.0 - 1.0 / 6.0); const _: () = assert!( DEFAULT_TOLERANCE > NARROWING_PERTURBATION && DEFAULT_TOLERANCE < UNRENORMALIZED_DEVIATION @@ -138,7 +144,7 @@ impl Error for SpotCheckError {} /// Verifies the representation contract on a uniform sample of rows. /// -/// `embeddings` holds the persisted representations in row order; a mapped `f32[N, 512]` artifact +/// `embeddings` holds the persisted representations in row order. A mapped `f32[N, 512]` artifact /// yields the slice directly. The check covers a corpus smaller than the sample size exhaustively, /// making the certification exact. /// diff --git a/libs/@local/graph/atlas/src/salt/fit/prepare/tests.rs b/libs/@local/graph/atlas/src/salt/fit/prepare/tests.rs index d004657ecfb..07ec1e2c031 100644 --- a/libs/@local/graph/atlas/src/salt/fit/prepare/tests.rs +++ b/libs/@local/graph/atlas/src/salt/fit/prepare/tests.rs @@ -46,10 +46,12 @@ impl Matrix { Self { storage, rows } } + /// Mutable access to row `row`'s components. fn row_mut(&mut self, row: usize) -> &mut [f32] { &mut self.storage.as_array_mut()[row * PROJECTOR_DIMENSIONS..][..PROJECTOR_DIMENSIONS] } + /// The resident rows as an aligned row-indexed slice. fn view(&self) -> &IdSlice> { IdSlice::from_raw( AlignedVecN::from_slice(&self.storage.as_array()[..self.rows * PROJECTOR_DIMENSIONS]) @@ -81,6 +83,10 @@ fn nodes_only(embeddings: Vec>) -> MemoryDataset MemoryDataset::new(nodes, vec![], vec![], HashMap::new(), HashMap::new()) } +/// Writes a three-node dataset and reads the `f32` array file back bit for bit. +/// +/// Writing a three-node dataset produces an `f32` array file of shape `[3, PROJECTOR_DIMENSIONS]` +/// whose row `i` is node row `i`'s embedding bit for bit, with three ids and type lists. #[tokio::test] async fn representations_persist_row_aligned_with_the_node_stream() { let embeddings = vec![unit(0), unit(7), unit(511)]; @@ -113,6 +119,7 @@ async fn representations_persist_row_aligned_with_the_node_stream() { } } +/// An empty dataset persists as a header-only array file with no ids or types. #[tokio::test] async fn empty_dataset_seals_an_empty_matrix() { let dataset = nodes_only(vec![]); @@ -141,7 +148,7 @@ fn spot_check_certifies_a_normalized_matrix() { ) .expect("a non-empty matrix under a sound budget checks"); - // The corpus sits far below the sample size, so the check is exhaustive and the certification + // The corpus lies far below the sample size: the check is exhaustive and the certification // exact. assert_eq!(check.rows, 5); assert_eq!(check.sampled_rows, 5); @@ -152,6 +159,7 @@ fn spot_check_certifies_a_normalized_matrix() { assert_eq!(check.confidence, open_unit_fraction!(0.999)); } +/// A NaN row and an over-norm row both appear in the check's defects, and the check fails. #[test] fn spot_check_lists_every_defective_sampled_row() { let mut matrix = Matrix::units(6); @@ -181,6 +189,10 @@ fn spot_check_lists_every_defective_sampled_row() { ); } +/// Passes a `4e-5` norm residual at the default tolerance and fails it a hundredfold tighter. +/// +/// A row `4e-5` over unit norm passes the default tolerance and fails a hundredfold tighter one, +/// which names it as a defect. #[test] fn spot_check_honours_a_configured_tolerance() { let mut matrix = Matrix::units(4); @@ -212,6 +224,7 @@ fn spot_check_honours_a_configured_tolerance() { ); } +/// A 700-row matrix under the default budget samples 688 rows and passes. #[test] fn spot_check_samples_large_matrices() { let matrix = Matrix::units(700); @@ -228,6 +241,7 @@ fn spot_check_samples_large_matrices() { assert!(check.passes()); } +/// `spot_check` over no rows fails with `SpotCheckError::Empty`. #[test] fn spot_check_rejects_an_empty_matrix() { assert_eq!( @@ -252,6 +266,10 @@ fn spool_root(name: &str) -> camino::Utf8PathBuf { dir } +/// Round-trips every presence combination of the three optional confidences. +/// +/// Every presence combination of the three optional confidences survives the record encode and +/// decode, along with the row ids and multiplicity. #[test] fn instance_records_round_trip_their_option_confidences() { // Every presence combination of the three scores survives the @@ -350,6 +368,7 @@ fn spool_round_trips_through_its_scratch_file() { assert_eq!(read_back, expected); } +/// A spool sealed without records has count zero and maps to no records. #[test] #[expect( clippy::significant_drop_tightening, @@ -367,6 +386,10 @@ fn empty_spool_maps_to_zero_readings() { assert!(mapped.records().is_empty()); } +/// Parses a valid record and fails one with an out-of-domain link confidence. +/// +/// A valid record parses, and planting an out-of-domain value in its link-confidence lane makes the +/// parse fail. #[test] #[expect( clippy::little_endian_bytes, @@ -386,8 +409,8 @@ fn spool_record_parse_refuses_an_out_of_domain_confidence() { Ok(_) ); - // The link confidence sits behind the record's row ids: `repr(C)` places it after the - // edge, relation, source, and target columns, one u64 lane each. + // The link confidence follows the record's row ids: `repr(C)` places it after the edge, + // relation, source, and target columns, one u64 lane each. let link_offset = 4 * size_of::(); let mut bytes = record.as_bytes().to_vec(); bytes[link_offset..link_offset + size_of::()].copy_from_slice(&2.0_f64.to_le_bytes()); diff --git a/libs/@local/graph/atlas/src/salt/fit/tests.rs b/libs/@local/graph/atlas/src/salt/fit/tests.rs index 1742ecbea73..54e9b1bea9f 100644 --- a/libs/@local/graph/atlas/src/salt/fit/tests.rs +++ b/libs/@local/graph/atlas/src/salt/fit/tests.rs @@ -86,9 +86,12 @@ use crate::{ }, }; +/// Row count of the fixture corpus. const NODES: usize = 48; +/// Landmark capacity the fixture selection runs at. const LANDMARKS: u32 = 8; +/// A fresh per-process scratch directory under the system temp dir, cleared if it already exists. fn scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -100,7 +103,9 @@ fn scratch(name: &str) -> Utf8PathBuf { dir } -/// A unit-norm pseudo-random representation for node `row`. +/// A unit-norm pseudo-random representation drawn from `rng`. +/// +/// Every component is uniform on `[-0.5, 0.5)`. Double-precision arithmetic normalizes the vector. fn representation(rng: &mut Xoshiro256PlusPlus) -> BoxedVecN { let mut components = [0.0_f32; PROJECTOR_DIMENSIONS]; for component in &mut components { @@ -124,15 +129,17 @@ fn representation(rng: &mut Xoshiro256PlusPlus) -> BoxedVecN MemoryDataset { dataset_with_edge_confidences([(None, None, None); 2]) } /// The base corpus with both edges' `(link, source, target)` confidence readings supplied. /// -/// [`dataset`] is the unscored form the other fit tests share. The readings are a parameter so a -/// corpus that violates the confidence contract differs from the clean one in nothing else. -/// Every row carries display text, so the staged identity artifacts persist real payloads. +/// [`dataset`] is the unscored form the other fit tests share. Taking the readings as a parameter +/// lets a corpus that violates the confidence contract differ from the clean one in nothing else. +/// Every row carries display text, and the staged identity artifacts therefore persist real +/// payloads. fn dataset_with_edge_confidences( readings: [( Option, @@ -288,10 +295,10 @@ fn classifier_fit_echo_round_trips_every_knob() { /// The fixture inserts `relative_cg_residual_tolerance`, `maximum_cg_iterations`, /// `maximum_consecutive_rejections`, `maximum_hvp_requests`, `maximum_objective_requests`, /// `maximum_gradient_requests` and `maximum_row_traversals` verbatim into an otherwise-default -/// solver echo, so the decode pins unknown-field tolerance for the current record shape. The -/// decoder ignores an unknown solver field instead of rejecting the record. The first two retired -/// with the inner CG recurrence, the third with the rejection budget radius underflow always -/// preceded, and the last four with the work budgets the iteration structure already bounds. +/// solver echo. The decode pins unknown-field tolerance for the current record shape: the decoder +/// ignores an unknown solver field instead of rejecting the record. None of the seven names a +/// field of the current [`SolverConfig`], whose loop has no inner CG recurrence, no +/// consecutive-rejection budget, and no work limit besides the outer-iteration cap. #[test] fn config_echo_decodes_the_retired_solver_knobs() { #[derive(Debug, serde::Serialize, serde::Deserialize)] @@ -358,12 +365,7 @@ fn config_echo_requires_every_setting() { } } -/// The budget echoes as its floor object, and the retired clamp's bare array still decodes. -/// -/// The bare four-constant array `[positive, total, floor, epsilon]` is the exact shape every -/// generation published under the enforcing clamp carries. The fixture pins the ratified constants -/// verbatim, so this decode is the standing witness that those manifests parse under the current -/// binary. The decode keeps the floor and discards the retired clamp coefficients. +/// Preserves a non-default diagnostic floor in the projector configuration. #[test] fn budget_echo_writes_the_floor_and_decodes_the_retired_clamp_array() { #[derive(Debug, serde::Serialize, serde::Deserialize)] @@ -410,6 +412,7 @@ fn config_echo_validates_the_group_budget() { assert!(error.to_string().contains("fraction in (0, 1]")); } +/// Builds a landmark-baseline configuration that skips projector training. fn config() -> FitConfig { FitConfig { seed: 7, @@ -428,12 +431,14 @@ fn config() -> FitConfig { } } -/// A deterministic classifier fitted from a synthetic corpus. +/// Fits a deterministic classifier from a synthetic corpus. /// -/// The supplied model input of every fixture fit. +/// The supplied model input of every fixture fit except the annotation-corpus fit, which fits its +/// own. fn fixture_classifier() -> Classifier { const ROWS: usize = 4; - // Coprime to the dimension, so no two corpus rows repeat. + // The pattern length 13 is coprime to `CANONICAL_DIMENSIONS`: the cycle enters each of the four + // rows at a different phase, and no two corpus rows repeat. const PATTERN: [f32; 13] = [ -0.75, -0.625, -0.5, -0.375, -0.25, -0.125, 0.0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, ]; @@ -491,7 +496,8 @@ fn fixture_input() -> ClassifierInput { /// Asserts the complete-generation fixture's snapshot counts and its multiplicity histogram. /// -/// The dataset streamed two single-typed edges, so the histogram is a single k = 1 entry. +/// The dataset streamed two single-typed edges: the histogram is the single entry `[2]`, the edge +/// count at multiplicity one. #[track_caller] fn assert_complete_generation_snapshot(repository: &SaltRepository) { assert_eq!(repository.metadata.snapshot.nodes, NODES as u64); @@ -550,9 +556,9 @@ async fn fit_publishes_a_complete_generation() { serde_json::from_slice(&document).expect("the document should deserialize"); assert_digests_match(&published_path, &repository); - // The call supplied no verdicts, so the manifest records the absence. + // With no verdicts supplied, the manifest records the absence. assert!(repository.files.reviewed_verdicts.is_none()); - // The call supplied the classifier, so the manifest records the source digest and stages no + // With the classifier supplied, the manifest records its source digest and stages no // annotation artifacts. assert!(repository.files.annotation_corpus.is_none()); assert!(repository.files.annotation_embeddings.is_none()); @@ -577,7 +583,7 @@ async fn fit_publishes_a_complete_generation() { assert!(repository.metadata.reproducibility.prior.is_none()); assert_eq!(repository.metadata.placement, Placement::LandmarkBaseline); - // The recorded evidence passed - a published generation implies it. + // A published generation implies the recorded evidence passed. assert!(repository.metadata.evidence.norm.passes()); assert_eq!( repository.metadata.evidence.recall.admission(), @@ -614,6 +620,8 @@ async fn fit_publishes_a_complete_generation() { ); } +/// Asserts every baseline coordinate is bit-equal to its landmark's layout coordinate. +/// /// Asserts every row's baseline coordinate is bit-equal to its assigned landmark's layout /// coordinate in the published skeleton. #[track_caller] @@ -710,8 +718,10 @@ fn assert_identities_translate(published: &Utf8Path) { /// Asserts the published postings against the fixture by hand. /// /// Every position carries its row's direct type, only the link type names a parent, and the -/// evidence records the representation split - at 48 points the dense threshold is one member and a -/// dense run costs two words, so both node types go dense and the empty link type stays a list. +/// evidence records the representation split. At 48 points a dense set costs two words (the header +/// and one word of bits, 16 bytes) against four bytes per list member, and a type goes dense from +/// five members up: both node types (24 members each) go dense and the empty link type stays a +/// list. #[track_caller] fn assert_postings_read_back(published: &Utf8Path, repository: &SaltRepository) { let postings = PostingsArchive::new( @@ -825,6 +835,11 @@ async fn policy_artifacts_publish_and_read_back() { ); } +/// Checks the published LOD columns: permutations, Morton order, wire range and metadata. +/// +/// The published LOD columns hold mutually inverse total permutations, a Morton column covering +/// every row, wire coordinates inside `[-1, 1]` that keep landmark-coincident rows coincident, and +/// metadata recording the histogram and the incident-degree ranking origin. #[tokio::test] async fn lod_columns_publish_in_base_order() { let root = GenerationRoot::new(scratch("lod")).expect("the root should open"); @@ -918,6 +933,10 @@ async fn lod_columns_publish_in_base_order() { assert_eq!(repository.metadata.ranking, RankingOrigin::IncidentDegree); } +/// Publishes a supplied reviewed-verdicts document byte for byte under its hash. +/// +/// A supplied reviewed-verdicts document publishes byte for byte under its manifest entry, bound to +/// the supplied file's hash, and still parses through the trainer's reader. #[tokio::test] async fn supplied_verdicts_publish_verbatim() { let root = GenerationRoot::new(scratch("verdicts")).expect("the root should open"); @@ -1261,10 +1280,12 @@ async fn prior_generation_seeds_reuse_and_retention() { assert_eq!(repository.metadata.evidence.cards.embedded, 0); // Equal corpus, seed, and config draw equal selection priorities, - // and retention prefers prior landmarks among equal candidates: - // the selection reproduces and every selected row is a retained - // one. The skeleton is then bit-identical, which the recorded - // digests certify. + // and the second selection picks the same rows: the retention + // phase reserves `ceil(capacity · retained_fraction)` seats (2 of + // the 8) for prior landmarks, and the rows filling them would win + // the free fill by priority regardless. Every selected row is + // therefore a retained one, and the skeleton is bit-identical, + // which the recorded digests certify. assert_eq!(repository.metadata.evidence.landmarks.retained, LANDMARKS); assert_eq!( repository.files.landmarks, @@ -1276,6 +1297,10 @@ async fn prior_generation_seeds_reuse_and_retention() { ); } +/// Publishes a human override's exact distribution as the selected and attraction mix. +/// +/// A human override for the fixture's link type publishes its exact distribution as both the +/// selected and attraction mix at applicability one, and the metadata echoes the config verbatim. #[tokio::test] async fn override_supersedes_the_classifier() { let root = GenerationRoot::new(scratch("override")).expect("the root should open"); @@ -1321,8 +1346,8 @@ async fn override_supersedes_the_classifier() { .expect("the link type resolves"); // The override's distribution is the selected one, asserted with - // applicability 1, so the attraction mix passes it through - // unchanged. + // applicability 1: the attraction mix `a · p + (1 - a) · Overlay` + // passes it through unchanged. assert_eq!(policy.selected.coincident.to_bits(), 0.25_f64.to_bits()); assert_eq!(policy.selected.proximal.to_bits(), 0.5_f64.to_bits()); assert_eq!(policy.attraction.coincident.to_bits(), 0.25_f64.to_bits()); @@ -1374,14 +1399,19 @@ async fn equal_seeds_publish_equal_generations() { .await .expect("the second fit should publish"); - // The generation id digests the metadata document, which in turn digests every artifact, so - // equal ids certify byte-equal generations. The converse follows from the wired stage set, - // whose every member is deterministic by construction. The pipeline promises no such contract - // of its own. Determinism is best effort, and a stage under the training carve-out rescopes - // this assertion to the deterministic artifacts. + // The generation id is the digest of the metadata document, and the document records every + // artifact's digest: equal ids certify byte-equal generations. Equal generations follow from + // equal inputs here because every stage the baseline configuration wires is deterministic by + // construction. The pipeline promises no such contract of its own: the trainer's backend need + // not accumulate gradients deterministically (`projector::train::fit`), and a configuration + // that trains a projector narrows this assertion to the artifacts outside that carve-out. assert_eq!(first.id(), second.id()); } +/// A zero-norm node fails the norm check and leaves the root empty. +/// +/// A corpus with one zero-norm node fails the norm check, and the failed fit leaves the root empty +/// with no transient state. #[tokio::test] async fn defective_corpus_publishes_nothing() { let path = scratch("defective"); @@ -1456,13 +1486,13 @@ fn minimal_schedule() -> TrainingSchedule { .expect("the fixture schedule is valid") } -/// The projector fixture's training run. +/// Builds the projector fixture's training options. /// /// Short enough for a test, long enough that the boundary and every step run. The hidden -/// architecture shrinks while the representation width keeps the pipeline's contract, so a -/// forward or training step costs a fraction of the ratified model's; the publish boundary's own -/// certificates (`compute::projector::tests`) pin the bit-exact publish contracts, and these -/// fixtures certify the fit's composition. +/// architecture shrinks while the representation width keeps the pipeline's contract, and a +/// forward or training step therefore costs a fraction of the ratified model's. The publish +/// boundary's own certificates (`compute::projector::tests`) pin the bit-exact publish contracts, +/// and these fixtures certify the fit's composition. fn projector_options() -> ProjectorOptions { let mut options = ProjectorOptions::ratified(); options.architecture = Architecture { @@ -1501,9 +1531,9 @@ fn projector_options() -> ProjectorOptions { /// A reviewed-Proximal verdict covering the fixture link row. /// -/// The versioned URL names ontology id 2 in the memory corpus's own id space (`memory://2/`), so -/// the resolution reaches the link row carrying the Proximal force and the boundary measures its -/// radius from reviewed pairs. +/// The versioned URL names ontology id 2 in the memory corpus's own id space (`memory://2/`). The +/// resolution therefore reaches the link row carrying the Proximal force, and the boundary +/// measures its radius from reviewed pairs. fn proximal_link_verdicts() -> SuppliedVerdicts { let document = concat!( r#"{"pair_verdicts":[],"schema":"atlas-reviewed-verdicts/1","sources":{},"#, @@ -1514,6 +1544,10 @@ fn proximal_link_verdicts() -> SuppliedVerdicts { SuppliedVerdicts::from_bytes(document.as_bytes()).expect("the fixture document admits") } +/// Defaults a bare config to the projector placement under `ratified()` options. +/// +/// A bare config defaults to the projector placement under `ratified()` options, whose schedule +/// runs 20,000 steps with the boundary at 5,000. #[test] fn default_placement_is_the_trained_projector() { // The conditioned projector is the pipeline's architecture; a bare @@ -1544,10 +1578,11 @@ async fn forceless_projector_publishes_the_baseline_step() { let root = GenerationRoot::new(scratch("projector-vacuous")).expect("the root should open"); let dataset = dataset(); - // An Overlay override strips the fixture's link type of force. The boundary then freezes - // nothing, the lens provably never trains, and the run skips the ladder whole. - // The zero force makes the run vacuous by construction (`admit` sees no force at all), so the - // schedule trains nothing and one step certifies the same orchestration a longer run would. + // An Overlay override strips the fixture's link type of force. `admit` then finds no force at + // all and the run is vacuous by construction, with no frozen radius and no relation term. The + // run skips the ladder whole, and the one scheduled step trains with the relation term absent + // and every other configured family drawn and evaluated (semantic pairs, ordinary and hard + // repulsion, landmark anchors): it certifies the same orchestration a longer run would. let mut options = projector_options(); options.schedule = minimal_schedule(); let config = FitConfig { @@ -1622,8 +1657,8 @@ async fn forceless_projector_publishes_the_baseline_step() { /// The recorded salt re-derives from the document's own input sections, the draw counts replay /// over the published attraction index, and the measured body's strata census the whole /// candidate pool. The trained fixture holds 48 rows and the two full-force edges (0, 1) and -/// (2, 3), so the pair domain censuses 2 oriented pairs and both draw, while the 44 rows -/// outside the edges form the control pool and `m = n = 2` controls draw from it. +/// (2, 3). The pair domain censuses 2 oriented pairs and both draw. The 44 rows outside the edges +/// form the control pool, from which `m = n = 2` controls draw. #[track_caller] fn assert_paired_replay(published: &Utf8Path, repository: &SaltRepository) { let paired = repository @@ -1703,8 +1738,8 @@ async fn trained_lens_publishes_the_canonical_step_aligned() { let dataset = dataset(); // A Proximal override gives the link type full force, and the reviewed verdict names the - // link row in the corpus's own id space, so the boundary measures its radius from the - // reviewed pairs. + // link row in the corpus's own id space: the boundary measures its radius from the reviewed + // pairs. let options = projector_options(); let verdicts = proximal_link_verdicts(); let config = FitConfig { @@ -1765,8 +1800,8 @@ async fn trained_lens_publishes_the_canonical_step_aligned() { assert_eq!(ladder.canonical.get().to_bits(), 1.0_f32.to_bits()); assert_eq!(ladder.canonical_index, ladder.steps.len() - 1); - // The baseline step is its own frame; every recorded loss is a - // real measurement. + // The baseline step is its own frame: the identity alignment and + // zero movement against the baseline. assert_eq!(ladder.steps[0].alignment, Similarity::IDENTITY); assert_eq!( ladder.steps[0].baseline_movement.get().to_bits(), @@ -1776,8 +1811,8 @@ async fn trained_lens_publishes_the_canonical_step_aligned() { let canonical = &ladder.steps[ladder.canonical_index]; assert!(canonical.adjacent_movement > d_non_negative!(0.0)); - // The paired-movement readout lands beside the steps, and its salt and draw replay from - // the published document alone. + // The ladder evidence holds the paired-movement readout beside the steps, and its salt and + // draw replay from the published document alone. assert_paired_replay(&published_path, &repository); // The publish boundary's certificates (`compute::projector::tests`) pin the column's @@ -1922,7 +1957,7 @@ fn assert_cluster_shares_one_point( /// /// Every published artifact covers the row domain, and every published neighbour and selected /// landmark is a representation's first row. The rows of one duplicate cluster share one neighbour -/// list, one landmark ordinal, and one published coordinate bit for bit, so the distinct-domain +/// list, one landmark ordinal, and one published coordinate bit for bit: the distinct-domain /// training evidence and the full-domain column describe one field. #[tokio::test] async fn duplicate_rows_train_distinct_and_publish_the_row_domain() { @@ -1932,9 +1967,10 @@ async fn duplicate_rows_train_distinct_and_publish_the_row_domain() { // The Proximal override gives the link type full force and the // reviewed verdict freezes a measured boundary radius: the whole // trained path runs over the quotient's distinct domain. The - // byte-identical reviewed pairs measure a small Proximal quantile, - // so this corpus wants a Coincident radius below that measurement, - // the same ordering the composed energy demands of production. + // byte-identical reviewed pairs measure a small Proximal radius, + // and `RelationLens::energy` composes no energy unless the + // Coincident radius lies strictly below it. The lens therefore + // takes a 0.01 Coincident radius in place of the fixture's 0.5. let verdicts = proximal_link_verdicts(); let mut options = projector_options(); options.lens = RelationLens::new( @@ -2035,9 +2071,9 @@ async fn duplicate_rows_train_distinct_and_publish_the_row_domain() { /// The vacuous placement unblocks a Proximal corpus lacking reviewed coverage. /// -/// A corpus whose relations carry Proximal force refuses to train without reviewed coverage - and -/// the vacuous placement is exactly what unblocks it: the same configuration trains and publishes -/// with the relation evidence withheld. +/// A corpus whose relations carry Proximal force refuses to train without reviewed coverage. The +/// vacuous placement unblocks it: the same configuration trains and publishes with the relation +/// evidence withheld from the trainer. #[tokio::test] async fn vacuous_placement_trains_without_reviews() { let dataset = dataset(); @@ -2085,9 +2121,9 @@ async fn vacuous_placement_trains_without_reviews() { let root = GenerationRoot::new(scratch("vacuous-trains")).expect("the root should open"); let mut options = projector_options(); options.vacuous = true; - // The vacuous flag empties the attraction index before admission, so - // the run is vacuous by construction regardless of the schedule: one - // step certifies the same orchestration a longer run would. + // The vacuous flag hands the trainer an empty attraction index, and + // admission finds no force regardless of the schedule: one step + // certifies the same orchestration a longer run would. options.schedule = minimal_schedule(); let vacuous_config = FitConfig { placement: PlacementOptions::Projector(options), @@ -2115,7 +2151,7 @@ async fn vacuous_placement_trains_without_reviews() { serde_json::from_slice(&document).expect("the document should deserialize"); // The placement trained, no radius froze, and the untrained lens - // publishes the baseline step directly - no ladder to measure. + // publishes the baseline step directly: there is no ladder to measure. assert_eq!(repository.metadata.placement, Placement::Projector); let evidence = repository .metadata @@ -2162,12 +2198,12 @@ async fn canonical_condition_outside_the_schedule_publishes_nothing() { let root = GenerationRoot::new(&path).expect("the root should open"); let dataset = dataset(); - // 0.3 names no step of the schedule, so the configuration - // contradicts itself and the fit must refuse to publish. The - // membership is decidable from the options alone, so the refusal - // lands before a single training step: the run's cost is the - // stages ahead of the placement. The reviewed verdict keeps the - // canonical mismatch as the configuration's only defect. + // 0.3 names no step of the schedule: the configuration contradicts + // itself and the fit must refuse to publish. The membership is + // decidable from the options alone, and the refusal precedes the + // first training step: the run's cost is the stages ahead of the + // placement. The reviewed verdict keeps the canonical mismatch as + // the configuration's only defect. let mut options = projector_options(); options.ladder.canonical = non_negative!(0.3); let verdicts = proximal_link_verdicts(); @@ -2331,10 +2367,6 @@ fn assert_adjacency_reads_back(published: &Utf8Path) { ); } -/// Asserts the published attraction index against the [`relation_dataset`] readings. -/// -/// Relation 2 retains three instances and relation 3 retains one, since the self-loop reading -/// carries no force and the drain discards it. The overridden weights and confidence provenance /// Reads a typed id column into the fixture tests' `u32` vocabulary. fn column_ids(file: &ArrayFile) -> Vec where @@ -2352,7 +2384,11 @@ where .collect() } -/// stay intact. +/// Asserts the published attraction index against the [`relation_dataset`] readings. +/// +/// Relation 2 retains three instances and relation 3 retains one: the self-loop reading carries no +/// force and the index build drops it. The overridden weights and confidence provenance stay +/// intact. #[track_caller] fn assert_attraction_reads_back(attraction: &AttractionArchive) { assert_eq!(attraction.rows(), NODES as u64); @@ -2367,8 +2403,8 @@ fn assert_attraction_reads_back(attraction: &AttractionArchive ArchivedOntologyTypeUuid { ArchivedOntologyTypeUuid::from_url(&url) } +/// Resolves store-identity verdicts against held versions when the version matches exactly. +/// +/// Against a column of store-identity versions, only the verdict reviewed at an exact held version +/// resolves. A foreign-store verdict and one at another version of a held base URL stay unresolved. #[test] fn store_identity_verdicts_resolve_by_reviewed_version() { // The staged column holds the generation's own type versions. @@ -2568,8 +2614,8 @@ fn store_identity_verdicts_resolve_by_reviewed_version() { // The fixture supplies one verdict from a foreign store, one reviewed at a version the column // does not hold, and one reviewed at a version it does. Only the exact reviewed version - // resolves, because versions are immutable and distinct, so a verdict for another version of - // the same base URL is evidence about a different card. + // resolves: versions are immutable and distinct, and a verdict for another version of the same + // base URL is evidence about a different card. let document = concat!( r#"{"pair_verdicts":[],"schema":"atlas-reviewed-verdicts/1","sources":{},"#, r#""type_verdicts":["#, @@ -2594,6 +2640,10 @@ fn store_identity_verdicts_resolve_by_reviewed_version() { assert_eq!(resolution.resolved[0].placement, PlacementClass::Proximal); } +/// Resolves a `memory://` verdict against a plain-number identity column. +/// +/// Against a plain-number identity column, a store-identity verdict stays unresolved while a +/// `memory://` verdict resolves to the row its authority names with its placement class. #[test] fn plain_number_corpus_resolves_the_memory_scheme() { // A plain-number column derives ids from the memory scheme alone: diff --git a/libs/@local/graph/atlas/src/salt/fit/verdicts/mod.rs b/libs/@local/graph/atlas/src/salt/fit/verdicts/mod.rs index 276b9257d90..38f71d382cb 100644 --- a/libs/@local/graph/atlas/src/salt/fit/verdicts/mod.rs +++ b/libs/@local/graph/atlas/src/salt/fit/verdicts/mod.rs @@ -3,12 +3,13 @@ //! The caller supplies a reviewed-verdicts document beside the corpus rather than deriving it, //! which puts the document in the same input category as the policy override table. //! [`SuppliedVerdicts`] runs the document's whole wire contract at construction through the verdict -//! reader and keeps the exact wire bytes, so the staged artifact is byte-identical to the supplied -//! file and the digest computed here is the supplied file's identity. +//! reader and keeps the exact wire bytes. The staged artifact is therefore byte-identical to the +//! supplied file, and the digest computed here is the supplied file's identity. //! -//! The fit carries the document into the generation without acting on it: fan-out from verdicts to -//! row pairs happens at the trainer's phase boundary, which consumes the staged artifact like every -//! other training input. +//! The fit stages the document verbatim and, in the placement stage, resolves its verdicts against +//! the staged ontology identity column into the corpus row domain. The trainer's phase boundary +//! consumes the resolved verdicts, fanning them out to row pairs over the attraction index when it +//! freezes the Proximal radius. use core::{error::Error, fmt}; use std::io; @@ -54,9 +55,10 @@ impl Error for SupplyError { /// One validated reviewed-verdicts document with its exact wire bytes. /// -/// A value of this type is admissible by existence: construction validated the document, so a fit -/// holding one stages the bytes verbatim and binds the digest without any further check. -/// Construction rejects a document that would fail admission before the fit spends anything. +/// A value of this type is admissible by existence. Construction validated the document, and a +/// fit holding one therefore stages the bytes verbatim and binds the digest without any further +/// check. Construction rejects a document that would fail admission before the fit spends +/// anything. #[derive(Debug, Clone, PartialEq)] pub(crate) struct SuppliedVerdicts { bytes: Box<[u8]>, diff --git a/libs/@local/graph/atlas/src/salt/fit/verdicts/tests.rs b/libs/@local/graph/atlas/src/salt/fit/verdicts/tests.rs index f2c6bc217df..396398b20cc 100644 --- a/libs/@local/graph/atlas/src/salt/fit/verdicts/tests.rs +++ b/libs/@local/graph/atlas/src/salt/fit/verdicts/tests.rs @@ -15,6 +15,7 @@ use crate::{ const DOCUMENT: &str = r#"{"pair_verdicts":[],"schema":"atlas-reviewed-verdicts/1","sources":{"cards.jsonl":"2a9934acae8bf210b6a3428e553b1bcc0e220a4de113940782cd573da1ea4f4b"},"type_verdicts":[{"class":"proximal","relation":"hash:https://hash.ai/@h/types/entity-type/delivers/","reviewer":"Bilal Mahmoud","versioned_url":"https://hash.ai/@h/types/entity-type/delivers/v/3"}]} "#; +/// A fresh per-process scratch directory named `name` under the system temp dir. fn scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -27,6 +28,10 @@ fn scratch(name: &str) -> Utf8PathBuf { dir } +/// Keeps the document bytes verbatim in `from_bytes` and hashes exactly those bytes. +/// +/// `SuppliedVerdicts::from_bytes` keeps the document bytes verbatim, trailing newline included, and +/// identifies them by the SHA-256 of those wire bytes. #[test] fn construction_preserves_bytes_and_binds_their_digest() { let supplied = SuppliedVerdicts::from_bytes(DOCUMENT.as_bytes()) @@ -43,6 +48,10 @@ fn construction_preserves_bytes_and_binds_their_digest() { assert_eq!(supplied.hash(), hasher.finalize()); } +/// Exposes the parsed document: one proximal type verdict and no pair verdicts. +/// +/// The supplied verdicts expose the parsed document: one proximal type verdict and no pair +/// verdicts. #[test] fn construction_exposes_the_validated_document() { let supplied = SuppliedVerdicts::from_bytes(DOCUMENT.as_bytes()) @@ -64,6 +73,10 @@ fn contract_violation_is_rejected_at_supply() { ); } +/// Reads a written document verbatim and maps a missing file and malformed JSON to their variants. +/// +/// `open` reads a written document verbatim, reports a missing file as `Io`, and malformed JSON as +/// `Invalid`. #[test] fn open_reads_the_file_and_reports_both_failure_shapes() { let dir = scratch("open"); diff --git a/libs/@local/graph/atlas/src/salt/importance/mod.rs b/libs/@local/graph/atlas/src/salt/importance/mod.rs index ad3a24ed4a9..9969eb60f05 100644 --- a/libs/@local/graph/atlas/src/salt/importance/mod.rs +++ b/libs/@local/graph/atlas/src/salt/importance/mod.rs @@ -1,14 +1,12 @@ -//! Importance signals: the configured column behind the delivery ranking. +//! Per-node importance scores for coarse delivery selection. //! -//! The base delivery order ranks rows by importance first ([`crate::salt::lod::rank`]), so the -//! importance column decides what a zoomed-out tile shows. [`ImportanceSignal`] is the derivation -//! trait: an implementation turns published generation artifacts into one `f32[N]` column, and -//! [`RankingConfig`] selects which one a fit runs - adding a signal is one implementation plus one -//! variant, and the exhaustive matches carry it into the config echo and the metadata origin -//! marker. +//! [`ImportanceSignal`] derives the primary sort key for [`crate::salt::lod::rank::Ranking`]. +//! Higher scores receive earlier ranks, affecting which points represent coarse cells. +//! [`RankingConfig`] selects a constant signal or incident degree. //! -//! Every signal is a pure function of published artifacts and the configuration: equal generations -//! derive equal columns, so the ranking stays reproducible from the manifest alone. +//! Both signals reproduce their columns from equal inputs. Full ranking replay additionally depends +//! on the priority and identity columns, seed, and sorting contract documented by +//! [`crate::salt::lod::rank::Ranking::new`]. use hashql_core::id::IdVec; @@ -17,7 +15,7 @@ use crate::{identity::NodeRowId, salt::adjacency::Adjacency}; #[cfg(test)] mod tests; -/// Selects the importance signal of one fit. +/// The importance signal selected for a fit. /// /// The manifest echoes the variant and the metadata's ranking origin mirrors it, so a published /// generation names the signal its delivery order ran under. @@ -25,9 +23,10 @@ mod tests; pub(crate) enum RankingConfig { /// A constant column. /// - /// The delivery order reduces to the seeded identity tiebreak, a deterministic unbiased sample. + /// With equal priority scores, ranking uses the seeded identity hash. This introduces no degree + /// preference and does not guarantee an unbiased sample. ConstantColumns, - /// Incident degree over the adjacency: hub entities deliver first. + /// Incident degree over the adjacency, favoring nodes with more incident edge slots. IncidentDegree, } @@ -37,27 +36,27 @@ const impl Default for RankingConfig { } } -/// One importance derivation over published generation artifacts. +/// A derivation of per-node ordinal importance scores. /// -/// The column is an ordinal sort key: the ranking consumes it through IEEE 754 `totalOrder` -/// comparisons alone, greater delivers first, and monotone transforms of a signal rank identically: -/// magnitudes are never read. Every entry is finite: under `totalOrder` a NaN fails nowhere, it -/// silently delivers its row at an extreme zoom. The column holds exactly one entry per node row. +/// Ranking compares scores with [`f32::total_cmp`], greater first, without using their magnitudes. +/// A transform preserves this comparison only if it preserves strict order and ties in the +/// resulting `f32` values. A merely nondecreasing transform can create ties, as can floating-point +/// rounding. /// -/// Implementations are deterministic: the column is a function of the artifacts and the -/// configuration alone, never of thread count or timing. -// The first signal whose entries are not finite by construction (a learned score read from an -// artifact) validates them behind a column newtype at its own boundary. The constant and -// integer-cast signals prove finiteness structurally. PERF: `derive` materializes one f32[N] column -// per fit (4 MB at a million rows) that the rank pass borrows and then drops. A lazy return cannot -// remove the column, because the rank comparator indexes by row. If the allocation ever shows in a -// fit profile, the fix is the house `derive_in(allocator)` variant. +/// Each derivation materializes a four-byte score per node for row-indexed comparisons. A million +/// rows require 4,000,000 score bytes, excluding container overhead. pub(crate) trait ImportanceSignal { /// Derives the importance column, one entry per node row. + /// + /// # Implementation Note + /// + /// Return exactly `rows` finite entries. Equal artifacts and configuration must produce equal + /// columns, never depending on thread count or timing. The ranking layer rejects no NaNs and + /// supplies no finiteness check for an implementation. fn derive(&self, rows: usize) -> IdVec; } -/// A signal that weighs every row the same. +/// A signal assigning positive zero to every row. #[derive(Debug, Copy, Clone)] pub(crate) struct ConstantImportance; @@ -67,20 +66,27 @@ impl ImportanceSignal for ConstantImportance { } } -/// Incident degree: each row's importance is the number of edge slots touching it. +/// A signal assigning each row its incident-edge slot count. +/// +/// Derivation takes O(N) time for N nodes. A self-loop counts twice, once in each direction, under +/// [`Adjacency`]'s degree contract. +/// +/// # Warning /// -/// Degrees read straight off the adjacency fenceposts, so the derivation is `O(N)` over an artifact -/// the fit already published. A self-loop occupies both slots of its node and counts twice, the -/// same reading the adjacency documents. Degrees convert to `f32` exactly up to 2^24 incident slots -/// per node; beyond that the ranking key rounds, which reorders only rows already within a quarter -/// of a percent of each other. +/// Integer degrees convert to `f32` exactly through 2²⁴. Larger degrees may round to the same +/// score. The conversion is nondecreasing, but distinct degrees can become ties resolved by +/// priority and identity hash. +/// +/// # Panics +/// +/// Derivation panics when `rows` differs from the adjacency's node count. #[derive(Debug, Copy, Clone)] pub(crate) struct DegreeImportance<'graph> { adjacency: &'graph Adjacency, } impl<'graph> DegreeImportance<'graph> { - /// Wraps the adjacency the degrees read from. + /// Selects the adjacency supplying the incident-edge counts. #[inline] #[must_use] pub(crate) const fn new(adjacency: &'graph Adjacency) -> Self { @@ -89,12 +95,6 @@ impl<'graph> DegreeImportance<'graph> { } impl ImportanceSignal for DegreeImportance<'_> { - /// Derives the degree column. - /// - /// # Panics - /// - /// This panics when `rows` disagrees with the adjacency's node domain, which the row-aligned - /// artifact contract excludes. #[expect( clippy::cast_precision_loss, reason = "degrees stay exactly representable in f32 far beyond any plausible fan-in; the \ diff --git a/libs/@local/graph/atlas/src/salt/importance/tests.rs b/libs/@local/graph/atlas/src/salt/importance/tests.rs index 06cbd346787..c3d4168684e 100644 --- a/libs/@local/graph/atlas/src/salt/importance/tests.rs +++ b/libs/@local/graph/atlas/src/salt/importance/tests.rs @@ -1,9 +1,7 @@ use super::{ConstantImportance, DegreeImportance, ImportanceSignal as _, RankingConfig}; use crate::{identity::NodeRowId, salt::adjacency::Adjacency}; -/// The five-node fixture. -/// -/// A parallel pair `0 → 1`, one `2 → 3`, a self-loop at 3, and node 4 untouched. +/// Builds five nodes with parallel edges, a self-loop, and an isolated node. fn fixture() -> Adjacency { let endpoints: [[NodeRowId; 2]; 4] = [ [NodeRowId::new(0), NodeRowId::new(1)], @@ -30,9 +28,8 @@ fn constant_column() { fn incident_degree_hand_count() { let adjacency = fixture(); - // A hand count over the fixture gives these degrees. Node 0 sends the parallel pair, node 1 - // receives it, node 2 sends once, node 3 receives twice and holds both slots of its self-loop, - // and node 4 touches nothing. + // nodes 0 and 1 each touch the parallel pair. Node 2 sends one edge. Node 3 receives that edge + // and has both slots of its self-loop: 1 + 2 = 3. Node 4 touches nothing. let column = DegreeImportance::new(&adjacency).derive(5); assert_eq!(*column.as_raw(), [2.0, 2.0, 1.0, 3.0, 0.0]); } diff --git a/libs/@local/graph/atlas/src/salt/knn/artifact.rs b/libs/@local/graph/atlas/src/salt/knn/artifact.rs index f719dd7f641..67b302ca96c 100644 --- a/libs/@local/graph/atlas/src/salt/knn/artifact.rs +++ b/libs/@local/graph/atlas/src/salt/knn/artifact.rs @@ -1,9 +1,9 @@ //! The k-NN table's published form: one sparse matrix file and its mapped reader. //! -//! A [`Knn`] table publishes as one [`crate::file::sprs`] file holding its -//! [`KnnMatrix`](super::table::KnnMatrix) verbatim. [`KnnArchive`] reopens the file over a -//! whole-file mapping and validates the table invariants once, so later pipeline stages read the -//! table without holding it on the heap. +//! A [`Knn`] table with a zero initial row pointer publishes as one [`crate::file::sprs`] file +//! holding its [`KnnMatrix`](super::table::KnnMatrix) verbatim. [`KnnArchive`] reopens the file +//! over a whole-file mapping and validates the table invariants. Views borrow the mapped matrix +//! regions without a heap copy. use core::{error::Error, fmt, marker::PhantomData}; use std::io; @@ -30,12 +30,14 @@ where /// Writes the table as a sparse matrix file. /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. - /// /// # Errors /// /// Returns an error when the underlying writer fails. + /// + /// # Panics + /// + /// This panics when the matrix's first row pointer is nonzero. [`Knn::new`] accepts such + /// matrices, but the sparse-file writer requires an initial zero. fn write_into(&self, write: impl io::Write) -> io::Result { let mut writer = Writer { accumulator: Sha256::new(), @@ -43,8 +45,8 @@ where }; write_matrix(&self.matrix(), &mut writer).map_err(|error| match error { WriteSprsError::Io(error) => error, - // A validated table is row-compressed, unsliced, and at - // least 2 x 2, so no non-IO write failure exists for it. + // validation establishes row-compressed storage and nonzero dimensions. The writer also + // requires an initial zero pointer, which Knn::new does not establish. error @ (WriteSprsError::Sliced | WriteSprsError::ZeroDimension { .. }) => { unreachable!("a validated table is writable: {error}") } @@ -93,10 +95,9 @@ impl Error for InvalidKnnFile { /// A published k-NN table opened over its mapped file. /// -/// Construction checks the table invariants once, so an open table only serves valid views; the -/// matrix regions stay in the page cache under memory pressure and off the heap. Each -/// [`view`](Self::view) re-checks the compressed-row structure ([`SprsFile::matrix`]'s contract), -/// so stages call it once and hold the view. +/// Construction checks the table's domain and neighbour invariants. Each [`view`](Self::view) +/// re-checks the compressed-row structure and value bit patterns under [`SprsFile::matrix`]'s +/// contract. Reuse that borrowed view for repeated row access. #[derive(Debug)] pub(crate) struct KnnArchive { file: SprsFile, @@ -111,8 +112,8 @@ where /// /// # Errors /// - /// Returns an error when the file does not hold the table's matrix layout or the matrix - /// violates a [`Knn`] invariant. + /// Returns [`InvalidKnnFile`] for an incompatible matrix layout or a violated [`Knn`] + /// invariant. pub(crate) fn new(file: SprsFile) -> Result { let matrix = file.matrix().map_err(InvalidKnnFile::Matrix)?; validate(matrix)?; @@ -123,7 +124,12 @@ where }) } - /// Borrows the validated table. + /// Borrows the table after rechecking its sparse structure and value bit patterns. + /// + /// # Complexity + /// + /// Each call takes O(rows + entries) time to validate the mapped regions. Reuse the resulting + /// view within an operation. #[must_use] pub(crate) fn view(&self) -> KnnView<'_, N> { let matrix = self diff --git a/libs/@local/graph/atlas/src/salt/knn/construction.rs b/libs/@local/graph/atlas/src/salt/knn/construction.rs index 7663a69f594..45b00f83015 100644 --- a/libs/@local/graph/atlas/src/salt/knn/construction.rs +++ b/libs/@local/graph/atlas/src/salt/knn/construction.rs @@ -1,14 +1,12 @@ //! k-nearest-neighbour list construction. //! -//! [`KnnConstruction`] separates what the pipeline needs (every row's nearest-neighbour list) from -//! how a constructor produces it. Both consumers read one [`NeighbourLists`] value. The recall spot -//! check reads sampled rows from it, and the persisted table slices its stored prefix from it, so -//! one construction at one width feeds both. +//! [`KnnConstruction`] produces one [`NeighbourLists`] value at a width sufficient for both recall +//! measurement and table storage. Keeping the wider lists lets a recall check inspect more +//! neighbours than the persisted prefix contains, without a second construction. //! -//! [`IndexConstruction`] adapts any [`NearestNeighboursIndex`] search backend to the seam. It -//! ingests every row and links the backend, then answers each row's list with one search. A -//! constructor that derives the lists directly without a search structure implements the trait -//! itself. +//! [`IndexConstruction`] produces these lists through a [`NearestNeighboursIndex`] search backend. +//! A constructor that derives lists directly can implement the trait without maintaining a search +//! index. use core::{ num::NonZero, @@ -32,16 +30,15 @@ use crate::{ /// Rows one batched loop covers between progress reports. /// -/// The insertion and the readback both report at this cadence. A corpus of a million rows draws a -/// couple of hundred observations, so a watching operator sees the counter move while the loops pay -/// one report per few thousand rows rather than one per row. +/// Insertion and readback each report once per 4,096 rows and at completion. A million-row loop +/// produces ceil(1,000,000 / 4,096) = 245 observations, retaining progress updates without per-row +/// reporting. const REPORT_CADENCE: usize = 4_096; -/// Whether a batched loop over `total` rows reports at `done` rows covered. +/// Tests whether `done` is a report-cadence multiple or the completed total. /// -/// A loop reports every [`REPORT_CADENCE`] rows and once more as its last row lands, so a corpus -/// below the cadence reports exactly once - at completion - and the last report of any corpus is -/// the complete one. +/// For a nonempty loop over `1..=total`, this reports every [`REPORT_CADENCE`] rows and at +/// completion. A loop below the cadence reports exactly once. const fn reports_at(done: usize, total: usize) -> bool { done.is_multiple_of(REPORT_CADENCE) || done == total } @@ -52,11 +49,11 @@ hashql_core::id::newtype! { pub(crate) struct NeighbourSlot(u32) } -/// Every row's nearest non-self neighbours, at one uniform width. +/// Per-row approximate neighbour lists at one uniform width. /// -/// Row-major storage: row `i` holds exactly [`width`](Self::width) entries in ascending `(distance, -/// id)` order, with distances on the `[0, 2]` cosine scale. The producing constructor guarantees -/// the entries, and the type only stores them. +/// Row `i` holds exactly [`width`](Self::width) entries. Producers must supply distinct non-self +/// neighbours in ascending `(distance, id)` order, with distances on the `[0, 2]` cosine scale. +/// This type checks the rectangular shape only. #[derive(Debug)] pub(crate) struct NeighbourLists { entries: IdMatrix>, @@ -66,7 +63,7 @@ impl NeighbourLists where N: Id, { - /// Wraps row-major entries whose per-row contract the producer established. + /// Stores row-major entries satisfying the producer's per-row contract. /// /// # Panics /// @@ -106,23 +103,26 @@ where } } -/// A constructor of every row's nearest-neighbour list. -/// -/// One construction serves one generation's rows: `embeddings` holds the l2-normalized projector -/// representations in node-row order, and the result holds each row's `width` nearest non-self -/// neighbours. The construction clamps a `width` at or beyond the corpus to every non-self row. -/// `rng` drives the constructor's randomized choices, so a seeded generator pins its sampling -/// streams. +/// A constructor of approximate neighbour lists over one row domain. pub(crate) trait KnnConstruction where N: Id, { + /// The failure [`construct`](Self::construct) reports. type Error; - /// Produces every row's `width` nearest non-self neighbours. + /// Produces every row's approximate non-self neighbours at a requested width. + /// + /// `embeddings` must hold l2-normalized projector representations in row order. The result + /// clamps `width` to the number of non-self rows. `rng` drives randomized choices. A seed + /// determines the random stream, without requiring deterministic parallel update order. + /// + /// # Implementation Note /// - /// The construction reports its batched loops and its named phases to `progress` as they - /// happen; it observes nothing the run acts on, so the lists are identical under any observer. + /// Implementations must satisfy the per-row [`NeighbourLists`] contract and report batched + /// loops and named phases through `progress`. Reports observe the construction without + /// supplying algorithm inputs. Parallel constructions can still vary between runs with the same + /// inputs and observer. /// /// # Errors /// @@ -138,20 +138,21 @@ where P: Progress + Sync; } -/// Adapts a [`NearestNeighboursIndex`] search backend to [`KnnConstruction`]. +/// A [`KnnConstruction`] that obtains lists from a [`NearestNeighboursIndex`]. /// /// The construction ingests every row and links the backend under `rng`, then queries each row's /// neighbours in parallel. The assembled lists are deterministic for a deterministic backend /// because the construction writes each row's results into that row's slot regardless of completion /// order. /// -/// The construction distrusts the backend's responses at the seam. A short result, a duplicate, or -/// a neighbour outside the row domain fails the construction. +/// A result of the wrong length, a duplicate neighbour or a neighbour outside the row domain fails +/// construction. Ordering, self-exclusion and distance semantics rely on the backend's trait +/// contract. #[derive(Debug)] pub(crate) struct IndexConstruction(I); impl IndexConstruction { - /// Wraps an empty backend. + /// Creates a list constructor from an empty search backend. pub(crate) const fn new(index: I) -> Self { Self(index) } @@ -183,11 +184,9 @@ where self.0 .insert_many(embeddings.iter().enumerate().map(|(row, components)| { - // A backend ingests the whole corpus inside one write - // transaction, so the insertion reports from the iterator - // it draws the rows through rather than from a batched - // call sequence - which would commit once per batch and - // let an observer change what the run costs. + // report iterator consumption without splitting insert_many into batches. + // Transactional backends can keep their single write transaction independently of + // the report cadence. let done = row + 1; if reports_at(done, rows) { progress.knn_insert(Batch { done, total: rows }); @@ -254,8 +253,7 @@ where slots.copy_from_slice(&found); - // Rows finish out of order, so the readback's position is - // how many rows have landed, never this row's index. + // count completed rows, never the row index: completion order is parallel. let done = covered.fetch_add(1, Ordering::Relaxed) + 1; if reports_at(done, rows) { progress.knn_readback(Batch { done, total: rows }); diff --git a/libs/@local/graph/atlas/src/salt/knn/descent.rs b/libs/@local/graph/atlas/src/salt/knn/descent.rs index 88a3ceb1e54..8751ef5ed85 100644 --- a/libs/@local/graph/atlas/src/salt/knn/descent.rs +++ b/libs/@local/graph/atlas/src/salt/knn/descent.rs @@ -1,41 +1,45 @@ //! NN-Descent k-nearest-neighbour list construction. //! -//! [`NnDescent`] derives every row's neighbour list directly, without a search structure: lists -//! start random and improve by local joins - each row introduces its current neighbours to each -//! other, and every introduction that beats a list's worst entry displaces it. The join converges -//! because similarity is locally transitive: when `a` and `b` are both near `x`, `a` and `b` are -//! likely near each other, so a row's own list is a high-yield candidate source for its -//! neighbours' lists. Random lists seed that feedback everywhere at once, and each accepted -//! displacement sharpens the candidate source for the next round; the audited cost on -//! generic-similarity corpora is far below the brute-force quadratic. +//! [`NnDescent`] starts every row with random non-self neighbours and improves its list by local +//! joins. Each row introduces its current neighbours to each other. A distinct candidate displaces +//! a list's worst entry only when its distance is strictly smaller. Neighbours of the same row can +//! be useful candidates for one another, which motivates the join as a search heuristic. Cosine +//! similarity is not transitive, and this heuristic does not guarantee exact nearest neighbours. //! //! # Shape of one iteration //! //! 1. **Candidate sampling.** Each row splits its list by the *new* flag (set on entries that have //! not yet participated in a join) and samples up to //! [`maximum_candidates`](NnDescentOptions::maximum_candidates) of each side. Sampling marks the -//! drawn new entries old, which keeps a later iteration from recomparing the same pairs. -//! 2. **Reversal.** The step transposes the sampled sets, so the rows listing a row also introduce -//! it. Sampling limits each reverse pool to the same cap so that rows which many others list -//! cannot join quadratically in their in-degree. +//! drawn new entries old. Old-old pairs are skipped in a local join, though the same pair can +//! meet again through another row or sampling path. +//! 2. **Reversal.** Transposing the sampled sets lets the rows listing a row introduce it too. +//! Sampling limits each reverse pool to the same cap so that rows which many others list cannot +//! join quadratically in their in-degree. //! 3. **Local join.** Every sampled new candidate of a row meets every other sampled candidate of //! that row, and the join offers each pair's cosine distance to both sides' lists. The cap //! bounds one row's join at O(cap²) distances regardless of degree skew. //! -//! Iteration stops when an iteration's accepted updates fall to -//! [`termination`](NnDescentOptions::termination) of the total entry count, or at -//! [`maximum_iterations`](NnDescentOptions::maximum_iterations). +//! For a finite non-negative [`termination`](NnDescentOptions::termination) rate τ and L stored +//! entries, the ideal update threshold is ceil(τ · L). The computed threshold converts L to f64, +//! multiplies by τ and applies ceil in f64, then casts to u64 with saturation. Rounding can change +//! the threshold from the ceiling of the real product. A finite rate can still overflow the f64 +//! product. +//! +//! Iteration stops when accepted updates are at most the computed threshold, or at +//! [`maximum_iterations`](NnDescentOptions::maximum_iterations). A low update rate measures +//! exhaustion of these sampled joins, not exact-neighbour recovery. //! //! # Determinism //! -//! Sampling streams derive from the seed alone: initialization and every per-row draw use a -//! generator keyed by `(seed, row, iteration)` through [`keyed_rng`]. Update application is -//! parallel and unordered, however, and a list's acceptances depend on the updates applied before -//! it, so converged lists need not agree between same-seed runs. +//! Initialization and every per-row draw use a generator keyed by `(seed, row, iteration)` through +//! [`keyed_rng`]. The resulting lists also depend on parallel update order: each acceptance changes +//! the threshold and membership for later offers. Converged lists need not agree between same-seed +//! runs. //! -//! The search backends share this property, because their parallel linking runs unordered the same -//! way. The recall spot check downstream arbitrates every construction, and that check alone -//! establishes the persisted table's contract. A replay of the construction never does. +//! The recall spot check measures approximation quality. Structural table validation separately +//! enforces neighbour counts, domains and distance bounds. Replaying construction never substitutes +//! for either check. use core::{ error::Error, @@ -66,11 +70,14 @@ use crate::{ // The candidate cap bounds one row's join work per iteration at O(cap²) distances regardless of // degree skew; 50 matches the widths this crate constructs at, where the audited corpus converged // to the admission floor with headroom in iterations to spare. +/// The default candidate cap per side, per row, per iteration. const DEFAULT_MAXIMUM_CANDIDATES: usize = 50; // The iteration cap is a backstop: convergence terminates the loop on every measured corpus first. +/// The default iteration cap. const DEFAULT_MAXIMUM_ITERATIONS: usize = 20; // One accepted update per thousand entries marks the join exhausted: beyond it, iterations trade // full join sweeps for noise-level list changes. +/// The default accepted-update fraction below which the join terminates. const DEFAULT_TERMINATION: f64 = 0.001; /// Pinned NN-Descent sampling, convergence, and termination settings. @@ -79,11 +86,13 @@ pub(crate) struct NnDescentOptions { /// Candidates sampled per side (new and old, forward and reverse) per row per iteration. /// /// Bounds one row's join work at quadratically many distances in the cap. Larger values buy - /// convergence quality with per-iteration cost. + /// convergence quality with per-iteration cost. By default, uses 50 candidates. Zero is clamped to one. pub maximum_candidates: usize = DEFAULT_MAXIMUM_CANDIDATES, - /// Iterations after which construction stops regardless of convergence. + /// Iteration limit, 20 by default. Zero returns the initial random lists. pub maximum_iterations: usize = DEFAULT_MAXIMUM_ITERATIONS, - /// The accepted-update fraction of the total entry count below which the join converges. + /// Accepted-update rate used to stop construction, 0.001 by default. + /// + /// For a finite non-negative rate τ and L stored entries, the ideal update threshold is ceil(τ · L). The stopping count uses f64 count conversion, multiplication and ceil, followed by a saturating u64 cast. Rounding and overflow can change it from the ceiling of the real product. One entry can be displaced repeatedly within an iteration, making this a rate rather than a fraction bounded by one. pub termination: f64 = DEFAULT_TERMINATION, } @@ -93,7 +102,7 @@ const impl Default for NnDescentOptions { } } -/// The NN-Descent construction failed. +/// A row domain that cannot be used for NN-Descent construction. #[derive(Debug)] pub(crate) enum NnDescentError { /// The corpus has at most one row. @@ -124,7 +133,7 @@ pub(crate) struct NnDescent { } impl NnDescent { - /// Wraps pinned options. + /// Configures the candidate sampling and stopping criteria. pub(crate) const fn new(options: NnDescentOptions) -> Self { Self { options } } @@ -140,14 +149,14 @@ struct Entry { /// One row's bounded neighbour list, ascending by `(distance, id)`. /// -/// The list mirrors the worst distance into an atomic beside the lock, which lets an offer reject -/// without contending. Every stored value is a worst read under the lock and the live worst only -/// decreases, so however unlock-and-store pairs interleave, the mirror never falls below the live -/// worst. A stale read is always at or above it, and a rejection against it is always sound. +/// Every accepted replacement strictly improves on the previous worst distance, making the live +/// worst non-increasing. The atomic mirror contains a worst value read under the lock, stored after +/// unlocking. Even when stores are reordered, that value is at least the current live worst. +/// Therefore a candidate at least as far as the mirror can be rejected without taking the lock. /// -/// `Relaxed` suffices because the mirror guards no other memory. Every admission re-checks under -/// the lock, and the lock orders the entries. A `Release`/`Acquire` pairing would only buy ordering -/// for data read outside the lock, and no such read exists. +/// The mirror is only a rejection threshold and publishes no entry data. Every candidate that +/// passes it is checked again under the mutex, which orders all entry access. Therefore relaxed +/// atomic operations suffice for this preliminary check. #[derive(Debug)] struct RowList { entries: Mutex>>, @@ -158,6 +167,12 @@ impl RowList where N: Id, { + /// Builds a list from its initial entries, sorted ascending by `(distance, id)`. + /// + /// # Panics + /// + /// Panics when `entries` is empty: a list always holds at least one neighbour, whose distance + /// seeds the worst mirror. fn new(mut entries: Vec>) -> Self { entries.sort_unstable_by(|lhs, rhs| { lhs.distance @@ -212,8 +227,7 @@ where /// Samples `count` of `pool` uniformly without replacement, taking the whole pool when it fits. /// -/// The pool is ascending afterwards on every path - the retirement scan in [`sample_forward`] -/// binary-searches it. +/// The pool is ascending afterwards on every path, allowing binary-search membership tests. fn sample_pool(pool: &mut Vec, count: usize, mut rng: impl Rng) where N: Id, @@ -228,7 +242,11 @@ where /// Initializes every row's list with `width` distinct random non-self rows. /// /// Sampling draws over a domain one short and shifts past the row itself, excluding it without -/// rejection. +/// rejection. `width` must be positive and smaller than `rows`. +/// +/// # Panics +/// +/// For a nonempty corpus, this panics when `width` is zero or exceeds the non-self domain. fn initialize( rows: usize, width: usize, @@ -271,7 +289,8 @@ where /// Samples each row's forward candidates and retires the drawn new entries. /// /// Splits each list by the *new* flag and samples each side to `cap`. It then clears the flag on -/// the sampled new entries so no later join recompares them. +/// the sampled new entries. Later iterations classify these entries as old when sampling this row's +/// forward list. fn sample_forward( lists: &IdSlice>, cap: usize, @@ -319,6 +338,10 @@ where } /// Transposes sampled candidate sets and limits each reverse pool. +/// +/// # Panics +/// +/// This panics when a target row is outside the `rows`-sized domain. fn reverse( forward: &IdSlice>, rows: usize, @@ -370,10 +393,8 @@ where /// One iteration's accepted updates per stored list entry. /// -/// This reading judges convergence. It falls toward [`termination`](NnDescentOptions::termination) -/// as the join exhausts itself. A join offers each pair to both sides and can displace one entry -/// more than once, so the reading is a rate rather than a share. Early iterations stand above -/// `1`. +/// A join offers each pair to both sides and can displace one entry more than once. The reading is +/// a rate that can exceed `1`, and need not decrease monotonically between iterations. #[expect( clippy::cast_precision_loss, reason = "an accepted-update count and an entry count both stay far below exact f64 integer \ @@ -418,9 +439,8 @@ where let seed = rng.random::(); let cap = self.options.maximum_candidates.max(1); - // The trait admits l2-normalized representations only, so the - // cosine distance reduces to one minus the dot product - a third - // of the full kernel's multiply-adds; the clamp absorbs + // For unit vectors, cosine distance is 1 − the dot product. The trait admits l2-normalized + // representations only, allowing this kernel to omit both norm sums. Clamping absorbs // unit-norm rounding at the range's ends. let distance = |lhs: N, rhs: N| -> NonNegative { let dot = embeddings[lhs].dot(&embeddings[rhs]); @@ -476,8 +496,7 @@ where }); let accepted = accepted.load(Ordering::Relaxed); - // Reported before the break, so the iteration that converged is - // the last one observed rather than the last one unobserved. + // include the terminating iteration in the observations. progress.descent_iteration(DescentIteration { iteration: iteration + 1, accepted_per_entry: accepted_per_entry(accepted, rows * width), @@ -529,6 +548,11 @@ mod tests { random::keyed_rng, }; + /// Builds a row list from `(distance, id)` pairs, all marked old. + /// + /// # Panics + /// + /// This panics when `pairs` is empty. fn list(pairs: &[(NonNegative, u32)]) -> RowList { RowList::new( pairs diff --git a/libs/@local/graph/atlas/src/salt/knn/error.rs b/libs/@local/graph/atlas/src/salt/knn/error.rs index 4402eb133fc..a7d6bfab54e 100644 --- a/libs/@local/graph/atlas/src/salt/knn/error.rs +++ b/libs/@local/graph/atlas/src/salt/knn/error.rs @@ -3,7 +3,7 @@ use core::{error::Error, fmt}; use super::table::KnnValidationError; use crate::math::OpenUnitFraction; -/// Building or spot-checking against a backend failed. +/// A failure to construct, validate or score a neighbour table. #[derive(Debug)] pub(crate) enum KnnError { /// The backend reported an error. @@ -14,7 +14,7 @@ pub(crate) enum KnnError { TooManyRows { rows: usize }, /// The requested table shape overflows the entry count. TooManyEntries { rows: usize, neighbours: usize }, - /// A confidence level at or below one half sizes no one-sided sample. + /// The confidence level gives a negative normal quantile. SampleConfidence { confidence: OpenUnitFraction }, /// Constructed lists are narrower than the table's stored width. ListsWidth { width: usize, neighbours: usize }, diff --git a/libs/@local/graph/atlas/src/salt/knn/hannoy.rs b/libs/@local/graph/atlas/src/salt/knn/hannoy.rs index 4201eda23a0..7df939a97c6 100644 --- a/libs/@local/graph/atlas/src/salt/knn/hannoy.rs +++ b/libs/@local/graph/atlas/src/salt/knn/hannoy.rs @@ -1,12 +1,11 @@ -//! LMDB-backed HNSW behind the nearest-neighbours seam. +//! LMDB-backed HNSW search on the crate's cosine-distance scale. //! -//! [`HannoyIndex`] adapts one [hannoy] index inside one [heed] LMDB environment to -//! [`NearestNeighboursIndex`]. The environment lives in a directory guarded by an advisory file -//! lock, so one process owns the index at a time. Item keys are node rows narrowed to hannoy's -//! `u32` key space. A generation whose row count exceeds `u32::MAX` does not fit this backend. +//! [`HannoyIndex`] provides [`NearestNeighboursIndex`] through one [hannoy] index in one [heed] +//! LMDB environment. An advisory lock excludes other handles using the same lock file for the +//! environment. Item keys must fit hannoy's `u32` key space. //! -//! The adapter rescales backend distances onto the crate's `[0, 2]` cosine scale before they cross -//! the seam, and it orders results by ascending `(distance, id)`. +//! Results are rescaled onto the crate's `[0, 2]` cosine scale and ordered by ascending `(distance, +//! id)`. Rescaling preserves the backend's values, with their own floating-point rounding. use core::{error::Error, fmt, num::TryFromIntError}; use std::{ @@ -28,20 +27,18 @@ use crate::{ random::Compat, }; -/// Reports the backend's own build phases to the run's observer. -/// -/// hannoy names its build steps through [`steppe::Progress`], whose implementors are `'static`, so -/// the builder owns its reporter for the length of the build and the bridge cannot borrow the -/// run's observer. It carries the observer's detached half instead. Only the step's name crosses -/// the seam. hannoy hands its counted sub-step once, before its counter has moved, so the position -/// it carries is always zero and reporting it would describe the phase's progress falsely. +/// A detached observer that reports the backend's build-phase names. struct BuildPhases(D); +// steppe requires a 'static reporter, and the builder owns it. The detached observer avoids +// borrowing the run's observer for that lifetime. impl steppe::Progress for BuildPhases where D: Progress + Send + Sync + 'static, { fn update(&self, sub_progress: impl steppe::Step) { + // the immediate callback sees a zero counter, which can advance if the step is + // retained. Report only the phase name rather than that initial position. self.0.knn_build_phase(&sub_progress.name()); } } @@ -50,27 +47,32 @@ where // the ground layer. M = 16 with M0 = 2 · M follows the Malkov-Yashunin paper's defaults (a // reasonable M range is 5-48 where higher values pay off only for extreme recall or // dimensionality). The recall spot check is the per-corpus arbiter. +/// HNSW connectivity on the upper layers: links per node. #[expect( clippy::min_ident_chars, reason = "M is the canonical HNSW connectivity name" )] const M: usize = 16; +/// HNSW connectivity on the ground layer, `2 · M`. const M0: usize = 32; +/// The index number within the LMDB environment. // One environment carries one index. const INDEX: u16 = 0; +/// The default memory-map bound, 1 TiB. const DEFAULT_MAP_SIZE: usize = 1 << 40; -// Sized by a full-scale sweep (985,932 rows, recall@50, exact references replaying the fit's -// streams). Construction 128 -> 256 buys ~+0.009 sampled aggregate recall (~0.893 -> ~0.902 against -// the 0.89 floor) for ~+90s build (155 -> 245s), and same-seed rebuilds spread ±0.007 (hannoy links -// in parallel, and the seed pins the level stream rather than the link order), so the margin must -// clear that spread. Search breadth measured inert. 64 -> 256 bought +0.002-0.005 on every build at -// 2.2x query cost, so 128 stays. It is 2.5x the deepest query in this crate (the 50-neighbour -// recall audit) and above hannoy's default of 100. Sweep instrument: `report::backend` (`report -// knn-backend`). Raise construction before search on a failed recall check. +// the 985,932-row recall@50 backend sweep used exact-reference streams derived from the fit seeds. +// Raising construction breadth from 128 to 256 improved sampled aggregate recall from about 0.893 +// to 0.902 against the 0.89 floor, at 245s build time instead of 155s. Same-seed rebuilds varied by +// about ±0.007, which the admission margin must accommodate. Raising search breadth from 64 to 256 +// improved recall by 0.002-0.005 at 2.2 times the query cost. Retaining search breadth 128 favors +// construction quality over that recurring query cost. Use `report::backend` (`report knn-backend`) +// to reassess these settings on another corpus. +/// The default build-time frontier breadth. const DEFAULT_EF_CONSTRUCTION: usize = 256; +/// The default search-time frontier breadth. const DEFAULT_EF_SEARCH: usize = 128; /// Pinned hannoy storage, build, and query settings. @@ -78,17 +80,17 @@ const DEFAULT_EF_SEARCH: usize = 128; pub(crate) struct HannoyIndexOptions { /// Upper bound of the LMDB memory map, in bytes. /// - /// The map reserves virtual addresses without allocating memory. Pages materialize as the index writes them, so the bound costs nothing until a write reaches it. A write beyond it fails with an [`MDB_MAP_FULL`](heed::MdbError::MapFull) environment error, which a larger bound prevents. The 1 TiB default covers about 4 KiB per item at [`PROJECTOR_DIMENSIONS`], two orders of magnitude beyond a million-row generation. + /// By default, reserves up to 1 TiB of virtual address space for database mapping. Physical memory use depends on accessed pages. Database growth beyond it fails with an [`MDB_MAP_FULL`](heed::MdbError::MapFull) environment error. Choose a larger bound when the index needs more mapped space. pub map_size: usize = DEFAULT_MAP_SIZE, /// Breadth of the candidate frontier while linking one item into the graph. /// /// Larger values buy link quality with one-time build cost, and link quality bounds the recall - /// any search breadth can reach afterwards. + /// any search breadth can reach afterwards. By default, uses 256 candidates. pub ef_construction: usize = DEFAULT_EF_CONSTRUCTION, /// Breadth of the candidate frontier while searching. /// /// A search never runs below the requested neighbour count. Larger values buy recall with - /// per-query cost, and the recall spot check is the arbiter of whether a setting suffices. + /// per-query cost. By default, uses 128 candidates, raised to at least the requested neighbour count. pub ef_search: usize = DEFAULT_EF_SEARCH, } @@ -98,7 +100,7 @@ const impl Default for HannoyIndexOptions { } } -/// The [`HannoyIndex`] backend failed. +/// A failure of the [`HannoyIndex`] backend. /// /// The message names the failing surface - index, environment, lock file, or key space - and the /// concrete fault chains beneath through [`Error::source`]. @@ -245,12 +247,15 @@ pub(crate) struct HannoyIndex { } impl HannoyIndex { - /// Opens the environment directory at `base` and claims its lock. + /// Opens an existing environment directory at `base` and claims its advisory lock. + /// + /// The directory must reside on a local filesystem with intact LMDB locking. Advisory exclusion + /// covers only handles that use the same lock file. /// /// # Errors /// /// Returns an error when creating the lock file fails, another handle holds the lock, or the - /// environment cannot open. + /// environment or index database cannot be opened or initialized. pub(crate) fn new( base: impl AsRef, options: HannoyIndexOptions, @@ -258,16 +263,22 @@ impl HannoyIndex { Self::open(base.as_ref(), options).map_err(HannoyIndexError) } - /// Opens the environment and claims the lock, in the backend's fault vocabulary. + /// Opens the environment and index database under an advisory lock. + /// + /// # Errors + /// + /// Returns [`IndexFault`] when lock acquisition, environment opening or database initialization + /// fails. fn open(base: &Utf8Path, options: HannoyIndexOptions) -> Result> { let lockfile = base.with_extension("lock"); let lock = File::create(&lockfile)?; lock.try_lock()?; - // SAFETY: The crate cannot rule out another process opening the same database, but every - // opening through this crate's controlled access surface first claims the directory's - // exclusive advisory lock. The handle holds the lock for its whole life. + // SAFETY: heed relies on LMDB locking and on the database files remaining free of non-LMDB + // mutation on a local filesystem. No unsafe LMDB flags are enabled here, and `_lock` + // retains advisory exclusion against cooperating opens until after the environment drops. + // The lock does not establish the local-filesystem or external-mutation assumptions. let env = unsafe { EnvOpenOptions::new() .map_size(options.map_size) @@ -289,6 +300,11 @@ impl HannoyIndex { } /// Inserts every embedding under its row key inside one write transaction. + /// + /// # Errors + /// + /// Returns [`IndexFault`] when a row key does not fit `u32`, an item cannot be inserted, or the + /// transaction fails. fn insert<'embedding, N>( &self, embeddings: impl IntoIterator>, @@ -311,6 +327,10 @@ impl HannoyIndex { } /// Links the inserted items into the HNSW graph inside one write transaction. + /// + /// # Errors + /// + /// Returns [`IndexFault`] when graph construction or the transaction fails. fn link

(&self, rng: impl Rng + SeedableRng, progress: &P) -> Result<(), IndexFault> where P: Progress, @@ -332,6 +352,11 @@ impl HannoyIndex { } /// Searches the configured breadth around a query vector. + /// + /// # Errors + /// + /// Returns [`IndexFault`] when opening a read transaction or index reader fails, or the query + /// fails. fn nns_by_vector( &self, query: &AlignedVecN, @@ -348,6 +373,11 @@ impl HannoyIndex { } /// Searches the configured breadth around an indexed item. + /// + /// # Errors + /// + /// Returns [`IndexFault`] when opening a read transaction or index reader fails, the key does + /// not fit `u32`, the row is absent, or the query fails. fn nns_by_item(&self, id: N, limit: usize) -> Result, IndexFault> where N: Id, @@ -355,8 +385,7 @@ impl HannoyIndex { let rtxn = self.env.read_txn()?; let reader = Reader::open(&rtxn, INDEX, self.db)?; - // hannoy's by_item excludes the queried item from its results, - // so `limit` maps through unchanged. + // by_item already excludes the queried item. Request exactly `limit` results. reader .nns(limit) .ef_search(self.options.ef_search) @@ -368,13 +397,13 @@ impl HannoyIndex { .ok_or(IndexFault::RowNotIndexed(id)) } - /// Maps one search result onto the seam's contract. + /// Orders search results by distance and id and restores the `[0, 2]` cosine scale. fn finish_search(mut results: Vec<(u32, f32)>) -> impl IntoIterator> where N: Id, { - // hannoy returns ascending distances with unspecified ties; the - // id tiebreak pins the seam's deterministic order. + // hannoy leaves distance ties unspecified. Order equal distances by id to satisfy the + // search contract. results.sort_unstable_by(|(lhs_id, lhs_distance), (rhs_id, rhs_distance)| { lhs_distance .total_cmp(rhs_distance) @@ -383,9 +412,9 @@ impl HannoyIndex { results.into_iter().map(|(id, distance)| Neighbour { id: N::from_u32(id), - // hannoy's cosine distance is (1 - cos) / 2 ∈ [0, 1]; - // doubling restores the crate's [0, 2] scale exactly, - // because scaling by a power of two is lossless. + // Multiplication by two is exact for finite f32 values in [0, 1]. hannoy returns (1 − + // cos) / 2 on that range for the admitted vectors. Therefore doubling restores the [0, + // 2] scale without additional rounding. distance: NonNegative::new_unchecked(distance * 2.0), }) } diff --git a/libs/@local/graph/atlas/src/salt/knn/mod.rs b/libs/@local/graph/atlas/src/salt/knn/mod.rs index 2235b34f074..0b0a36e3627 100644 --- a/libs/@local/graph/atlas/src/salt/knn/mod.rs +++ b/libs/@local/graph/atlas/src/salt/knn/mod.rs @@ -2,28 +2,29 @@ //! //! The deliverable is [`table::Knn`], a directed cosine k-nearest-neighbour table over the //! projector representations, stored as a compressed sparse row matrix whose row `i` holds the `k` -//! nearest non-self neighbours of node row `i` with their cosine distances. +//! selected non-self neighbours of node row `i` with their cosine distances. Construction is +//! approximate. //! //! Every row stores exactly `k` entries. No row references itself or repeats a neighbour, every //! distance is finite in `[0, 2]`, and a row's entries ascend by neighbour row. The default `k` is //! [`DEFAULT_NEIGHBOURS`]. The bound applies to semantic sampling and does not limit relation //! edges. //! -//! [`construction::KnnConstruction`] separates the table's semantics from how a constructor -//! produces neighbour lists. [`construction::IndexConstruction`] wraps a [`NearestNeighboursIndex`] -//! search backend, where [`hannoy::HannoyIndex`] is the LMDB-backed HNSW production backend. -//! [`descent::NnDescent`] derives the lists directly by local joins, without a search structure. -//! The landmark assignment keeps querying a backend by vector, while the table build needs only the -//! lists. +//! [`construction::KnnConstruction`] produces neighbour lists independently of the persisted +//! table's representation. [`construction::IndexConstruction`] queries a [`NearestNeighboursIndex`] +//! search backend, including the LMDB-backed HNSW backend [`hannoy::HannoyIndex`]. +//! [`descent::NnDescent`] derives lists directly by local joins. Use a search backend when you also +//! need vector queries, and a list constructor when only per-row neighbours are required. //! -//! Exact comparison admits a construction. [`recall::spot_check_lists`] intersects sampled rows of -//! the produced lists with brute-force [`AlignedVecN`] cosine rankings and reads the aggregate -//! recall's one-sided bound against a configured minimum ([`recall::SpotCheckOptions`]), so an -//! admission stands on what the sample demonstrates rather than on a point estimate. +//! [`recall::spot_check_lists`] compares sampled lists with brute-force [`AlignedVecN`] cosine +//! rankings. It classifies an aggregate recall interval against the minimum in +//! [`recall::SpotCheckOptions`]. The interval uses a normal approximation, with the limitations +//! described in [`recall`]. Recall evidence measures approximation quality. Separate table +//! validation establishes structural invariants. //! -//! The validated table publishes as one sparse matrix file ([`crate::file::sprs`]) holding its -//! matrix verbatim; [`artifact::KnnArchive`] reopens it over a whole-file mapping, so stages after -//! the build read the table from the page cache without holding it on the heap. +//! A table with a zero initial row pointer publishes as one sparse matrix file +//! ([`crate::file::sprs`]) holding its matrix verbatim. [`artifact::KnnArchive`] reopens it over a +//! whole-file mapping, allowing table reads without a heap copy of the matrix regions. //! //! # Reproducibility boundary //! @@ -87,6 +88,7 @@ pub(crate) trait NearestNeighboursIndex where N: Id, { + /// The backend failure every fallible operation reports. type Error; /// Ingests projector representations keyed by node row. @@ -101,14 +103,14 @@ where /// Links the search structure over every inserted row. /// - /// `rng` drives the backend's randomized construction. Sampling streams derive from the seed, - /// but linking applies updates in parallel and unordered, so same-seed builds need not agree. - /// The recall spot check downstream is the arbiter of a construction, never a replay. + /// `rng` drives the backend's randomized construction. A seed determines its random stream, but + /// does not fix a parallel update order. Same-seed builds need not agree. /// - /// The link is the construction's long phase and only the backend knows its parts, so a backend - /// reports them as they begin through [`knn_build_phase`](Progress::knn_build_phase). A backend - /// that hands the reporting on to machinery owning it takes the observer's detached half, never - /// this borrow. + /// # Implementation Note + /// + /// Implementations report their construction phases as they begin through + /// [`knn_build_phase`](Progress::knn_build_phase). Reporting machinery that outlives this + /// call's observer borrow must take its detached observer, never retain the borrow. /// /// # Errors /// @@ -119,8 +121,8 @@ where /// Returns up to `limit` nearest neighbours of `query`. /// - /// The query is positional, so a row whose stored vector equals the query shows up in the - /// results. + /// The query excludes no row by identity. A row whose stored vector equals the query is + /// eligible for the results. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/knn/recall.rs b/libs/@local/graph/atlas/src/salt/knn/recall.rs index 3bf4016d0aa..8b4ed9cdcfa 100644 --- a/libs/@local/graph/atlas/src/salt/knn/recall.rs +++ b/libs/@local/graph/atlas/src/salt/knn/recall.rs @@ -1,36 +1,47 @@ -//! Exact recall spot check for approximate search backends. +//! Sampled recall against brute-force cosine rankings. //! -//! For each sampled node row, the check compares an approximate neighbour list with a brute-force -//! cosine ranking over the same projector matrix. Both rankings exclude the query row and resolve -//! equal distances by ascending row. Recall is the total intersection count divided by the total -//! number of exact neighbours across the sample, and the check admits a backend when that -//! aggregate's lower bound clears the configured [minimum](SpotCheckOptions::minimum_recall). +//! For each sampled row, the check compares an approximate neighbour list with the brute-force +//! ranking over the same projector matrix. The reference excludes the query row and resolves equal +//! kernel distances by ascending row. Recall counts literal neighbour-id intersections, including +//! when independently computed distances differ near a tie. "Exact" describes exhaustive ranking by +//! the crate's floating-point cosine kernel. //! -//! The check sizes the sample in three stages, because the criterion is an aggregate mean whose -//! per-row variance is a corpus property. A pilot measures the mean's deviation, the aggregate's -//! clearance of the minimum, and the rate the brute force runs at. Those measurements size one -//! fresh verdict sample. Its count resolves the *measured* clearance at the configured -//! [confidence](SpotCheckOptions::confidence), with a floor at the pilot's size and a cap from the -//! corpus and from what the [budget](SpotCheckOptions::budget) buys at the measured rate. The -//! verdict sample alone decides, and it decides by interval, because [`RecallAdmission`] reads the -//! recall's one-sided bound against the minimum, never the point estimate. No fixed margin shows up -//! anywhere. What a decision has to resolve is the clearance the run measures, so a backend far -//! above the floor settles at the pilot's size and one near the floor draws until the budget stops -//! it. +//! # Sampling and decision //! -//! The pilot sizes but does not vote. A bound holds at its stated confidence only over data the -//! sizing never saw, so the check draws the verdict sample fresh and reads it once. A check -//! that re-read a growing sample until the bound cleared the floor admits a backend sitting exactly -//! on the floor sooner or later, whatever confidence it printed. The knobs are scale-free, and the -//! check measures both the variance and the clearance rather than reading them from configuration. -//! (An acceptance-sampling budget, which was this check's original sizing, certifies all-pass -//! criteria and guarantees nothing about a mean: per-row recall is strongly bimodal, and at the -//! acceptance-sized 688 rows the check refused sound backends on sampling noise.) +//! A pilot estimates the per-row standard deviation, the aggregate's clearance of the configured +//! [minimum](SpotCheckOptions::minimum_recall), and the scoring rate. These measurements size one +//! fresh verdict sample, floored at the pilot's size and capped by the corpus and the +//! [budget](SpotCheckOptions::budget) at the measured rate. The verdict sample alone decides. If +//! the pilot already covers the corpus, it is a census and supplies the verdict directly. //! -//! The exact side of the check stands alone as [`ExactReference`]: one sampled brute-force -//! reference scores any number of backends or backend settings, so a parameter sweep pays the exact -//! rankings once instead of once per grid point. [`spot_check_lists`] composes the two halves -//! over already-constructed neighbour lists. +//! Let `N` be the corpus size, `n` the number of sampled rows, and `k` the number of exact non-self +//! neighbours per row. For each sampled row `i`, Rᵢ is its intersection count divided by `k`. The +//! aggregate R̄ is the total intersection count divided by `n · k`, and `s` is the sample standard +//! deviation of the Rᵢ values. At [confidence](SpotCheckOptions::confidence) `c`, the normal +//! quantile `z = Φ⁻¹(c)` gives the implemented half-width h = z · s / √n · √((N − n) / (N − 1)). A +//! census has h = 0. For minimum μ, [`RecallAdmission`] admits when R̄ − h ≥ μ and refuses when +//! R̄ + h < μ. Otherwise the admission remains unresolved. Admission never uses the point estimate +//! alone when h > 0. +//! +//! The pilot's measured clearance δ = |R̄ − μ| sizes ceil((z · s / δ)²) verdict rows before the +//! floor and caps apply. Zero clearance requests the largest affordable sample. This targets the +//! difference the decision must resolve without a fixed margin. An all-pass defect-rate sample size +//! does not supply a bound on this mean's error. +//! +//! The interval uses a normal approximation with an estimated deviation. Its confidence is nominal, +//! not a finite-sample, distribution-free coverage guarantee. Small samples and zero observed +//! spread can understate uncertainty. Drawing the verdict afresh avoids reusing the pilot's +//! observed recall in the decision and avoids repeated stopping tests on a growing verdict sample. +//! Fresh draws may overlap the pilot's rows. The normal approximation's limitations remain despite +//! this separation. +//! +//! # Reuse and precision +//! +//! [`ExactReference`] keeps sampled brute-force rankings for scoring multiple backends or settings +//! against identical queries. [`spot_check_lists`] scores already-constructed lists. Seeded draws +//! repeat for fixed sample sizes. Floating-point reduction order can change the measured deviation +//! and resulting size or half-width in the final bits, and budget-limited sizing also depends on +//! measured wall time. use alloc::collections::BinaryHeap; use core::{cmp::Ordering, default::Default, num::NonZero, time::Duration}; @@ -53,30 +64,29 @@ use crate::{ random::{mean_sample_size, normal_quantile, sample_ids}, }; -// The defaults are the backend admission criterion, recall@50 ≥ 0.89: -// the criterion is aggregate recall over the sample, so a long per-row -// tail cannot fail a backend whose aggregate holds. +// admission uses aggregate recall@50 ≥ 0.89 with its measured interval. It imposes no per-row +// recall floor. +/// The default `k` of the measured recall. const DEFAULT_NEIGHBOURS: NonZero = nz!(50); +/// The default aggregate recall floor a backend must clear. const DEFAULT_MINIMUM_RECALL: UnitFraction = unit_fraction!(0.89); -// A one-in-a-hundred risk that the aggregate's sampling error exceeds -// the reported resolution in the admitting direction. The sample grows -// as the square of the normal quantile, so the sample size prices the level in rows. 0.999 costs -// ~1.8x the sample this one sizes, and 0.95 costs half of it while admitting one backend in twenty -// whose true aggregate sits below the floor. +// sample size scales with the square of the normal quantile. At fixed spread and margin, 0.999 uses +// about 1.8 times the rows of 0.99, while 0.95 uses about half. These are nominal +// normal-approximation confidence levels. +/// The default confidence level of the admission. const DEFAULT_CONFIDENCE: OpenUnitFraction = open_unit_fraction!(0.99); -// The acceptance-era sample size, kept as the pilot: large enough to -// read the per-row deviation within a few percent, small enough that -// a decisively good or bad backend settles at ~19s of brute force. +// the pilot trades precision of its spread estimate against the cost of a full-corpus scan per +// sampled row. Reassess its size using the measured deviation and scoring rate across +// representative corpora. +/// The default pilot sample size. const DEFAULT_PILOT: NonZero = nz!(688); -// How long a build may spend proving its own admission. The value is a policy decision about a -// machine's time rather than a measured quantity. Ten minutes covers the sizing at the scale the -// check runs at: the full-scale backend sweep (985,932 rows) measured healthy builds at ~0.902 -// against the 0.89 floor with a per-row deviation of ~0.32 (near-tie rows score -// ~0.5 on any ANN index), so a healthy build's clearance sizes ~3,850 -// rows, ~108s of brute force, and a build clearing by half of that -// sizes four times as many. Past the budget the check reports the -// resolution it reached instead of spending a run's afternoon on a -// difference no decision turns on. +// ten minutes is a sizing policy that the pilot's measured rate converts into a row cap. The check +// itself runs without a wall-clock deadline. The 985,932-row backend sweep measured recall around +// 0.902 with per-row deviation around 0.32. At confidence 0.99, z ≈ 2.326 and clearance +// 0.902 − 0.89 = 0.012 request ceil((2.326 · 0.32 / 0.012)²) ≈ 3,848 rows before caps. Halving the +// clearance quadruples the request. The time-derived cap limits that growth, and the result +// reports the achieved width. +/// The default wall-clock budget of one check. const DEFAULT_BUDGET: Duration = Duration::from_secs(600); /// Pinned sampling and admission settings for one recall spot check. @@ -85,24 +95,21 @@ pub(crate) struct SpotCheckOptions { /// Exact neighbours compared per sampled row. /// /// A corpus smaller than this compares every non-self row. This is the `k` of the measured - /// recall@k, independent of the persisted table's neighbour count. + /// recall@k, independent of the persisted table's neighbour count. By default, compares 50 neighbours. pub neighbours: NonZero = DEFAULT_NEIGHBOURS, - /// Minimum admitted aggregate recall over the sample. + /// Minimum admitted aggregate recall over the sample, 0.89 by default. pub minimum_recall: UnitFraction = DEFAULT_MINIMUM_RECALL, - /// One-sided confidence that the aggregate's sampling error stays inside the reported - /// [resolution](RecallSpotCheck::resolution). + /// Nominal one-sided confidence for the normal-approximation interval, 0.99 by default. /// - /// Above one half, so the sizing quantile stays non-negative; the check refuses a smaller - /// confidence before it samples. + /// Values below one half are refused before sampling. At one half the quantile and reported [resolution](RecallSpotCheck::resolution) are zero. pub confidence: OpenUnitFraction = DEFAULT_CONFIDENCE, /// Rows of the sizing pilot. /// - /// A corpus smaller than this compares every row exhaustively. The pilot measures the per-row deviation, the aggregate's clearance of the minimum, and the rate the brute force runs at, and its size floors the verdict sample, because a normal bound over a sample too small to estimate its own deviation resolves nothing. + /// By default, samples 688 rows. A corpus no larger than this is compared exhaustively. Otherwise the pilot estimates the spread, clearance and scoring rate used to size the fresh verdict sample. Its size floors that sample. pub pilot: NonZero = DEFAULT_PILOT, - /// Wall clock the verdict sample may spend, at the rate the pilot measured. + /// Estimated verdict-sample time budget, ten minutes by default. /// - /// A sizing beyond the budget's reach draws what the budget affords and records the resolution - /// it achieved. [`ZERO`](Duration::ZERO) draws the verdict sample at the pilot's size. + /// The pilot's measured rate converts this duration into a row cap, floored at the pilot's size. This is a sizing estimate rather than a runtime deadline and excludes the pilot's own cost. [`ZERO`](Duration::ZERO) selects the pilot-size floor when the pilot has a positive measured duration. An unmeasurably short pilot imposes no time-derived cap. pub budget: Duration = DEFAULT_BUDGET, } @@ -131,25 +138,25 @@ pub struct RecallSpotCheck { pub expected: u64, /// Sample standard deviation of per-row recall over the verdict sample. /// - /// What this sample measured, not what sized it: the pilot's own reading of the spread is what - /// chose this sample's size. + /// The pilot's separate deviation sizes this sample. This value describes the verdict rows. pub deviation: DNonNegative, /// The admission minimum the check ran under. pub minimum_recall: UnitFraction, - /// The one-sided sampling resolution the verdict sample achieved, in recall units. + /// The normal-approximation half-width of the verdict sample, in recall units. /// - /// The half-width `z · deviation / sqrt(sampled_rows)` the [admission](Self::admission) - /// reading compares against the minimum, narrowed by the finite-population factor. Zero - /// when the sample is the corpus, because a census has no sampling error to bound. + /// The [admission](Self::admission) reading uses h = z · s / √n · √((N − n) / (N − 1)), with + /// `s` the verdict deviation, `n` its sample size, `N` the corpus size and `z` the normal + /// quantile of the configured confidence. Zero when the sample is the corpus, because a census + /// has no sampling error to bound. pub resolution: DNonNegative, - /// The one-sided confidence the resolution holds at. + /// The nominal one-sided confidence used for the resolution. pub confidence: OpenUnitFraction, } -/// What one recall spot check demonstrated about its backend. +/// A recall interval's position relative to the configured admission minimum. /// -/// The reading compares the aggregate's one-sided interval with the admission minimum, so it -/// separates a backend proven good from one proven bad from a sample that settles neither. +/// The interval has the normal-approximation limitations described in +/// [`recall`](crate::salt::knn::recall). #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum RecallAdmission { /// The recall's lower bound clears the minimum. @@ -173,17 +180,16 @@ impl RecallSpotCheck { recall } - /// Returns what the sample demonstrated about the configured admission minimum. + /// Classifies the recall interval against the configured admission minimum. /// - /// Admission asks the interval rather than the point estimate. This reading admits a backend - /// when its recall's lower bound clears the minimum. It refuses one when the upper bound falls - /// below the minimum, and it reports [`Unresolved`](RecallAdmission::Unresolved) when the - /// achieved [resolution](Self::resolution) spans the minimum. A sample that ran out of budget - /// has measured something, though not the thing the floor asks about. + /// Admits when the lower endpoint reaches the minimum and refuses when the upper endpoint falls + /// below it. Otherwise returns [`Unresolved`](RecallAdmission::Unresolved), including when the + /// budget-limited sample does not separate the interval from the minimum. /// - /// Each side spends the confidence once, and a one-sided quantile bounds both risks: for any - /// one true recall only one of the two errors is possible, so admitting a backend below the - /// minimum and refusing one above it each stay at `1 - confidence`. + /// For a fixed true recall, only one of false admission or false refusal is possible: the true + /// value is either below the minimum or at least the minimum. Each comparison uses its + /// corresponding one-sided normal quantile. Therefore no two-sided correction is needed to + /// target either error separately, subject to the interval's normal-approximation limitations. #[inline] #[must_use] pub fn admission(&self) -> RecallAdmission { @@ -199,6 +205,7 @@ impl RecallSpotCheck { } } +/// One brute-force reference neighbour, ordered by `(distance, row)`. #[derive(Debug, Copy, Clone)] struct ExactNeighbour { row: N, @@ -239,7 +246,11 @@ where } } -/// Returns the `limit` exact nearest non-self neighbours of `query`. +/// Returns up to `limit` non-self neighbours of `query` by exhaustive kernel ranking. +/// +/// # Panics +/// +/// This panics when `query` is outside `embeddings`. fn exact_neighbours( embeddings: &IdSlice>, query: N, @@ -280,8 +291,8 @@ where /// One sampled brute-force reference, reusable across backends. /// -/// The sample and its exact neighbour lists depend only on the corpus and the sampling draw, so one -/// reference scores any number of backends or backend settings against identical queries. +/// The sample and its exact neighbour lists depend only on the corpus and the sampling draw. Reuse +/// one reference to score any number of backends or settings against identical queries. #[derive(Debug)] pub(crate) struct ExactReference { /// Sampled rows and their exact neighbours, ascending within each row's list. @@ -296,13 +307,13 @@ where { /// Samples query rows and computes their exact cosine rankings in parallel. /// - /// `embeddings` holds the projector representations in row order; a mapped `f32[T, 512]` + /// `embeddings` holds the projector representations in row order. A mapped `f32[T, 512]` /// artifact yields the slice directly. A `sample_size` beyond the corpus compares every row, /// and a `neighbours` beyond the corpus compares every non-self row. /// /// # Errors /// - /// Returns an error when the corpus holds fewer than two rows. + /// Returns [`KnnError`] when the corpus holds fewer than two rows. pub(crate) fn new( embeddings: &IdSlice>, neighbours: NonZero, @@ -354,7 +365,11 @@ where /// /// Sampled rows read their list prefix at the reference depth and compare in parallel. Lists /// narrower than the reference depth score what they hold. The reading carries raw counts and - /// the per-row spread, and admission criteria live with the caller. + /// the per-row spread, without applying an admission minimum. + /// + /// # Panics + /// + /// This panics when a sampled query row is outside `lists`. pub(crate) fn score_lists(&self, lists: &NeighbourLists) -> Scoring { let depth = self.neighbours_per_row.min(lists.width()); let (matched, squares) = self @@ -404,12 +419,11 @@ where /// /// This scoring queries sampled rows through /// [`search_by_id`](NearestNeighboursIndex::search_by_id) and compares them in parallel. The - /// reading carries raw counts and the per-row spread, and admission criteria live with the - /// caller. + /// reading carries raw counts and the per-row spread, without applying an admission minimum. /// /// # Errors /// - /// Returns an error when the backend fails a query. + /// Returns [`KnnError`] when the backend fails a query. pub(crate) fn score(&self, index: &I) -> Result> where I: NearestNeighboursIndex + Sync, @@ -497,7 +511,7 @@ impl Scoring { /// Computes the sample standard deviation of per-row recall. /// -/// Derived from the aggregate counts and the sum of squared per-row recalls. +/// Computes sample deviation from aggregate counts and squared per-row recalls. /// /// The per-row sum needs no separate accumulator: it is the matched total divided by the comparison /// depth. @@ -515,20 +529,17 @@ fn deviation(rows: usize, matched: u64, expected: u64, squares: f64) -> DNonNega // the root real. let variance = (count * mean).mul_add(-mean, squares).max(0.0) / (count - 1.0); - // Per-row recalls lie in [0, 1], so the squared sum stays within the row count and the - // clamped variance is finite non-negative. The root of such a value is in domain. + // Per-row recalls lie in [0, 1]. Their squared sum is bounded by the row count, and the clamp + // removes a negative rounding residual. The resulting finite, non-negative variance has a + // finite, non-negative root. DNonNegative::new_unchecked(variance.sqrt()) } /// Returns the rows that resolve the pilot's measured clearance of the admission minimum. /// -/// [`mean_sample_size`]'s identity, with the clearance the pilot measured standing where a -/// configured margin otherwise would. What a decision has to resolve is how far the aggregate sits -/// from the floor, and only the run knows that. -/// -/// A caller that has already read a quantile out of `confidence` leaves one way for the sizing to -/// come back empty: an aggregate sitting exactly on the floor, which no finite sample resolves and -/// which therefore asks for every row a budget allows. +/// Uses [`mean_sample_size`] with the absolute difference between the pilot recall and the minimum +/// as its margin. A pilot exactly on the minimum returns [`usize::MAX`] before the corpus and +/// budget caps apply. fn sizing_rows( piloted: &Scoring, minimum_recall: UnitFraction, @@ -541,12 +552,10 @@ fn sizing_rows( }) } -/// Returns the rows `budget` buys at the rate `measured` rows took to sample and score. +/// Estimates affordable rows from the measured sampling and scoring rate. /// -/// The exact reference scans the whole corpus per sampled row, so the pilot's own cost per row is -/// the verdict sample's cost per row. The budget converts to rows against a rate this machine -/// demonstrated in the same run, never against a per-row cost recorded from another one. A pilot -/// too fast to time affords everything. +/// Each reference query scans the whole corpus. The pilot's rate estimates the verdict sample's +/// cost, without guaranteeing equal per-row time. A zero measured duration returns [`usize::MAX`]. fn budget_rows(budget: Duration, elapsed: Duration, measured: usize) -> usize { #[expect( clippy::cast_precision_loss, @@ -570,10 +579,9 @@ fn budget_rows(budget: Duration, elapsed: Duration, measured: usize) -> usize { /// Computes the one-sided half-width of the aggregate's sampling interval. /// -/// `z * deviation / sqrt(n)`, narrowed by the finite-population factor `sqrt((N - n) / (N - 1))`: -/// the aggregate is a mean over rows drawn without replacement from a corpus of `N`, so a sample -/// that reaches the corpus has no sampling error left to bound and the point estimate is the -/// population value. +/// For sample size `n`, corpus size `N`, sample deviation `s` and normal quantile `z`, the +/// implemented width is z · s / √n · √((N − n) / (N − 1)). The correction makes a census's width +/// zero. For a proper subsample this is a plug-in normal approximation. fn resolution(quantile: f64, scored: &Scoring, rows: usize) -> DNonNegative { #[expect( clippy::cast_precision_loss, @@ -589,8 +597,8 @@ fn resolution(quantile: f64, scored: &Scoring, rows: usize) -> DNonNegative { .max(0.0) .sqrt(); - // The caller refused a confidence at or below one half, so the quantile is non-negative; - // the deviation, the root, and the clamped correction are non-negative by construction. + // the confidence check excludes negative quantiles. The sample's deviation, the root and the + // clamped correction are non-negative. DNonNegative::new_unchecked(quantile * scored.deviation / sampled.sqrt() * correction) } @@ -599,6 +607,11 @@ fn resolution(quantile: f64, scored: &Scoring, rows: usize) -> DNonNegative { /// A pilot runs first, its measurements size the verdict sample, and the fresh verdict sample alone /// decides. Both draws come from the one generator, and the two entry points differ only in what /// `score` compares against. +/// +/// # Errors +/// +/// Returns [`KnnError`] when the confidence has a negative normal quantile, the corpus has fewer +/// than two rows, or `score` fails. fn staged_check( embeddings: &IdSlice>, options: SpotCheckOptions, @@ -611,8 +624,8 @@ where let rows = embeddings.len(); let quantile = normal_quantile(options.confidence); if quantile < 0.0 { - // A one-sided bound needs a non-negative quantile: a confidence at or below one half - // would size no sample and read a negative resolution. + // a confidence below one half gives a negative quantile and would produce a negative + // resolution. return Err(KnnError::SampleConfidence { confidence: options.confidence, }); @@ -626,8 +639,8 @@ where let elapsed = started.elapsed(); let scored = if pilot.sampled_rows() >= rows { - // A pilot that covered the corpus is a census. No sample size remains to choose, so the - // sizing had no freedom to bias and the reading is exactly what the pilot measured. + // a census already measures every row. Reuse exactly that reading without a second + // exhaustive pass. piloted } else { // Stages two and three. The pilot sizes the verdict sample @@ -661,16 +674,19 @@ where }) } -/// Measures recall of constructed lists against exact cosine rankings, sizing the sample in three -/// stages. +/// Measures recall of constructed lists against sampled brute-force rankings. /// -/// Sizing runs the three stages above, and scoring reads the lists in place, so the verdict sample -/// pays only for its exact rankings. +/// Uses the pilot and verdict sampling described in the [module](crate::salt::knn::recall). The +/// lists must cover the same row domain as `embeddings`. /// /// # Errors /// -/// Returns an error when the corpus holds fewer than two rows or the confidence is degenerate -/// ([`SampleConfidence`](KnnError::SampleConfidence)). +/// Returns [`KnnError`] when the confidence is below one half or the corpus holds fewer than two +/// rows. +/// +/// # Panics +/// +/// This panics when a sampled query row is outside `lists`. #[tracing::instrument(skip_all)] pub(crate) fn spot_check_lists( lists: &NeighbourLists, @@ -697,14 +713,14 @@ where /// [admission](RecallSpotCheck::admission) reading. A pilot that already covers the corpus is /// exhaustive and is itself the reading. /// -/// Both draws come from the one generator, so a seeded check replays exactly whenever the budget -/// leaves the sizing alone. A run whose verdict sample the budget truncates samples what its own -/// machine afforded, and records the [resolution](RecallSpotCheck::resolution) it reached. +/// Both draws come from the one generator. Fixed-size seeded draws repeat, but floating-point +/// reduction order and measured timing can affect sizing and +/// [resolution](RecallSpotCheck::resolution). /// /// # Errors /// -/// Returns an error when the corpus has at most one row, when the confidence is degenerate -/// ([`SampleConfidence`](KnnError::SampleConfidence)), or when a backend query fails. +/// Returns [`KnnError`] when the confidence is below one half, the corpus holds fewer than two +/// rows, or a backend query fails. #[cfg(test)] // The knn tests score fixture backends through the full sampling path. pub(crate) fn spot_check( index: &I, diff --git a/libs/@local/graph/atlas/src/salt/knn/report/backend.rs b/libs/@local/graph/atlas/src/salt/knn/report/backend.rs index b71552b6aa8..cbacbf25d7d 100644 --- a/libs/@local/graph/atlas/src/salt/knn/report/backend.rs +++ b/libs/@local/graph/atlas/src/salt/knn/report/backend.rs @@ -2,13 +2,13 @@ //! //! The sweep builds one hannoy index per (fit seed, `ef_construction`) grid cell and scores it at //! every `ef_search` value against the exact reference of every distinct seed's sample. `ef_search` -//! is a query-time setting, so one build serves its whole search row by reopening the persisted -//! environment. The sweep removes the environment as soon as it finishes that row, which keeps peak +//! is a query-time setting. Reopening the persisted environment reuses one build for its whole +//! search row. The sweep removes the environment as soon as it finishes that row, which keeps peak //! disk at one index. //! -//! The production check reads the grid's diagonal, a build scored against its own seed's sample. -//! Off-diagonal readings separate build quality from sample hardness, and the sweep's decision -//! surface is the worst recall a setting produced anywhere in the grid. +//! Scoring every build against every seed's sample separates build variation from sample variation. +//! The minimum across observed grid cells describes this sweep only. It is neither a confidence +//! bound nor the fit's staged admission reading. use alloc::borrow::Cow; use core::{ @@ -42,8 +42,8 @@ use crate::{ /// Fit seeds the sweep replays by default. /// -/// The list holds three distinct seeds and repeats one of them, so the grid carries both build -/// nondeterminism and seed spread. +/// Repeating seed zero measures build nondeterminism. Seeds one and two add variation from the +/// seed. pub(crate) const DEFAULT_SEEDS: &[u64] = &[0, 0, 1, 2]; /// `ef_construction` values swept by default: the deployed setting and the one below it. pub(crate) const DEFAULT_CONSTRUCTIONS: &[usize] = &[128, 256]; @@ -57,6 +57,7 @@ pub(crate) const DEFAULT_SEARCHES: &[usize] = &[64, 128, 192, 256]; // values: a moved default fails compilation here instead of silently leaving the swept grid // without the deployed setting. const _: () = { + /// Reports whether `values` holds `value`, at compile time. const fn contains(values: &[usize], value: usize) -> bool { let mut index = 0; while index < values.len() { @@ -78,11 +79,15 @@ const _: () = { pub(crate) struct Options { /// Fit seeds whose build and sample streams the sweep replays. /// - /// A repeated seed rebuilds the same configuration again, measuring build nondeterminism. + /// A repeated seed rebuilds the same configuration again, measuring build nondeterminism. By default, uses [`DEFAULT_SEEDS`]. pub seeds: Cow<'static, [u64]> = Cow::Borrowed(DEFAULT_SEEDS), /// `ef_construction` values, with one index build per (seed, value). + /// + /// By default, uses [`DEFAULT_CONSTRUCTIONS`]. pub constructions: Cow<'static, [usize]> = Cow::Borrowed(DEFAULT_CONSTRUCTIONS), /// `ef_search` values, swept per built index. + /// + /// By default, uses [`DEFAULT_SEARCHES`]. pub searches: Cow<'static, [usize]> = Cow::Borrowed(DEFAULT_SEARCHES), } @@ -99,7 +104,7 @@ pub(crate) struct Point { pub sample_seed: u64, /// The query-time frontier breadth of this reading. pub ef_search: usize, - /// Aggregate recall@50 against the exact reference. + /// Aggregate recall at the sweep's comparison depth against the exact reference. pub recall: f64, } @@ -112,7 +117,7 @@ pub(crate) struct Build { pub ef_construction: usize, /// Wall clock of insert plus link. pub wall: Duration, - /// One reading per `ef_search` value, in grid order. + /// One reading per (`ef_search`, distinct sample seed), in grid order. pub points: Vec, } @@ -145,7 +150,7 @@ pub(crate) struct Sweep { } impl Sweep { - /// The swept `ef_construction` values, in grid order. + /// Returns distinct `ef_construction` values in first-appearance order. fn constructions(&self) -> impl IntoIterator { let mut values: Vec = Vec::new(); for build in &self.builds { @@ -156,7 +161,7 @@ impl Sweep { values } - /// The swept `ef_search` values, in grid order. + /// Returns distinct `ef_search` values in first-appearance order. fn searches(&self) -> impl IntoIterator { let mut values: Vec = Vec::new(); for point in self.builds.iter().flat_map(|build| &build.points) { @@ -167,10 +172,9 @@ impl Sweep { values } - /// The worst recall one setting produced across every build and sample that measured it. + /// Returns the lowest observed recall for a setting across every build and sample. /// - /// This reading is the sweep's decision surface, because a setting earns admission on what it - /// guarantees rather than on its best grid cell. + /// Returns positive infinity when the sweep contains no matching reading. fn minimum_recall(&self, ef_construction: usize, ef_search: usize) -> f64 { self.builds .iter() @@ -282,8 +286,8 @@ impl Error for SweepError { /// An index persisted into its scratch directory, bound to the breadth it was built at. /// -/// Scoring reopens the environment with this breadth, so the reopened options describe the index -/// that exists on disk. [`Self::remove`] consumes the value together with the directory it names. +/// Reopening retains the original build breadth in the options. [`Self::remove`] consumes the value +/// and removes its directory. struct BuiltIndex { /// The scratch directory holding the persisted environment. directory: Utf8PathBuf, @@ -395,8 +399,8 @@ fn score_grid( /// /// # Errors /// -/// Returns a [`SweepError`] when reading the representations fails, when an index build fails, or -/// when a sampled query fails. +/// Returns [`SweepError`] when setup, scratch-directory management, reference computation or a +/// backend operation fails. pub(crate) fn sweep(root: &GenerationRoot, options: &Options) -> Result { let started = Instant::now(); @@ -406,8 +410,8 @@ pub(crate) fn sweep(root: &GenerationRoot, options: &Options) -> Result)> = Vec::new(); let mut reference_costs = Vec::new(); for &seed in &*options.seeds { diff --git a/libs/@local/graph/atlas/src/salt/knn/report/descent.rs b/libs/@local/graph/atlas/src/salt/knn/report/descent.rs index bd07536cd1b..1196191c05c 100644 --- a/libs/@local/graph/atlas/src/salt/knn/report/descent.rs +++ b/libs/@local/graph/atlas/src/salt/knn/report/descent.rs @@ -1,9 +1,10 @@ //! The NN-Descent construction audit over a published generation. //! -//! Constructions run at the production width - the wider of the spot check's depth and the stored -//! neighbour count - replaying the production fit's `knn-link` stream per seed, and one exact -//! reference scores every reading. A repeated seed measures construction nondeterminism. A -//! candidate cap is the knob the audit sweeps. +//! Constructions use the wider of the default spot-check depth and stored neighbour count, clamped +//! to the corpus's non-self domain. Each seed uses the fit's `knn-link` stream derivation. One +//! exact reference, drawn from the first seed or zero if the list is empty, scores every +//! construction. Repeated seeds expose construction nondeterminism while the comparison varies +//! candidate caps. use core::{ fmt::{self, Display}, @@ -29,11 +30,11 @@ use crate::{ /// Fit seeds the audit replays by default. /// -/// The list holds two distinct seeds and repeats one of them, so the readings carry construction -/// nondeterminism as well as seed spread. +/// Repeating seed zero measures build nondeterminism. Seed one adds variation from the seed. pub(crate) const DEFAULT_SEEDS: &[u64] = &[0, 0, 1]; -/// Candidate caps audited by default: the constructor's own setting, so a bare invocation reads the -/// deployed construction. +/// Candidate caps audited by default. +/// +/// Uses the constructor's default cap. pub(crate) const DEFAULT_CANDIDATES: &[usize] = &[NnDescentOptions::default().maximum_candidates]; /// One NN-Descent construction reading. @@ -45,7 +46,7 @@ pub(crate) struct Reading { pub maximum_candidates: usize, /// Wall clock of the construction. pub construct_wall: Duration, - /// Aggregate recall@50 against the exact reference. + /// Aggregate recall at the audit's comparison depth against the exact reference. pub recall: f64, } @@ -100,8 +101,7 @@ impl Display for Audit { /// /// # Errors /// -/// Returns an [`AuditError`] when reading the representations fails, when computing the reference -/// fails, or when a construction fails. +/// Returns [`AuditError`] when setup, reference computation or a construction fails. pub(crate) fn audit( root: &GenerationRoot, seeds: &[u64], diff --git a/libs/@local/graph/atlas/src/salt/knn/report/mod.rs b/libs/@local/graph/atlas/src/salt/knn/report/mod.rs index 8b45b03dcbb..e6a0fad4b56 100644 --- a/libs/@local/graph/atlas/src/salt/knn/report/mod.rs +++ b/libs/@local/graph/atlas/src/salt/knn/report/mod.rs @@ -1,18 +1,21 @@ -//! Audits and sweeps over a published generation's neighbour construction. +//! Neighbour-construction comparisons over published projector representations. //! -//! Each instrument reopens the active generation's projector representation artifact and replays -//! the production fit's random streams for the stage it measures. It then scores one construction -//! against an exact reference. [`backend`] sweeps the hannoy backend over its `ef_construction` × -//! `ef_search` grid and [`descent`] audits NN-Descent constructions across candidate caps. An -//! instrument observes the construction stage. No fit consumes anything here, and a reading -//! describes a generation that is already published. +//! [`backend`] sweeps hannoy over its `ef_construction` × `ef_search` grid. [`descent`] compares +//! NN-Descent candidate caps. Both read the active generation's representation artifact and return +//! measurements without changing its published artifacts. //! -//! Every instrument replays [`stage_rng`](crate::salt::fit::stage_rng) per fit seed, so a grid -//! point reproduces what a live fit at that seed and setting would have measured, and a repeated -//! seed measures the construction's own nondeterminism rather than seed spread. An instrument -//! computes the exact reference once per distinct seed and scores every reading against it. +//! Both comparisons use the fit's [`stage_rng`](crate::salt::fit::stage_rng) derivation. A repeated +//! seed exposes construction nondeterminism separately from seed spread. The backend sweep reuses +//! one exact reference per distinct seed and scores every build against every reference. The +//! descent comparison uses one reference from its first seed, or seed zero when the seed list is +//! empty. //! -//! An instrument returns its readings and its host renders them. +//! # Measurement scope +//! +//! These comparisons operate on the published corpus rows with fixed-size reference samples of up +//! to 2,048 queries. The fit constructs on its distinct-representation quotient and uses staged +//! recall sampling. These readings share that seed derivation without replaying the fit's table or +//! admission interval. They compare settings within each report's fixed corpus and sampling design. use core::{ error::Error, @@ -39,12 +42,13 @@ pub(crate) mod descent; #[cfg(test)] mod tests; -// The reference size stays fixed because an instrument compares settings against each other. One -// sizing of the samples yields SE ~0.007 at the measured per-row deviation and resolves the -// construction effect. The production check stages its sizing per reading instead. +// a fixed reference lets settings share the same queries. At a per-row deviation of 0.32, 2,048 +// rows give an uncorrected standard error of 0.32 / √2048 ≈ 0.0071. The fit's check sizes each +// verdict separately. +/// The maximum reference sample size every comparison uses. const REFERENCE_ROWS: NonZero = NonZero::new(2_048).expect("the reference size is nonzero"); -/// A setup failure that stopped an instrument from reading the published representations. +/// A failure to access the published representations for comparison. #[derive(Debug)] pub(crate) enum SetupError { /// Reading the root's current-generation pointer failed. @@ -119,7 +123,7 @@ impl Error for Audit } } -/// A wall-clock reading, rendered on the scale every instrument measures on. +/// A wall-clock duration formatted in seconds to one decimal place. /// /// Seconds to one decimal, padded to the caller's width so a reading aligns inside a column of /// them. @@ -154,9 +158,8 @@ impl Representations { /// /// # Errors /// - /// Returns a [`SetupError`] when reading the current-generation pointer fails, when the root - /// holds no activated generation, or when the generation or its representation artifact fails - /// to open. + /// Returns [`SetupError`] when the active generation or its representation artifact cannot be + /// opened. fn open(root: &GenerationRoot) -> Result { let id = root .current() @@ -179,7 +182,7 @@ impl Representations { /// /// # Errors /// - /// Returns [`SetupError::Width`] when the artifact holds another element type or width. + /// Returns [`SetupError`] when the artifact holds another element type or width. fn rows(&self) -> Result<&IdSlice>, SetupError> { self.file .vectors() diff --git a/libs/@local/graph/atlas/src/salt/knn/report/tests.rs b/libs/@local/graph/atlas/src/salt/knn/report/tests.rs index 0e40b9d4272..aa71817c2ff 100644 --- a/libs/@local/graph/atlas/src/salt/knn/report/tests.rs +++ b/libs/@local/graph/atlas/src/salt/knn/report/tests.rs @@ -9,13 +9,17 @@ use crate::{ file::array::{ArrayFile, ArrayVariant, ArrayWriter, Dim}, }; -/// A uniquely named file in the system temporary directory, removed on drop. +/// A scratch array file for mapped representation fixtures. struct TempFile { path: PathBuf, } impl TempFile { /// Writes an f32 array file holding one `width`-component row per value, filled with it. + /// + /// # Panics + /// + /// This panics when file creation or array writing fails. fn representation_rows(values: &[f32], width: usize) -> Self { static COUNTER: AtomicU64 = AtomicU64::new(0); @@ -46,6 +50,10 @@ impl Drop for TempFile { } /// Maps `written` as a fixture generation's representations. +/// +/// # Panics +/// +/// This panics when the array file cannot be opened. fn representations(written: &TempFile) -> Representations { Representations { generation: "3a" diff --git a/libs/@local/graph/atlas/src/salt/knn/table.rs b/libs/@local/graph/atlas/src/salt/knn/table.rs index 0db310f1f38..f18ef2509d6 100644 --- a/libs/@local/graph/atlas/src/salt/knn/table.rs +++ b/libs/@local/graph/atlas/src/salt/knn/table.rs @@ -1,4 +1,6 @@ -//! The validated k-nearest-neighbour table. +//! Structural validation and row access for approximate neighbour tables. +//! +//! Table validation establishes the stored shape and distance range. Recall is measured separately. use core::{error::Error, fmt, marker::PhantomData, num::NonZero}; @@ -12,7 +14,7 @@ use sprs::{CsMatI, CsMatViewI}; use super::{Neighbour, construction::NeighbourLists, error::KnnError}; use crate::math::NonNegative; -/// A neighbour matrix violated a [`Knn`] invariant. +/// A violation of a [`Knn`] matrix's domain or neighbour contract. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum KnnValidationError { /// The matrix uses column-compressed storage. @@ -82,6 +84,10 @@ impl Error for KnnValidationError {} /// /// Structural invariants (in-bounds, strictly ascending row entries, consistent pointers) hold for /// any existing [`KnnMatrixView`]. This check covers the domain invariants layered on top. +/// +/// # Errors +/// +/// Returns [`KnnValidationError`] when a domain or neighbour invariant is violated. pub(super) fn validate(matrix: KnnMatrixView<'_>) -> Result<(), KnnValidationError> { if !matrix.is_csr() { return Err(KnnValidationError::ColumnCompressed); @@ -133,9 +139,8 @@ pub(super) fn validate(matrix: KnnMatrixView<'_>) -> Result<(), KnnValidationErr /// The table's matrix layout: `u32` neighbour columns under `u64` row pointers. /// -/// Columns are `u32` because a `u32` encoding carries node rows end to end (the wire row ids, the -/// search backend's item keys, the persisted column region). Row pointers are `u64` so the -/// persisted and resident layouts coincide and a mapped table has the same type as a built one. +/// The column and row-pointer widths match the persisted matrix regions, allowing mapped and +/// resident tables to share the same matrix type. pub(crate) type KnnMatrix = CsMatI; /// A borrowed [`KnnMatrix`]. @@ -144,13 +149,16 @@ pub(crate) type KnnMatrixView<'view> = CsMatViewI<'view, NonNegative, u32, u64>; /// The persisted directed k-nearest-neighbour table of one generation. /// /// A square compressed sparse row matrix over the node-row domain. Row `i` stores the cosine -/// distances of the `k` nearest non-self neighbours of node row `i`, keyed by neighbour row in +/// distances of `k` selected non-self neighbours of node row `i`, keyed by neighbour row in /// ascending row order. Every row stores exactly `k` entries, no row references itself, and every /// distance is finite in `[0, 2]`. Entries within a row are strictly ascending by column, which -/// makes duplicate neighbours unrepresentable. +/// makes duplicate neighbours unrepresentable. These invariants do not establish that the selected +/// neighbours are the exact nearest ones. +/// +/// A `0.0` distance is a stored value, never an absent entry. /// -/// A `0.0` distance is a stored value (duplicate embeddings are exactly coincident), never an -/// absent entry. +/// Validation accepts checked offset row pointers. [`KnnView::row`] requires an initial zero, and +/// writing through [`WriteInto`](crate::file::WriteInto) panics when that pointer is nonzero. #[derive(Debug, Clone)] pub(crate) struct Knn(KnnMatrix, PhantomData); @@ -160,29 +168,26 @@ where { /// Validates a neighbour matrix against the table invariants. /// + /// Accepts offset row pointers without rebasing them. [`Knn`] documents the additional + /// row-access and publication requirements. + /// /// # Errors /// - /// Returns an error when the matrix is not row-compressed, not square over at least two rows, - /// ragged, or self-referencing, when its per-row neighbour count is zero or not below the row - /// count, or when a stored distance lies outside the finite `[0, 2]` range. + /// Returns [`KnnValidationError`] when the matrix violates a table invariant. pub(crate) fn new(matrix: KnnMatrix) -> Result { validate(matrix.view())?; Ok(Self(matrix, PhantomData)) } - /// Slices each row's stored prefix from constructed lists and assembles the validated table. + /// Assembles a validated table from each constructed list's leading prefix. /// /// Each row keeps its `neighbours` nearest entries - the lists' leading prefix - rekeyed into /// the matrix's ascending-column order. /// /// # Errors /// - /// Returns [`KnnError::Invalid`] when the row domain or the assembled table violates a [`Knn`] - /// invariant, [`KnnError::ListsWidth`] when the lists are narrower than the stored width, - /// [`KnnError::TooManyRows`] when the row domain exceeds the table's `u32` column encoding, - /// [`KnnError::TooManyEntries`] when the requested shape overflows the entry count, - /// [`KnnError::NeighbourOutOfBounds`] when a list references a row outside the row domain, and - /// [`KnnError::DuplicateNeighbour`] when a list stores the same neighbour twice. + /// Returns [`KnnError`] for an unsupported table shape, insufficient list width, or invalid + /// selected neighbours. #[tracing::instrument(skip_all)] pub(crate) fn from_lists( lists: &NeighbourLists, @@ -296,6 +301,9 @@ where } /// Borrowed rows of one validated [`Knn`] table. +/// +/// [`Self::row`] additionally requires a zero initial row pointer. Table validation accepts offset +/// pointers without establishing this requirement. #[derive(Debug, Clone)] pub(crate) struct KnnView<'view, N>(KnnMatrixView<'view>, PhantomData); @@ -303,10 +311,9 @@ impl<'view, N> KnnView<'view, N> where N: Id, { - /// Wraps a matrix whose invariants already hold. + /// Borrows a matrix satisfying the table invariants. /// - /// The caller promises the matrix passed [`validate`]; the wrapper performs no checks of its - /// own. + /// The matrix must satisfy [`validate`]. #[inline] #[must_use] pub(super) const fn new_unchecked(matrix: KnnMatrixView<'view>) -> Self { @@ -341,9 +348,17 @@ where /// Returns row `row`'s neighbours in ascending row order. /// + /// The initial row pointer must be zero. [`Knn::new`] accepts offset pointers without rebasing + /// them, while this accessor uses the stored values as indices into the raw column and distance + /// slices. Nonzero offsets can select the wrong entries before a later row exceeds those + /// slices. The row and its successor, and the returned neighbour ids, must be representable by + /// `N`. + /// /// # Panics /// - /// This panics when `row` is outside the table's row domain. + /// Panics when `row` or its successor lies outside the pointer slice, a row pointer does not + /// fit usize, or the resulting range lies outside the raw column or distance slice. Row-id + /// operations also inherit [`Id`]'s domain requirements. pub(crate) fn row(&self, row: N) -> impl Iterator> + 'view { // outer_view reborrows at `&self`; the raw storage carries the // view's own lifetime. diff --git a/libs/@local/graph/atlas/src/salt/knn/tests.rs b/libs/@local/graph/atlas/src/salt/knn/tests.rs index 21e44eedc77..21348f2f489 100644 --- a/libs/@local/graph/atlas/src/salt/knn/tests.rs +++ b/libs/@local/graph/atlas/src/salt/knn/tests.rs @@ -51,6 +51,11 @@ struct Matrix { } impl Matrix { + /// Copies `rows` into the head of the aligned storage. + /// + /// # Panics + /// + /// Panics when `rows` exceeds the fixture capacity of 128. fn new(rows: &[[f32; PROJECTOR_DIMENSIONS]]) -> Self { let mut storage = BoxedVecN::zero(); let (chunks, _) = storage @@ -66,6 +71,7 @@ impl Matrix { } } + /// Borrows the initialized fixture rows as a SIMD-aligned, row-indexed slice. fn view(&self) -> &IdSlice> { IdSlice::from_raw( AlignedVecN::from_slice(&self.storage.as_array()[..self.rows * PROJECTOR_DIMENSIONS]) @@ -75,11 +81,15 @@ impl Matrix { } /// A brute-force reference backend over resident rows. +/// +/// Insertion requires consecutive dense ids starting at the next stored row. Id queries require an +/// already-stored row. Violating either fixture condition panics. struct ExactIndex { rows: Vec>, } impl ExactIndex { + /// Creates an index already holding `rows`. fn from_rows(rows: &[[f32; PROJECTOR_DIMENSIONS]]) -> Self { Self { rows: rows @@ -117,6 +127,11 @@ impl ExactIndex { all } + /// Ranks every other stored row against the stored row `id`. + /// + /// # Panics + /// + /// This panics when `id` is not a resident row. fn ranked_by_id(&self, id: NodeRowId) -> Vec> { let row = usize::try_from(id.as_u64()).expect("test rows fit usize"); self.ranked(&self.rows[row], Some(row)) @@ -184,6 +199,8 @@ struct FarthestIndex(ExactIndex); struct ShortIndex(ExactIndex); /// A misbehaving backend repeating its nearest neighbour. +/// +/// Id searches panic when fewer than two results remain after truncation. struct DoubledIndex(ExactIndex); /// A misbehaving backend naming rows outside the domain. @@ -191,10 +208,14 @@ struct EscapingIndex(ExactIndex); /// A backend degraded by a per-row offset. /// -/// Row `i` skips the nearest `i & 7` candidates, so per-row recall spans a linear ramp and any -/// non-degenerate sample measures real spread. +/// Row `i` skips the nearest `i & 7` candidates. At depth 50 with at least 57 non-self candidates, +/// its recall is `1 - (i & 7) / 50`, ranging from 0.86 to 1.0. struct MixedIndex(ExactIndex); +/// Delegates every index operation except `search_by_id` to [`ExactIndex`]. +/// +/// Expands to an uninhabited `Error` and implementations of `insert_many`, `build`, and +/// `search_by_vector`. A degraded backend supplies only `search_by_id`. macro_rules! delegate_all_but_search_by_id { () => { type Error = !; @@ -302,14 +323,18 @@ impl NearestNeighboursIndex for MixedIndex { } } +/// Creates a row that is zero except for `value` at `component`. +/// +/// # Panics +/// +/// This panics when `component` is outside the projector width. fn axis(component: usize, value: f32) -> [f32; PROJECTOR_DIMENSIONS] { let mut row = [0.0; PROJECTOR_DIMENSIONS]; row[component] = value; row } -/// The vectors `e0`, `e1`, `e0 + e1`, and `-e0`, where known geometry gives every pairwise -/// distance. +/// Creates `e0`, `e1`, `e0 + e1` and `-e0` for hand-derived cosine distances. fn plane_fixture() -> [[f32; PROJECTOR_DIMENSIONS]; 4] { let mut mix = [0.0; PROJECTOR_DIMENSIONS]; mix[0] = 1.0; @@ -317,9 +342,10 @@ fn plane_fixture() -> [[f32; PROJECTOR_DIMENSIONS]; 4] { [axis(0, 1.0), axis(1, 1.0), mix, axis(0, -1.0)] } -/// Distinct unit vectors fanned through the `(0, 1)` plane. +/// Builds planar directions at angular increments of `step`. /// -/// Distance is strictly monotone in index separation. +/// In real arithmetic, cosine distance increases with angular separation within a half turn. The +/// trigonometric rows and distances here retain floating-point rounding. fn fan_fixture(rows: usize, step: f32) -> Vec<[f32; PROJECTOR_DIMENSIONS]> { (0..rows) .map(|index| { @@ -336,10 +362,12 @@ fn fan_fixture(rows: usize, step: f32) -> Vec<[f32; PROJECTOR_DIMENSIONS]> { .collect() } +/// Returns the neighbour width `2` for the small plane fixtures. fn two_neighbours() -> NonZero { NonZero::new(2).expect("two is nonzero") } +/// Creates the fixed-seed generator for neighbour-construction fixtures. fn test_rng() -> Xoshiro256PlusPlus { Xoshiro256PlusPlus::seed_from_u64(0x0A75) } @@ -359,9 +387,8 @@ enum Reported { /// An observer keeping every observation a construction reported, in arrival order. /// -/// Cloneable and shareable because the seam hands the backend an observer of its own: every clone -/// records into the one log, so the backend's phases and the loops around it arrive interleaved as -/// the construction reported them. +/// Every clone, including a detached observer, appends to one shared log. The log preserves the +/// interleaving of backend phases with insertion and readback observations. #[derive(Debug, Clone, Default)] struct RecordingProgress(Arc>>); @@ -374,7 +401,7 @@ impl RecordingProgress { .push(reported); } - /// Every observation so far, in arrival order. + /// Copies every observation so far in arrival order. fn reported(&self) -> Vec { self.0 .lock() @@ -382,14 +409,13 @@ impl RecordingProgress { .clone() } - /// The batches one loop reported, in arrival order. + /// Selects one loop's batches in arrival order. fn batches(&self, select: fn(&Reported) -> Option) -> Vec { self.reported().iter().filter_map(select).collect() } } impl Progress for RecordingProgress { - /// A detached half shares the log, so it records into the same fixture. type Detached = Self; fn detach(&self) -> Self { @@ -413,7 +439,7 @@ impl Progress for RecordingProgress { } } -/// The insertion's batch, when the observation is one. +/// Extracts a batch from an insertion observation. const fn inserted(reported: &Reported) -> Option { match reported { Reported::Insert(batch) => Some(*batch), @@ -421,7 +447,7 @@ const fn inserted(reported: &Reported) -> Option { } } -/// The readback's batch, when the observation is one. +/// Extracts a batch from a readback observation. const fn readback(reported: &Reported) -> Option { match reported { Reported::Readback(batch) => Some(*batch), @@ -429,8 +455,7 @@ const fn readback(reported: &Reported) -> Option { } } -/// The smallest lists `Knn::from_lists` accepts: one neighbour per row over three rows, each -/// column distinct from its own row. +/// Creates a three-row cycle with one non-self neighbour per row. fn tiny_neighbour_lists() -> NeighbourLists { let entries: Box<[Neighbour]> = Box::new([ Neighbour { @@ -449,15 +474,6 @@ fn tiny_neighbour_lists() -> NeighbourLists { NeighbourLists::new(entries, 1) } -/// `Knn::from_lists` iterates rows through rayon's `par_chunks_mut`, and rayon's thread pool runs -/// under miri at real cost. This measures the smallest fixture's miri wall time so the review -/// letter can cite it, rather than assume it. -/// -/// Measured: under miri this fails rather than merely running slowly. Rayon's crossbeam-epoch -/// registry trips the same known Stacked Borrows false positive already worked around on -/// `build_matches_hand_computed_neighbours` above, about 1m46s wall clock in, at the failing -/// retag. The site is downgraded to out-with-reason on this measurement; the ignore below is the -/// existing crate pattern for the same false positive, not a new one. #[test] fn from_lists_over_the_smallest_fixture_measures_the_miri_cost() { let lists = tiny_neighbour_lists(); @@ -470,14 +486,16 @@ fn from_lists_over_the_smallest_fixture_measures_the_miri_cost() { assert_eq!(knn.view().rows(), 3); assert_eq!(knn.view().neighbours(), 1); - // Not a correctness assertion: printed so the harness's transcript carries the measurement - // whether or not `--nocapture` is passed by the caller measuring it under miri. if std::env::var_os("MIRI_KNN_TIMING").is_some() { eprintln!("from_lists_over_the_smallest_fixture_measures_the_miri_cost: {elapsed:?}"); } } /// Constructs lists over `embeddings` through an initially empty backend. +/// +/// # Errors +/// +/// Returns [`KnnError`] when construction fails. fn lists_via( index: I, embeddings: &IdSlice>, @@ -490,6 +508,8 @@ where IndexConstruction::new(index).construct(embeddings, width, test_rng(), &NoProgress) } +// the diagonal angle is π/4, giving cosine distance 1 − 1/√2. Stored distances are the crate +// kernel's rounded values, compared bit-exactly below. #[test] fn build_matches_hand_computed_neighbours() { let rows = plane_fixture(); @@ -655,10 +675,8 @@ fn descent_converges_on_known_geometry() { matched += ids.iter().filter(|id| reference.contains(id)).count(); } - // The join's update application is parallel and unordered, so the - // converged lists are not replayable; on this smooth fan geometry - // the join converges to (near-)exact lists under any order, and - // the bound leaves room for the residual variance. + // parallel update order can change the selected lists. The match threshold allows residual + // variation while requiring near-exact recovery on this fan. assert!( matched >= 230, "descent matched {matched}/256 exact neighbours", @@ -667,9 +685,7 @@ fn descent_converges_on_known_geometry() { #[test] fn descent_passes_the_admission_gate() { - // l2-normalized, honouring the construction's input contract: the - // production pipeline admits representations through the norm spot - // check before any construction sees them. + // l2-normalize the fixture to satisfy the construction's input contract. let rows: Vec<[f32; PROJECTOR_DIMENSIONS]> = { let mut rng = Xoshiro256PlusPlus::seed_from_u64(7); core::iter::repeat_with(|| { @@ -693,8 +709,7 @@ fn descent_passes_the_admission_gate() { let matrix = Matrix::new(&rows); let embeddings = matrix.view(); - // The width is the wider of the spot check's depth and the stored count, so the admitted lists - // and the persisted table are the same lists. + // one construction supplies the full recall depth and the narrower stored prefix. let width = recall::SpotCheckOptions::default() .neighbours .max(DEFAULT_NEIGHBOURS); @@ -770,8 +785,8 @@ fn an_observed_construction_reports_its_insertion_then_its_readback() { ) .expect("the fixture is well-formed"); - // A corpus below the report cadence reports each loop once, as its last row lands. This backend - // names no build phases, so the whole log is the two loops' completions, insertion first. + // below the cadence, each loop reports exactly once at its last row. This backend names no + // build phases between the loops. let complete = Batch { done: 64, total: 64, @@ -782,6 +797,8 @@ fn an_observed_construction_reports_its_insertion_then_its_readback() { ); } +// the exact fixture backend is deterministic, isolating the observer's effect from parallel +// construction variability. #[test] fn watching_a_construction_does_not_change_its_lists() { let rows = fan_fixture(64, 0.02); @@ -854,9 +871,7 @@ fn an_observed_descent_reports_every_iteration_it_ran() { ); } - // The reading is a convergence reading: the join accepts less as - // the lists sharpen, and the last iteration is the one that met the - // termination threshold. + // compare only the endpoints: parallel joins need not produce a monotone accepted rate. let [first, last] = [iterations.first(), iterations.last()].map(|reading| { reading .expect("the construction ran at least one iteration") @@ -1002,8 +1017,7 @@ fn spot_check_fails_a_degraded_backend() { #[test] fn spot_check_honours_configured_options() { - // The same degraded backend passes under a laxer minimum: the - // criterion travels with the options, and the evidence records it. + // the farthest-50 backend has recall 41/50 = 0.82 on 60 rows, above this 0.8 minimum. let rows = fan_fixture(60, 0.02); let matrix = Matrix::new(&rows); let index = FarthestIndex(ExactIndex::from_rows(&rows)); @@ -1098,12 +1112,10 @@ fn spot_check_sizes_the_verdict_sample_to_the_measured_clearance() { let matrix = Matrix::new(&rows); let index = MixedIndex(ExactIndex::from_rows(&rows)); - // Row `i` matches exactly `50 - (i & 7)` of its exact top 50, so - // per-row recall ramps 0.86..1.0 and any pilot mixing residues - // measures real spread. Against a minimum a third of a percent - // below the aggregate, the clearance the pilot measures sizes a - // sample far past the corpus, so the verdict sample is exhaustive: - // ids 0..59 sum their residues to 7 · 28 + 6 = 202 skipped rows. + // row `i` matches exactly `50 - (i & 7)` of its exact top 50, giving per-row recall 0.86..1.0. + // Across ids 0..59, residues sum to 7 · 28 + 6 = 202 skipped rows. Aggregate recall is (3000 − + // 202) / 3000 ≈ 0.9327. The seeded pilot's clearance from 0.93 requests at least the corpus + // size. let check = recall::spot_check( &index, matrix.view(), @@ -1120,14 +1132,11 @@ fn spot_check_sizes_the_verdict_sample_to_the_measured_clearance() { assert_eq!(check.matched, 60 * 50 - 202); assert_eq!(check.expected, 60 * 50); assert!(check.deviation > d_non_negative!(0.0)); - // A census of the corpus leaves no sampling error to bound, so the - // aggregate itself clears the minimum. + // a census has zero sampling width. assert_eq!(check.resolution, d_non_negative!(0.0)); assert_eq!(check.admission(), recall::RecallAdmission::Admitted); } -/// A budget that buys nothing leaves the verdict sample at the pilot's size, and a shortfall it -/// cannot demonstrate reads as unresolved rather than refused. #[test] fn spot_check_stops_at_the_sampling_budget() { let rows = fan_fixture(60, 0.02); @@ -1154,10 +1163,8 @@ fn spot_check_stops_at_the_sampling_budget() { check.sampled_rows, 4, "the budget buys no rows past the pilot" ); - // Those four rows skip 18 of their 200 exact neighbours, so the - // aggregate reads 0.91 - below the minimum, and by less than the - // 0.047 such a sample resolves. A shortfall the sample cannot - // demonstrate is not a refusal. + // four rows skip 18 of their 200 exact neighbours: recall is 182/200 = 0.91. The achieved width + // below exceeds the 0.94 − 0.91 = 0.03 shortfall. This sample cannot demonstrate a refusal. assert_eq!(check.matched, 200 - 18); assert_eq!(check.recall(), 0.91); assert!(check.recall() < check.minimum_recall.get()); @@ -1330,8 +1337,8 @@ fn hannoy_honours_the_seam_contract() { (0.0..=2.0).contains(&neighbour.distance), "distances arrive on the [0, 2] cosine scale", ); - // hannoy accumulates in f32; the rescaled seam distance agrees - // with the crate kernel to vector-sum rounding. + // hannoy and the crate kernel accumulate separately in f32. Allow their rounded distances + // to differ within this fixture's tolerance. assert!( (neighbour.distance - exact).abs() < 1e-4, "backend distance {} disagrees with exact distance {exact}", @@ -1359,7 +1366,7 @@ fn hannoy_honours_the_seam_contract() { let _: Result<(), std::io::Error> = std::fs::remove_dir_all(&dir); } -/// 128 fixture rows, each component drawn uniformly from `[-1, 1)` under seed 7. +/// Draws 128 fixture rows with components uniform in `[-1, 1)` under seed 7. fn uniform_rows() -> Vec<[f32; PROJECTOR_DIMENSIONS]> { let mut rng = Xoshiro256PlusPlus::seed_from_u64(7); core::iter::repeat_with(|| { @@ -1375,8 +1382,13 @@ fn uniform_rows() -> Vec<[f32; PROJECTOR_DIMENSIONS]> { /// Asserts the production path admits the fixture rows. /// -/// The production path puts a fresh backend behind the construction wrapper. Exact recall admits -/// the lists, and the table comes from those same lists. +/// A fresh backend supplies the lists for both the recall check and the table. +/// +/// # Panics +/// +/// This panics when the fixture directory or index cannot be prepared, construction or table +/// validation fails, the fixture counts differ from 128 rows and 50-wide lists, or the recall check +/// does not admit the lists. #[track_caller] fn assert_construction_path_admits( base: &camino::Utf8Path, diff --git a/libs/@local/graph/atlas/src/salt/ladder/error.rs b/libs/@local/graph/atlas/src/salt/ladder/error.rs index 29324881207..fdac70a2789 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/error.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/error.rs @@ -9,7 +9,7 @@ use crate::math::NonNegative; pub(crate) enum ConditionsError { /// Fewer than two steps: nothing to compare across. TooFew { - /// steps offered. + /// Steps offered. count: usize, }, /// The schedule's first step is not the exact zero-condition value `0.0`. @@ -64,7 +64,7 @@ impl Error for ConditionsError {} pub(crate) enum LadderError { /// The field count does not match the schedule. FieldCount { - /// steps in the schedule. + /// Steps in the schedule. conditions: usize, /// Fields offered. fields: usize, @@ -80,7 +80,8 @@ pub(crate) enum LadderError { }, /// A step's field has no Procrustes alignment onto the compared field. /// - /// Its points are coincident or the covariance cancels exactly. + /// Alignment fails when the field has fewer than two rows, when its rounded variance or + /// covariance is unusable, or when the coefficients lie outside the accepted `f32` range. Degenerate { /// Position of the unalignable field. index: usize, @@ -123,7 +124,7 @@ impl Error for LadderError {} /// A rejected canonical selection. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum CanonicalError { - /// The requested value is not a step of the measured ladder. + /// The requested value is absent from the supplied conditions. UnknownStep { /// The requested condition. value: NonNegative, diff --git a/libs/@local/graph/atlas/src/salt/ladder/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/mod.rs index 24f9ca20d35..0f98e1e68ba 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/mod.rs @@ -1,32 +1,35 @@ -//! The condition ladder. +//! Projected layouts across relation-lens conditions, aligned for comparison. //! -//! One projected layout per relation-lens condition, aligned and measured against its neighbours. +//! A generation publishes one coordinate field: the configured canonical step's layout aligned into +//! the baseline frame. Every other step is a measurement counterfactual, the same jointly trained +//! [projector](crate::salt::projector) evaluated at a different lens strength. Its measurements +//! persist as evidence, and its coordinates never publish as the canonical field. //! -//! A generation publishes one coordinate field, the configured canonical step's layout aligned into -//! the baseline frame. Every other step is a measurement counterfactual (the same jointly trained -//! model projected at a different lens strength) that persists as evidence and never publishes as -//! the coordinate field. +//! [`Conditions`] defines the zero-condition step at `0.0`, with relation conditioning disabled in +//! the jointly trained model. This baseline is not a separately trained relation-free model. The +//! steps ascend strictly and every value is finite. The schedule has no configured step-count cap. +//! Projection evaluates every step. Each non-baseline measurement then traverses the paired fields +//! for two fits and two residuals. The baseline's alignment is the identity and both of its +//! movements are zero. //! -//! [`Conditions`] carries the schedule, valid by construction. The schedule opens at the -//! zero-condition value `0.0`, the jointly trained model with the lens off rather than a -//! relation-free model trained on its own. The steps ascend strictly and every value is finite. The -//! step count has no upper bound. Each step costs one projection pass plus four alignment passes, -//! so the schedule length is configuration rather than a format limit. +//! [`measure_ladder`] fits each non-baseline field onto the baseline and separately onto its +//! predecessor using unweighted, orientation-preserving Procrustes alignment +//! ([`Similarity::fit_uniform_par`]). For source points sᵢ and target points tᵢ over N +//! corresponding rows, the model chooses scale a > 0, rotation R and translation b to minimize Σᵢ +//! ‖aR sᵢ + b − tᵢ‖². Movement is √(Σᵢ ‖aR sᵢ + b − tᵢ‖²/N). It measures residual deformation after +//! removing the source's global similarity freedom, in the target frame's units. Scaling the target +//! scales this residual, and reflections remain outside the fit family. //! -//! [`measure_ladder`] derives each step's evidence. Every step aligns onto the baseline and onto -//! its predecessor with the unweighted Procrustes fit ([`Similarity::fit_uniform_par`]); the RMS -//! movement the alignment cannot explain is the step's real geometric change, invariant under the -//! scale, rotation, and translation freedom the projector never promises to pin down. The -//! measurements are diagnostics: they persist as evidence and surface as structured log events, and -//! they never block publication. +//! Fits accumulate moments in `f64` and narrow their coefficients to `f32`. Residuals apply the +//! widened coefficients in `f64`. Rounding, cancellation and parallel summation limit exact +//! invariance and bitwise repeatability. A field can fail alignment despite having finite +//! coordinates. Measurement errors abort this operation. Successful movement and loss values are +//! diagnostics, with no acceptance threshold here. //! -//! [`select_canonical`] names the step that publishes as the canonical field. The configured -//! condition names an exact member of the measured schedule, so configuration picks a step rather -//! than an interpolation point. -//! -//! Projection itself is the conditioned projector's inference (`salt/projector`), and the per-step -//! relation loss is its frozen objective. Both enter here as constructed domain values, so the -//! boundary between the stages stays artifact-level. +//! [`select_canonical`] selects the configured condition by exact membership, without interpolation +//! or a quality-based choice. Projection and frozen relation-loss evaluation happen before +//! [`Field`] construction. This module checks field counts and lengths, while row correspondence +//! and common model provenance remain input requirements. use alloc::borrow::Cow; @@ -58,8 +61,8 @@ pub(crate) struct Conditions { impl Conditions { /// The reference schedule, the baseline plus four evenly spaced steps. /// - /// An unvalidated starting point carried over from the legacy pipeline. The ladder's own - /// movement evidence revises it. + /// The values are structurally valid. Their spacing is an uncalibrated starting point to + /// revisit using the ladder's movement evidence. pub(crate) const REFERENCE: Self = Self { values: Cow::Borrowed(&[ NonNegative::new_unchecked(0.0), @@ -74,8 +77,8 @@ impl Conditions { /// /// # Errors /// - /// Returns an error when the schedule has fewer than two steps, the first step is not zero, - /// or a step does not strictly exceed its predecessor. + /// Returns [`ConditionsError`] when the values violate the schedule's minimum length, baseline + /// or ordering. pub(crate) fn new( values: impl Into>, ) -> Result { @@ -131,8 +134,9 @@ const impl Default for Conditions { /// One step's projected field with its frozen relation loss. /// -/// `I` is the step frames' shared row domain. The coordinates arrive proven finite, so the -/// alignment fits consume them with no rescan and a non-finite frame is unrepresentable here. +/// `I` is the step frames' shared row domain. Coordinates must obey [`FinitePointField`]'s +/// finiteness contract, and each row must identify the same subject in every field. The field does +/// not validate the relationship between coordinates and loss. #[derive(Debug, Copy, Clone)] pub(crate) struct Field<'coordinates, I> { /// The step's projected coordinates, row-aligned with every other step's. @@ -143,9 +147,7 @@ pub(crate) struct Field<'coordinates, I> { pub relation_loss: DNonNegative, } -/// One fit's whole ladder configuration. -/// -/// The schedule and the step that publishes. +/// The projection schedule and the condition selected for canonical coordinates. /// /// The canonical value names a schedule member exactly ([`select_canonical`]): equality on /// [`NonNegative`] is bit equality. A value outside the schedule is a configuration @@ -153,11 +155,11 @@ pub(crate) struct Field<'coordinates, I> { /// so a fit refuses the contradiction before it trains. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct LadderOptions { - /// The condition schedule the ladder projects. + /// The condition schedule, [`Conditions::REFERENCE`] by default. pub conditions: Conditions = Conditions::REFERENCE, /// The condition whose aligned field publishes as the canonical coordinates. /// - /// `1.0` is the full-strength lens, matching the reference pipeline's canonical condition. + /// The full-strength condition `1.0` by default. pub canonical: NonNegative = NonNegative::ONE, } @@ -170,13 +172,11 @@ const impl Default for LadderOptions { impl LadderOptions { /// Returns the canonical step's position in the schedule. /// - /// The canonical value names a schedule member exactly, so the index exists exactly when - /// the configuration is self-consistent. The membership is a property of the options alone, - /// decidable before any step projects. + /// Membership depends only on these options and can be checked before projection. /// /// # Errors /// - /// Returns an error when the canonical value names no step of the schedule. + /// Returns [`CanonicalError`] when the canonical value names no schedule member. pub(crate) fn canonical_index(&self) -> Result { canonical_position(self.conditions.values().iter().copied(), self.canonical) } @@ -184,8 +184,11 @@ impl LadderOptions { /// Returns the position of the canonical `value` among `conditions`. /// -/// The one membership rule both the options check and the measured selection apply: equality on -/// [`NonNegative`] is bit equality. +/// Returns the first match using [`NonNegative`]'s bit equality. +/// +/// # Errors +/// +/// Returns [`CanonicalError`] when no condition matches `value`. fn canonical_position( mut conditions: impl Iterator, value: NonNegative, @@ -206,24 +209,28 @@ pub(crate) struct StepMeasurement { /// /// The identity for the baseline itself. pub alignment: Similarity, - /// RMS movement against the baseline field after alignment. + /// RMS movement after alignment onto the baseline, in baseline-frame units. pub baseline_movement: DNonNegative, - /// RMS movement against the preceding field after alignment. + /// RMS movement after alignment onto the predecessor, in predecessor-frame units. pub adjacent_movement: DNonNegative, } /// Aligns and measures a condition ladder. /// -/// `fields[i]` is the projection of the whole corpus at `conditions.values()[i]`; rows correspond -/// across fields. Each non-baseline step fits its alignment onto the baseline and onto its -/// predecessor in parallel, and the returned measurements carry one entry per step in schedule -/// order. +/// `fields[i]` must be the whole-corpus projection at `conditions.values()[i]`, with corresponding +/// rows across every field. Non-baseline steps fit onto the baseline and then onto their +/// predecessor. Each fit and residual uses parallel reductions. The result has one measurement per +/// step in schedule order and echoes every supplied loss without recomputing it. +/// +/// # Complexity +/// +/// O(SN) work and O(S) result storage for S steps of N rows, excluding the already supplied +/// coordinate fields. /// /// # Errors /// -/// Returns an error when the field count does not match the schedule, a field's rows differ from -/// the baseline's, or a field admits no similarity alignment (coincident points or an exactly -/// cancelling covariance). +/// Returns [`LadderError`] for field-count or row-count disagreement, or when a similarity fit +/// rejects a pair. pub(crate) fn measure_ladder( conditions: &Conditions, fields: &[Field<'_, I>], @@ -276,7 +283,7 @@ pub(crate) fn measure_ladder( Ok(measurements) } -/// The step authorized to publish as the canonical field. +/// The selected step's position and baseline alignment. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct CanonicalSelection<'ladder> { /// The step's position in the schedule. @@ -292,12 +299,13 @@ pub(crate) struct CanonicalSelection<'ladder> { /// Selects the step publishing as the canonical field. /// -/// The value must be an exact member of the measured schedule, so the canonical condition names -/// an existing step. Equality on [`NonNegative`] is bit equality. +/// The value must be an exact member of `measurements`. Equality on [`NonNegative`] is bit +/// equality. This returns the first matching entry without checking schedule order, row +/// correspondence or measurement quality. /// /// # Errors /// -/// Returns an error when the value names no step. +/// Returns [`CanonicalError`] when the value names no step. pub(crate) fn select_canonical( measurements: &[StepMeasurement], value: NonNegative, @@ -313,11 +321,12 @@ pub(crate) fn select_canonical( }) } -/// Fits the alignment of `source` onto `target` and measures the RMS movement it cannot explain. +/// Fits `source` onto `target` and measures the residual in target-frame units. /// -/// Returns [`None`] exactly when the fit rejects the pair as degenerate: fewer than two rows, -/// coincident points, or an exactly cancelling covariance. The residual of a successful fit -/// over the proven-finite fields is total. +/// Returns [`None`] when [`Similarity::fit_uniform_par`] rejects the pair, including length +/// disagreement, fewer than two rows, unusable rounded moments or unrepresentable coefficients. A +/// successful fit establishes the equal, nonempty fields and finite coefficients required by the +/// residual. fn aligned_movement( source: &FinitePointField, target: &FinitePointField, diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/census/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/census/mod.rs index 244d059f363..6883cd9729c 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/census/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/census/mod.rs @@ -1,25 +1,26 @@ -//! The candidate census and the draw. +//! Candidate populations and bounded, repeatable paired-movement samples. //! //! [`Draw::over`] samples one generation's attraction index. The census walks two candidate //! domains. The pair domain holds every distinct oriented `(source, target)` pair among the -//! force-bearing Proximal instances: the edges of the groups whose Proximal class weight is -//! positive, deduplicated across groups with orientation kept. The control domain holds every -//! nonparticipant corpus row: a row that no retained instance of any force class names as an -//! endpoint. +//! [force-bearing Proximal instances](crate::salt::relation::attraction::AttractionEdge): the edges +//! of the groups whose Proximal class weight is positive, deduplicated across groups with +//! orientation kept. The control domain holds every nonparticipant corpus row: a row that no +//! retained instance of any force class names as an endpoint. //! //! The draw orders each domain ascending by `(order key, subject)` and keeps a bounded prefix: //! `n = min(P, SAMPLE_CAP)` of the `P` candidate pairs and `m = min(Q, n)` of the `Q` candidate -//! rows. The subject tie-break keeps the order total without assuming the keyed hash never -//! collides, so the draw is a function of the rule and the salt over the index bytes alone. An -//! empty pair domain short-circuits into the `P = 0` outcome: zero counts on both domains -//! and no control population at all. +//! rows. The subject tie-break keeps the order total even when digests collide. The draw is a +//! function of the rule, salt, corpus row count and index regions. An empty pair domain +//! short-circuits into the `P = 0` outcome: zero counts on both domains and no control population +//! at all. //! //! Scratch stays bounded by the index and the draw. The deduplication buffer holds the Proximal //! instances and the participant set spends one bit per corpus row, while each selection works //! in a heap of at most its own sample size. The census refuses an index whose group ranges or edge //! endpoints contradict its own geometry ([`CensusError`]) instead of reading around the -//! contradiction; every other domain rule stays `salt::relation`'s artifact contract, validated -//! where the domain types live. +//! contradiction. Supply regions satisfying the [attraction +//! index](crate::salt::relation::attraction::AttractionIndex)'s remaining invariants. The census +//! checks neither complete group coverage of the edge region nor each edge's force factor. #[cfg(test)] mod tests; @@ -36,17 +37,19 @@ use crate::{ identity::{EdgeRowId, NodeRowId}, }; -/// The pair-sample cap. +/// The maximum number of sampled pairs. /// -/// The Dvoretzky-Kiefer-Wolfowitz bound `2 · exp(−2 · n · ε²) ≤ δ` at `ε = 0.01` and `δ = 10⁻⁶` -/// has the exact integer minimum 72,544, and the cap adds an eleven-row margin above it. A capped -/// draw therefore holds the sample's whole empirical distribution within one percentile point of -/// its population's, with failure probability at most one in a million, and a smaller pair domain -/// draws whole. -// `pub(super)`: the evidence body documents its `pairs_selected` bound by naming this cap. +/// The cap uses the Dvoretzky-Kiefer-Wolfowitz calibration for an independent random sample: 2 · +/// exp(−2nε²) ≤ δ, where n is the sample count, ε is the maximum empirical-CDF error and δ is the +/// failure-probability bound. At ε = 0.01 and δ = 10⁻⁶, n ≥ ⌈ln(2/δ)/(2ε²)⌉ = 72,544. The cap adds +/// eleven rows. +/// +/// This deterministic digest-ordered, without-replacement draw does not itself establish that +/// random-sampling model or its probability guarantee. A pair domain at or below the cap is +/// measured in full. The calibration supplies no per-stratum control guarantee. pub(super) const SAMPLE_CAP: usize = 72_555; -/// The index contradiction the census refused. +/// An invalid group range or endpoint encountered during the census. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum CensusError { /// A group's edge range contradicts the edge region. @@ -55,8 +58,9 @@ pub(crate) enum CensusError { group: u64, /// The range's first edge position. start: u64, - /// The range's one-past-last edge position, the next group's start or the edge count - /// for the final group. + /// The range's one-past-last edge position. + /// + /// The next group's start, or the edge count for the final group. end: u64, /// The edge count the range must stay within. edges: u64, @@ -97,7 +101,7 @@ where impl Error for CensusError where I: Id {} -/// One oriented candidate pair, the source row and then the target row. +/// An oriented source-target pair eligible for sampling. /// /// The derived order is the draw's subject tie-break: ascending `(source, target)`. The subject /// encoding behind the primary key is the rule's ([`DrawRule::pair_order_key`]). @@ -111,16 +115,16 @@ pub(crate) struct Pair { /// One completed draw over a generation's attraction index. /// -/// The selections keep draw order, ascending `(order key, subject)`, the order every downstream -/// fold consumes. The candidate counts census the whole domains, so evidence records candidates -/// beside selections without retaining an identity. +/// The selections keep draw order, ascending `(order key, subject)`. Candidate counts cover the +/// whole corresponding domains when the pair population is nonempty. An empty pair population +/// records zero controls without counting nonparticipants. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct Draw { /// The distinct force-bearing Proximal pair count `P`. pair_candidates: u64, /// The `n = min(P, SAMPLE_CAP)` drawn pairs, in draw order. pairs: Vec, - /// The nonparticipant corpus row count `Q`. + /// The nonparticipant corpus row count `Q`, or zero when no pairs were eligible. control_candidates: u64, /// The `m = min(Q, n)` drawn control rows, in draw order. controls: Vec, @@ -130,9 +134,19 @@ impl Draw { /// Takes one generation's draw over its attraction index. /// /// `rows` is the corpus row count the endpoints index into, and `groups` and `edges` are the - /// index's regions in file order. [`AttractionFile`] hands out all three. One rule and salt - /// over one index always produce one draw, so a replay that re-derives the salt re-derives - /// the selections. + /// index's regions in file order, obtainable from [`AttractionFile`]. The same rule, salt, row + /// count and regions always produce the same selections. A positive Proximal group weight + /// admits all its edges, including self-pairs, without rechecking their confidence or strength. + /// + /// # Complexity + /// + /// For G groups, E edges, L Proximal-group edges, N rows and quotas n and m, work is O(G + E + + /// L log(L + 1) + P log(n + 2) + N log(m + 2)). Scratch is O(G + L + N/8 + n + m) bytes up to + /// record-size factors. When P = 0, the participant allocation and row sweep are skipped. + /// + /// # Panics + /// + /// Panics for a nonempty pair domain when `rows` exceeds [`usize::MAX`]. /// /// # Errors /// @@ -265,14 +279,11 @@ impl Draw { /// Marks every corpus row a retained instance names as an endpoint. /// -/// The complement is the control candidate domain. [`Draw::over`] censuses its control pool from -/// this set, and the evidence writer re-derives it for the collateral strata's candidate sweep, -/// so both walks share one participant definition. +/// The complement defines control eligibility for sampling and for the collateral-stratum census. /// /// # Panics /// -/// This panics when an edge names an endpoint at or beyond `rows`. The census's endpoint sweep -/// establishes the bound before either caller arrives here. +/// Panics when `rows` exceeds [`usize::MAX`] or an endpoint is outside `0..rows`. pub(super) fn participants( rows: u64, edges: &[EdgeRecord], @@ -291,8 +302,12 @@ pub(super) fn participants( /// Resolves each group's edge range, refusing boundaries the edge region contradicts. /// /// Group `i` spans `first_edge[i] .. first_edge[i + 1]`, with the final group ending at the edge -/// count. A backwards boundary or one past the region has no consistent reading, so the census -/// refuses it rather than walking a range the file cannot hold. +/// count. The first range need not start at zero, and an empty group slice returns no ranges even +/// when edges exist. +/// +/// # Errors +/// +/// Returns [`CensusError`] for a backwards boundary or a boundary past the edge count. fn edge_ranges( groups: &[GroupRecord], edges: &[EdgeRecord], @@ -323,8 +338,9 @@ fn edge_ranges( /// Selects the `quota` least candidates, ascending. /// -/// A bounded max-heap carries the running selection: a candidate below the current worst -/// replaces it, so the walk streams its domain while scratch stays proportional to the quota. +/// Streams the candidates through a bounded max-heap, replacing the current largest selected value +/// when a smaller candidate appears. Returns at most `quota` entries, or an empty vector at quota +/// zero. Storage is O(quota), and each candidate takes O(log(quota + 2)) work. fn select(candidates: impl Iterator, quota: usize) -> Vec { let mut selected = BinaryHeap::with_capacity(quota); for candidate in candidates { diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/census/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/census/tests.rs index d64de0bfa5a..26e140628ba 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/census/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/census/tests.rs @@ -1,9 +1,9 @@ //! Candidate census and draw expectations. //! -//! The oracles restate the draw as a full sort over independently recollected candidate -//! domains, so the bounded selection is checked against a plain statement of the same order. -//! The subject keys are the rule's own ([`DrawRule::pair_order_key`], -//! [`DrawRule::row_order_key`]), whose encodings identity's own tests pin independently. +//! Full-sort oracles independently collect both candidate domains and compare their ordered +//! prefixes with bounded selection. The subject keys are the rule's own +//! ([`DrawRule::pair_order_key`], [`DrawRule::row_order_key`]), whose encodings identity's own +//! tests pin independently. //! //! [`DrawRule::pair_order_key`]: crate::salt::ladder::paired::identity::DrawRule::pair_order_key //! [`DrawRule::row_order_key`]: crate::salt::ladder::paired::identity::DrawRule::row_order_key @@ -20,6 +20,7 @@ use crate::{ }, }; +/// Builds a source-target pair from literal row numbers. fn pair(source: u64, target: u64) -> Pair { Pair { source: node(source), @@ -27,7 +28,7 @@ fn pair(source: u64, target: u64) -> Pair { } } -/// An index of two Proximal-bearing groups around one Coincident-only group. +/// Builds two Proximal-bearing groups around one Coincident-only group. /// /// The Proximal groups duplicate one pair across groups and carry one orientation-reversed pair /// and one self-pair. The Coincident-only group's endpoints participate without pairing. The @@ -49,7 +50,11 @@ fn mixed_index() -> (Vec, Vec>) { ) } -/// Restates the pair draw as a full sort: gate, dedup, key, prefix. +/// Selects distinct oriented Proximal pairs by full sort and prefix. +/// +/// # Panics +/// +/// Panics if a group offset does not fit `usize` or its edge range is invalid. fn oracle_pairs( salt: DrawSalt, groups: &[GroupRecord], @@ -267,7 +272,8 @@ fn one_index_one_draw_and_a_rotated_salt_permutes() { let rotated = Draw::over(rule(), rotated_salt, rows, &groups, &edges) .expect("the fixture index is well-formed"); - // Both pools are thin, so a rotated salt permutes each whole domain rather than reselecting. + // both pools fit within their quotas. These salts change the order, while each selected set + // remains the full domain. assert_ne!( rotated.pairs(), first.pairs(), diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/mod.rs index dc25b678c05..88c0fbbc289 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/mod.rs @@ -1,25 +1,27 @@ -//! The persisted paired-movement evidence body and its aggregation. +//! Persisted paired-movement outcomes and distribution summaries. //! -//! [`PairedMovementEvidence`] is the block the metadata document embeds beside the step -//! measurements: the draw metadata a replay re-derives, then a tri-state outcome. A -//! [`MovementOutcome::Measured`] body carries the aggregate families, and the other two -//! structurally cannot. [`MovementOutcome::Vacuous`] records an empty pair domain, while -//! [`MovementOutcome::Failed`] retains a typed refusal beside the completed draw counts. Every -//! aggregate family sits beside its population count, and its value fields exist exactly when -//! that count is positive, so neither an empty population nor an empty stratum can read as a -//! measured zero. The body holds no pair or row identity, and its size is a function of the -//! quantile grid and the strata alone. +//! [`PairedMovementEvidence`] records draw metadata beside an outcome. +//! [`MovementOutcome::Measured`] carries aggregate families. [`MovementOutcome::Vacuous`] records +//! an empty pair domain, and [`MovementOutcome::Failed`] retains a typed refusal with any completed +//! draw counts. Successful bodies omit selected identities. An endpoint failure can name a rejected +//! row. Produced bodies have a fixed quantile grid and at most ten strata. //! -//! Aggregation forms each per-pair difference from one reading's own fields in `f64` and never -//! subtracts step aggregates. Means commute with subtraction while fractions and quantiles do -//! not, so the shortcut would fabricate readings no pair produced. Quantiles follow the nearest -//! rank, and every accumulating sum is one serial `f64` fold in draw order, so one draw -//! reproduces its aggregates bit for bit. +//! Aggregation forms every per-pair difference in `f64` and never subtracts step aggregates. For +//! pair i, Δdᵢ = dᵢ,canonical − dᵢ,zero and Δrᵢ = rankᵢ,canonical − rankᵢ,zero. Negative +//! differences indicate contraction or rank improvement. Fractions count strict negatives, and +//! nearest-rank quantiles summarize the difference populations. Means commute with subtraction in +//! real arithmetic, but separate rounded folds can differ. Quantiles and contraction fractions +//! require the paired readings themselves. //! -//! The collateral strata stand on the candidate population. Every nonparticipant row's -//! anchor-distance reading defines the ten boundaries, so the strata are a function of the -//! census rather than the draw, and each stratum's candidate and selected counts read how the -//! draw spread across the census. +//! Every accumulating sum uses a serial `f64` fold in draw order. Equal ordered readings reproduce +//! the same aggregates under the same arithmetic semantics. This does not establish equal upstream +//! frames or cross-platform square-root results. +//! +//! Collateral strata use all nonparticipants' distances to the sampled pair endpoints. Their +//! boundaries depend on the pair sample but not on which controls were selected. Candidate and +//! selected counts show the control sample's distribution across those boundaries. The producers +//! attach a displacement family exactly when a stratum has selected rows. These relationships are +//! not validated by deserialization. #[cfg(test)] mod tests; @@ -41,10 +43,13 @@ const DECILES: u32 = 10; /// The paired-movement evidence body of one ladder record. /// -/// The body persists no pair or row identity. A corpus holder re-derives the selected -/// identities by recognizing the rule, re-deriving the salt, and rerunning the keyed order, so -/// the omission limits the payload rather than claiming secrecy. The outcome -/// flattens beside these fields under its `outcome` tag. +/// A successful body omits selected pair and row identities. They can be re-derived from the same +/// metadata, attraction index and draw conventions. This limits payload size without providing +/// secrecy. An endpoint failure can include a rejected row. +/// +/// The outcome flattens beside the draw metadata under its `outcome` tag. Public fields and derived +/// deserialization do not validate count relationships, rule recognition or consistency between an +/// outcome and its metadata. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct PairedMovementEvidence { /// The draw rule that produced the sample. @@ -55,10 +60,11 @@ pub(crate) struct PairedMovementEvidence { pub rank_window: u64, /// The distinct force-bearing Proximal pair count `P`. pub pair_candidates: u64, - /// The drawn pair count `n`, the candidates bounded by the pair-sample cap - /// ([`SAMPLE_CAP`](super::census::SAMPLE_CAP)). + /// The number of drawn pairs. + /// + /// The readout bounds this count by [`SAMPLE_CAP`](super::census::SAMPLE_CAP). pub pairs_selected: u64, - /// The nonparticipant corpus row count `Q`. + /// The nonparticipant corpus row count Q, or zero when no control census completed. pub control_candidates: u64, /// The drawn control count `m = min(Q, n)`. pub controls_selected: u64, @@ -69,9 +75,9 @@ pub(crate) struct PairedMovementEvidence { /// What one paired-movement readout resolved to. /// -/// The variants are structural. Aggregates exist only inside [`Self::Measured`], so a vacuous -/// or failed readout cannot carry a partial family, and the tag alone tells a reader the whole -/// shape. +/// Aggregate fields exist only inside [`Self::Measured`]. A vacuous or failed value cannot carry a +/// partial family. Construction and deserialization do not validate a measured value's counts or +/// stratum contents. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(rename_all = "kebab-case", tag = "outcome")] pub(crate) enum MovementOutcome { @@ -81,9 +87,8 @@ pub(crate) enum MovementOutcome { pairs: PairAggregates, /// The collateral strata over the drawn controls. /// - /// A nonempty candidate population yields every stratum, individually empty when the - /// draw is thin. No strata at all under the `Q = 0` reading, which keeps every - /// control count zero and every control value field absent. + /// The producer emits ten strata for a nonempty control population, with individually + /// empty strata allowed. At Q = 0 it emits no strata. deciles: Vec, }, /// The pair domain was empty (`P = 0`). @@ -104,9 +109,8 @@ pub(crate) enum MovementOutcome { /// The aggregate families over the drawn pairs. /// -/// Each difference forms per pair from one [`PairMovement`]'s own fields. A negative distance -/// change contracts, and a negative rank change improves rank, so the fractions read the share -/// of pairs the canonical step moved toward their partners. +/// Each difference forms per pair from one [`PairMovement`]'s own fields. Negative distance changes +/// indicate contraction, and negative rank changes indicate rank improvement. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct PairAggregates { /// The population count `n`, every drawn pair. @@ -124,9 +128,10 @@ pub(crate) struct PairAggregates { impl PairAggregates { /// Aggregates the drawn pairs' readings, in draw order. /// - /// Every difference forms directly from a reading's own fields in `f64`, never by - /// subtracting persisted step aggregates. Means commute with subtraction while fractions - /// and quantiles do not, so the subtraction shortcut fabricates readings no pair produced. + /// Every difference forms directly from a reading's fields, never by subtracting step + /// aggregates. Distance differences use `f64`, and differences of the `u32` ranks are exactly + /// representable in `f64`. The difference populations must meet [`MovementAggregate::over`]'s + /// finite-partial-sum requirement. /// /// # Panics /// @@ -147,9 +152,10 @@ impl PairAggregates { let rank = DFinite::from(i64::from(reading.rank_canonical) - i64::from(reading.rank_zero)); - // The strict-less is a plain numeric comparison, because a subtraction never - // produces `-0.0` under round-to-nearest, so the total order's `-0.0 < +0.0` case - // is unreachable here. + // Non-negative operands have canonical positive zero. Their difference cannot underflow + // to a negative zero under round-to-nearest, and equal operands give positive zero. + // Therefore the total-order comparison with zero agrees with numeric strict negativity + // for these differences. if distance < DFinite::ZERO { contracted += 1; } @@ -178,9 +184,10 @@ impl PairAggregates { /// One reading family's nearest-rank quantiles and mean. /// -/// The value fields of one aggregate family. The population count lives beside the family, on -/// [`PairAggregates::count`] for the pair families and on [`ControlDecile::selected`] for a -/// stratum's displacement family, and the family exists exactly when that count is positive. +/// For produced pair and control evidence, the associated count lives on [`PairAggregates::count`] +/// or [`ControlDecile::selected`]. Those producers create a family exactly when the count is +/// positive. This type carries no count and validates no cross-field relationships on +/// deserialization. #[derive(Debug, Copy, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct MovementAggregate { /// The nearest-rank reading at fraction 0.05. @@ -200,10 +207,14 @@ pub(crate) struct MovementAggregate { impl MovementAggregate { /// Aggregates one reading family, in draw order. /// - /// The mean folds the readings serially in the given order. The quantiles sort a copy - /// ascending. The draw order breaks reading ties by stable identity, which keeps the sort - /// total without ever moving a value across a rank, so the ascending value sequence alone - /// reproduces every persisted quantile and the identities stay out of the aggregation. + /// The mean folds in the supplied order. Every running partial sum must remain finite. + /// Individual [`DFinite`] values alone do not establish this requirement. The quantiles sort a + /// copy using [`DFinite`]'s total order, which distinguishes zero signs. Equal values need no + /// identity tie-break because exchanging them leaves each quantile unchanged. + /// + /// # Complexity + /// + /// O(n log n) work and O(n) copied values for n readings. /// /// # Panics /// @@ -214,17 +225,16 @@ impl MovementAggregate { panic!("an aggregate family exists only for a positive population"); }; - // One fixed serial fold in draw order. `Iterator::sum` happens to fold in order too, - // but the loop states the contract rather than inheriting it. The fold keeps the typed - // escape op by its totality theorem. Every reading is a frame distance difference or a - // rank difference, bounded below 2¹³¹, and the population is bounded by the corpus - // rows, below 2³². The serial sum therefore stays below 2¹⁶³, far inside `f64`. + // Finite f32 frame coordinates bound wide distance differences below 2¹³¹ in magnitude. + // With fewer than 2⁶⁴ readings on supported targets, those populations keep even the + // absolute sum below 2¹⁹⁵, far inside f64. Therefore the paired-frame readout can use the + // finite sum directly. Other inputs must meet the documented partial-sum requirement. let mut sum = DFinite::ZERO; for &reading in readings { sum += reading; } - // Finite with no check: dividing the bounded sum by a count of at least one shrinks it. + // dividing a finite sum by a count of at least one preserves finiteness. let mean = (sum / DPositive::from_usize(len)).finish_unchecked(); let mut sorted = readings.to_vec(); @@ -243,21 +253,23 @@ impl MovementAggregate { /// One collateral stratum of the control readout. /// -/// The strata partition the anchor-distance axis at each tenth of the candidate population, so -/// the boundaries are a function of the census rather than the draw. The candidate count reads -/// the stratum's share of the census, and the selected count reads the stratum's share of the -/// draw. +/// The strata partition distances to the sampled pair endpoints at each tenth of the full +/// nonparticipant population. Their boundaries depend on the pair sample, but not on the selected +/// control rows. Counts record each stratum's share of the census and the control sample. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct ControlDecile { - /// The stratum's upper anchor-distance boundary, the nearest-rank reading at its tenth of - /// the candidate population. + /// The stratum's inclusive upper anchor-distance boundary. + /// + /// The nearest-rank reading at its tenth of the candidate population. pub upper: DNonNegative, /// Candidate rows this stratum holds. pub candidates: u64, /// Drawn rows this stratum holds. pub selected: u64, - /// The displacement family over the stratum's drawn rows, present exactly when `selected` - /// is positive. + /// The displacement family, produced exactly when `selected` is positive. + /// + /// Absence serializes as `null`. Deserialization does not check its relationship to + /// `selected`. pub displacement: Option, } @@ -265,19 +277,21 @@ impl ControlDecile { /// Builds the ten collateral strata. /// /// `candidates` holds every nonparticipant row's zero-step nearest-anchor distance and is - /// sorted in place. `readings` holds the drawn controls' readings in draw order, and each - /// drawn reading re-derives through the one metric, so it is one of the candidate readings. - /// A reading joins the first stratum whose upper boundary reaches it. Equal readings - /// therefore share a stratum, and a boundary tie leaves the later stratum without - /// candidates. + /// sorted in place. `readings` must hold sampled controls from that population in draw order, + /// using the same anchor positions and distance metric. Displacements must meet + /// [`MovementAggregate::over`]'s partial-sum requirement. Membership and uniqueness are not + /// checked. + /// + /// A reading joins the first stratum whose upper boundary reaches it. Equal readings share a + /// stratum, and repeated boundaries leave later strata without candidates. /// /// Returns no strata when the candidate population is empty. That is the `Q = 0` /// reading: every control count stays zero and every control value field stays absent. /// /// # Panics /// - /// This panics when readings arrive while the candidate population is empty, or when a - /// reading exceeds the census maximum. Both contradict the draw's own construction. + /// Panics when `readings` is nonempty but `candidates` is empty, or when a reading's anchor + /// distance exceeds the candidate maximum. pub(super) fn over(candidates: &mut [DNonNegative], readings: &[ControlMovement]) -> Vec { if candidates.is_empty() { assert!( @@ -295,8 +309,8 @@ impl ControlDecile { for tenth in 1..=DECILES { let upper = nearest_rank(candidates, f64::from(tenth) / 10.0); - // Candidates at or below the boundary, cumulatively. The tenth boundary is the - // population maximum, so the final stratum absorbs the remainder. + // count candidates cumulatively through each boundary, with the tenth at the population + // maximum. let cumulative = candidates.partition_point(|&reading| reading <= upper); uppers.push(upper); census.push((cumulative - below) as u64); @@ -329,10 +343,9 @@ impl ControlDecile { /// The typed refusal a failed readout retains. /// -/// Each variant mirrors its producer field for field, so the persisted reason names exactly -/// what refused. [`CensusError`] supplies the index contradictions and [`MovementError`] the -/// row-count contradiction; a non-finite frame never reaches the readout, because the frames -/// arrive as proven-finite fields. +/// Each variant retains every field of its producer. [`CensusError`] supplies index contradictions +/// and [`MovementError`] supplies frame-length disagreement. Finiteness belongs to the input field +/// contract, with no failure variant here. #[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] #[serde(rename_all = "kebab-case", tag = "cause")] pub(crate) enum FailureReason { @@ -397,14 +410,13 @@ impl From for FailureReason { /// Reads the nearest-rank quantile at `fraction` over ascending readings. /// -/// The reading is the first whose cumulative unit count reaches `fraction` of the population. -/// With population `N`, that is the reading at one-based rank `⌈fraction · N⌉`, evaluated in -/// `f64` so every replay computes the same rank. +/// `sorted` must be ascending and nonempty, and `fraction` must lie in (0,1]. The one-based rank is +/// ceil(fraction · N), with both the conversion of N and the product evaluated in `f64`. The +/// rounded product can select a different rank than exact rational arithmetic at a boundary. /// /// # Panics /// -/// This panics when `sorted` is empty. An aggregate family exists only for a positive -/// population. +/// Panics when `sorted` is empty or the computed rank is zero or exceeds its length. fn nearest_rank(sorted: &[T], fraction: f64) -> T { debug_assert!( fraction > 0.0 && fraction <= 1.0, diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/tests.rs index cfbee86886f..e88036f454a 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/evidence/tests.rs @@ -1,9 +1,7 @@ //! Evidence aggregation and wire-shape expectations. //! -//! The quantile oracle restates the cumulative quantile rule directly, and the aggregate pins -//! derive by hand from exact decimal literals. The serialized equalities pin each outcome -//! kind's whole wire shape, so a drifted field name or a defaulted absence fails against an -//! independent statement of the contract. +//! A cumulative-count oracle checks quantiles. Hand-derived binary-representable fixtures check +//! aggregate values, and complete serialized values check each outcome's wire shape. #![expect( clippy::float_cmp, @@ -36,8 +34,14 @@ use crate::{ #[test] fn nearest_rank_quantiles_restate_the_cumulative_rule() { - // The oracle restates the definition directly: walk the ascending readings and take - // the first whose cumulative unit count reaches the fraction of the population. + /// Computes the nearest-rank quantile by definition. + /// + /// Walks the ascending readings and returns the first whose cumulative unit count reaches + /// `fraction` of the population. + /// + /// # Panics + /// + /// Panics if `readings` is empty or no cumulative count reaches `fraction` of the population. fn oracle(readings: &[f64], fraction: f64) -> f64 { let mut sorted = readings.to_vec(); sorted.sort_unstable_by(f64::total_cmp); @@ -109,9 +113,10 @@ fn pair_aggregates_pin_exact_decimal_literals() { let aggregates = PairAggregates::over(&readings); - // Δd = [-0.5, 0.25, -1.5, 0.0] sorts to [-1.5, -0.5, 0.0, 0.25]. Over four readings the - // one-based nearest ranks are ⌈0.2⌉ = 1, ⌈1⌉ = 1, ⌈2⌉ = 2, ⌈3⌉ = 3, and ⌈3.8⌉ = 4. Every - // literal is exact in f64, so the serial fold gives (-0.5 + 0.25 - 1.5 + 0.0) / 4 exactly. + // Δd = [−0.5, 0.25, −1.5, 0.0] sorts to [−1.5, −0.5, 0.0, 0.25]. Over four readings the + // one-based nearest ranks are ⌈0.2⌉ = 1, ⌈1⌉ = 1, ⌈2⌉ = 2, ⌈3⌉ = 3 and ⌈3.8⌉ = 4. The Δd + // readings and every partial sum are exactly representable in f64. Their serial mean is (−0.5 + + // 0.25 − 1.5 + 0.0) / 4 = −0.4375. assert_eq!(aggregates.count, 4); assert_eq!(aggregates.distance.q05, d_finite!(-1.5)); assert_eq!(aggregates.distance.q25, d_finite!(-1.5)); @@ -133,8 +138,9 @@ fn pair_aggregates_pin_exact_decimal_literals() { #[test] fn the_mean_folds_serially_in_draw_order() { - // 2^54 absorbs a unit exactly, so the fold order decides the sum. Draw order absorbs the - // first unit and keeps the second, while an ascending fold would absorb both and read zero. + // At magnitude 2⁵⁴, adding one rounds back to the same value. Draw order loses the first unit, + // cancels the large values and retains the final unit, giving mean 0.25. An ascending fold + // loses both units and gives zero. let big = 18_014_398_509_481_984.0_f64; assert_eq!(big, (2.0_f64).powi(54), "the literal is 2^54"); @@ -168,9 +174,8 @@ fn pair_first_quantiles_defeat_aggregate_subtraction() { let aggregates = PairAggregates::over(&readings); - // Δd = [-9, 10, -18] has median -9. The step medians are 20 and 12, whose difference -8 - // is a reading no pair produced, so an implementation that subtracts persisted step - // aggregates cannot reproduce the family. + // Δd = [−9, 10, −18] has median −9. The step medians are 20 and 12, whose difference is −8. + // Therefore subtracting the step medians cannot reproduce the pair-difference median. let zero_median = MovementAggregate::over(&[d_finite!(10.0), d_finite!(20.0), d_finite!(30.0)]).q50; let canonical_median = diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/fixtures.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/fixtures.rs index 316c17fc2a5..017df7eef63 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/fixtures.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/fixtures.rs @@ -1,4 +1,4 @@ -//! Shared inputs of the paired-movement acceptance tests. +//! Common metadata and geometry fixtures for paired-movement tests. //! //! The identity pins freeze the exact bytes [`snapshot`] and [`reproducibility`] serialize //! into, and the writer pins record the salt they derive, so every sibling's tests must agree @@ -31,7 +31,9 @@ pub(super) fn digest(seed: &str) -> Sha256Digest { hasher.finalize() } -/// The frozen configuration half of the preimage inputs. +/// Builds a seeded fit configuration with a 512-landmark capacity. +/// +/// Unspecified fields use the current defaults and contribute to the pinned preimage bytes. pub(super) fn config() -> FitConfig { FitConfig { seed: 0xC2, @@ -45,7 +47,7 @@ pub(super) fn config() -> FitConfig { } } -/// The frozen snapshot half of the preimage inputs. +/// Builds snapshot metadata for 1,000 nodes and 4,000 edges. pub(super) fn snapshot() -> Snapshot { Snapshot { axes: None, @@ -55,7 +57,7 @@ pub(super) fn snapshot() -> Snapshot { } } -/// The frozen reproducibility half of the preimage inputs. +/// Builds the configuration and embedder echo with no prior generation. pub(super) fn reproducibility() -> Reproducibility { Reproducibility { config: config(), @@ -64,14 +66,18 @@ pub(super) fn reproducibility() -> Reproducibility { } } -/// The recognized initial draw rule. +/// Returns the recognized initial draw rule. pub(super) fn rule() -> DrawRule { RuleIdentity::INITIAL .recognize() .expect("the crate carries its own initial identity") } -/// The salt the frozen inputs derive. +/// Derives the fixture metadata's draw salt. +/// +/// # Panics +/// +/// Panics if the fixture metadata does not serialize. pub(super) fn salt() -> DrawSalt { rule() .derive_salt(&snapshot(), &reproducibility()) @@ -83,7 +89,11 @@ pub(super) fn node(value: u64) -> NodeRowId { NodeRowId::new(value) } -/// Builds one attraction group record with the given Proximal class weight. +/// Builds a group with the supplied Proximal weight and fixed Coincident weight and strength. +/// +/// # Panics +/// +/// Panics when `proximal` is negative or non-finite. pub(super) fn group(relation: u64, edge_offset: u64, proximal: f32) -> GroupRecord { let weight = |value: f32| NonNegative::new(value).expect("the fixture weights are in domain"); GroupRecord::new( @@ -107,7 +117,9 @@ pub(super) fn edge(source: u64, target: u64) -> EdgeRecord ) } -/// Views a finite point slice as a proven corpus-row frame. +/// Interprets fixture points as a corpus-row frame. +/// +/// Every coordinate must be finite. pub(super) fn frame(points: &[Vec2]) -> &FinitePointField { FinitePointField::new_unchecked(IdSlice::from_raw(points)) } diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/identity/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/identity/mod.rs index b1eaed64998..d72a06aacc8 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/identity/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/identity/mod.rs @@ -1,20 +1,17 @@ -//! The draw-rule identity and salt. +//! Versioned derivation of the paired-movement draw salt and subject order. //! -//! The paired-movement sample is replayable without persisting pair identities. The draw is a -//! pure function of the generation's declared inputs under a rule that fixes every convention -//! behind it. [`RuleIdentity`] names one such convention set, and the evidence body records -//! which one produced its draw. A replay recognizes the recorded identity -//! ([`RuleIdentity::recognize`]), re-derives the salt under that identity's retained conventions, -//! and compares it with the recorded value. An unknown identity or a salt mismatch invalidates the -//! replay and nothing else: the readout is evidence, and the generation it came from stays -//! published. +//! [`RuleIdentity`] names the conventions needed to replay a draw without persisting the selected +//! pair identities. Recognize the recorded identity with [`RuleIdentity::recognize`], re-derive the +//! salt and compare it with the recorded value before sampling the same attraction index. An +//! unknown identity or a salt mismatch prevents this replay check from establishing agreement. +//! These operations do not change a published generation. //! -//! The salt preimage is the ordered two-field projection of the generation's metadata document, -//! `snapshot` and then `reproducibility`. Both sections exist before ladder evaluation, and no -//! ladder output can enter them. Byte-identical inputs therefore share one draw, and any changed -//! input (snapshot, configuration, embedder, or prior-generation chain) rotates the -//! pseudo-random sample. Placement and evidence stay outside the preimage, so the projection is -//! an input identity rather than a full-content identity. +//! The salt preimage serializes the metadata's `snapshot` and then `reproducibility`, excluding +//! placement outputs and evidence. Equal serialized projections yield equal salts. The projection +//! contains declared inputs, not the attraction-index bytes or a complete corpus identity. +//! Replaying the draw also requires the same candidate domain and row identities. Changed +//! projections normally change the digest, subject to SHA-256 collisions, and even different salts +//! can select the same sample. #[cfg(test)] mod tests; @@ -32,9 +29,9 @@ use crate::{ /// One generation's derived draw salt. /// -/// A keyed digest of the generation's input identity (the encoded salt preimage) under the -/// rule's salt domain tag. The evidence body records it beside the rule identity, and a replay -/// re-derives and compares it before selecting any identity. +/// SHA-256 of a public domain tag followed by the encoded metadata projection. The evidence records +/// it beside [`RuleIdentity`] for replay comparison. This derivation uses no secret key and +/// supplies no authentication. #[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] #[serde(transparent)] #[repr(transparent)] @@ -43,14 +40,14 @@ pub(crate) struct DrawSalt(Sha256Digest); impl DrawSalt { /// The identity-1 salt domain tag. /// - /// Both tags terminate with a newline neither tag contains, so no tag is a prefix of the other - /// and the two derivation families cannot collide. Every hash input this module forms places - /// its single variable-length component last, after fixed-width components only, so an - /// input parses uniquely (see [`Sha256`] on framing). + /// Salt and order tags are distinct and end at their first newline. Neither tag is a prefix of + /// the other. Every input places its single variable-length component after the fixed-width + /// components. This framing separates the derivations' preimages, without ruling out hash + /// collisions. const DOMAIN: &[u8] = b"atlas.paired-movement.salt.1\n"; } -/// The keyed order key of one draw subject. +/// A salted digest ordering one draw subject. /// /// Keys order bytewise on the derived digest, the sampler's primary sort key. A key is never /// persisted: the salt and the subject encoding re-derive it. @@ -58,22 +55,21 @@ impl DrawSalt { pub(super) struct OrderKey(Sha256Digest); impl OrderKey { - /// The identity-1 order domain tag; the shared framing contract lives on - /// [`DrawSalt::DOMAIN`]. + /// The identity-1 order domain tag. + /// + /// [`DrawSalt::DOMAIN`] states the shared framing contract. const DOMAIN: &[u8] = b"atlas.paired-movement.order.1\n"; } /// Identity of one paired-movement draw rule. /// -/// An identity fixes the whole draw convention: the keyed-hash algorithm, its domain tags, the -/// stable-identity encodings, the ordering rule, and the exact preimage serializer with its -/// formatting options. Changing any of them, the serializer's dependency semantics included, -/// requires a new identity, and an existing identity's conventions never move, so a recorded -/// draw stays re-derivable byte for byte. The identities are replay semantics rather than -/// configuration and are independent of the repository version. +/// An identity fixes the hash algorithm and domain tags, subject encodings, ordering rule and exact +/// preimage serialization. Changes to any of these conventions require a new identity. An existing +/// identity's encoding of the same inputs must never change. This includes the metadata schema and +/// serializer dependency semantics, which this type alone cannot freeze. /// -/// The wire form is a bare integer, and deserialization is lossless for identities this crate -/// does not carry, so a reader reports exactly what it refused. +/// The wire form is a bare `u32`. Deserialization preserves unknown identities exactly, allowing a +/// reader to report the refused value. #[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] #[serde(transparent)] #[repr(transparent)] @@ -86,11 +82,10 @@ impl RuleIdentity { /// the preimage serializer, and bytewise digest order for the keys. pub(crate) const INITIAL: Self = Self(1); - /// Returns this identity's operations, or `None` for an identity this crate does not carry. + /// Returns this identity's operations, or [`None`] for an unknown identity. /// - /// Recognition is the one dispatch: every [`DrawRule`] operation afterwards is total over - /// its inputs. `None` invalidates the replay that consumed the recorded identity, and - /// nothing else, because the generation it came from is already published. + /// Recognition selects supported conventions. Serialization through the returned [`DrawRule`] + /// remains fallible. #[must_use] pub(crate) const fn recognize(self) -> Option { match self { @@ -100,11 +95,10 @@ impl RuleIdentity { } } -/// The identity-1 salt preimage, the ordered two-field projection of the metadata document. +/// The identity-1 projection of snapshot and reproducibility metadata. /// -/// Declaration order is serialization order: `snapshot`, then `reproducibility`. The fields -/// borrow [`SaltMetadata`]'s own types, so the projection serializes through exactly the -/// production document's `serde` paths, the validating config echo included. +/// Serialization emits `snapshot`, then `reproducibility`, through [`SaltMetadata`]'s field types +/// and their serialization implementations. /// /// [`SaltMetadata`]: crate::file::salt::metadata::SaltMetadata #[derive(serde::Serialize)] @@ -113,11 +107,9 @@ struct SaltPreimage<'document> { reproducibility: &'document Reproducibility, } -/// The salt preimage did not serialize. +/// A salt-preimage serialization or output-write failure. /// -/// Unreachable for this crate's document types: the projection is a strict subset of the -/// document the production writer serializes on every publish. The seam stays typed rather -/// than panicking, matching the seal path's posture. +/// The source retains the underlying [`serde_json::Error`]. #[derive(Debug)] pub(crate) struct EncodeError(serde_json::Error); @@ -135,10 +127,9 @@ impl Error for EncodeError { /// The operations of one recognized draw rule. /// -/// [`RuleIdentity::recognize`] is the only constructor, so a value always names a convention -/// set this crate implements. The bodies below are identity 1's. A later identity extends the -/// recognition dispatch and retains these conventions unchanged under -/// [`RuleIdentity::INITIAL`], so replay under the earlier identity keeps its exact bytes. +/// [`RuleIdentity::recognize`] constructs only supported identities. These operations implement +/// [`RuleIdentity::INITIAL`]. Adding an identity must retain the existing identity's exact +/// derivation for its recorded inputs. #[derive(Debug, Copy, Clone)] pub(crate) struct DrawRule { /// The identity these operations implement. @@ -147,8 +138,6 @@ pub(crate) struct DrawRule { impl DrawRule { /// Returns the identity these operations implement. - /// - /// The evidence body records it beside the derived salt, so a replay dispatches back here. #[must_use] pub(super) const fn identity(self) -> RuleIdentity { self.identity @@ -156,17 +145,14 @@ impl DrawRule { /// Writes the salt preimage, the ordered two-field projection of the metadata document. /// - /// Identity 1 serializes `snapshot` and then `reproducibility` into `write` with - /// [`serde_json::to_writer_pretty`] through the document types' own `serde` implementations: - /// the same dependency, entry point, and pretty-format semantics as the production - /// repository document writer ([`StagedGeneration::seal`]), because they are one crate. - /// Equal projections yield equal bytes even when document content outside them differs. + /// Identity 1 writes `snapshot` and then `reproducibility` using + /// [`serde_json::to_writer_pretty`] through the metadata types' own serialization + /// implementations. Equal projections yield equal bytes even when other document content + /// differs. /// /// # Errors /// - /// [`EncodeError`] when the projection does not serialize. - /// - /// [`StagedGeneration::seal`]: crate::file::generation::StagedGeneration::seal + /// Returns [`EncodeError`] for serialization or output-write failure. #[expect( clippy::unused_self, reason = "the receiver is the recognition proof: only `RuleIdentity::recognize` mints it, \ @@ -190,9 +176,9 @@ impl DrawRule { /// Derives the draw salt of one generation. /// - /// The salt is the SHA-256 of the salt domain tag followed by the written preimage - /// ([`Self::write_preimage`]). It keys the order under which the sampler draws candidate - /// pairs: byte-identical inputs share it, and any changed input rotates it. + /// Computes SHA-256 of the salt domain tag followed by [`Self::write_preimage`]'s bytes. + /// Byte-identical projections share a salt. Digest equality alone does not prove equal inputs + /// or an equal attraction index. /// /// # Errors /// @@ -217,7 +203,7 @@ impl DrawRule { Ok(DrawSalt(hasher.finalize())) } - /// Returns the keyed order key of one draw subject. + /// Hashes a draw subject under the order tag and public salt. /// /// The key is the SHA-256 of the order domain tag, the salt's raw digest bytes, and then /// `subject`, the draw subject's encoded stable identity. The subject encoding is part of @@ -237,12 +223,11 @@ impl DrawRule { OrderKey(hasher.finalize()) } - /// Returns the keyed order key of one candidate pair ([`Self::order_key`]). + /// Returns the salted order key of an oriented pair. /// - /// Identity 1 encodes the pair subject as 16 bytes, the source row and then the target row, - /// each in the row id's persisted little-endian form. Orientation is kept: a pair and its - /// reverse are distinct subjects. Both components are fixed width, so the encoding frames - /// itself. + /// Identity 1 encodes the source row followed by the target row, each as an 8-byte + /// little-endian integer. Fixed-width components make the pair encoding unambiguous. For + /// distinct endpoints, a pair and its reverse have different preimages. #[must_use] pub(super) fn pair_order_key( self, @@ -256,10 +241,10 @@ impl DrawRule { self.order_key(salt, &subject) } - /// Returns the keyed order key of one candidate control row ([`Self::order_key`]). + /// Returns the salted order key of a control row. /// - /// Identity 1 encodes the row subject as the row id's persisted little-endian form: 8 bytes - /// where a pair subject holds 16, so no row shares a preimage with a pair under one salt. + /// Identity 1 encodes the row as an 8-byte little-endian integer. The distinct lengths of row + /// and pair subjects keep their preimages separate under one salt. #[must_use] pub(super) fn row_order_key(self, salt: DrawSalt, row: NodeRowId) -> OrderKey { self.order_key(salt, row.as_bytes()) diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/identity/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/identity/tests.rs index d081c33ce7b..d3ef7653060 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/identity/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/identity/tests.rs @@ -1,9 +1,8 @@ //! Draw-rule identity and salt expectations. //! -//! The formula oracles restate the identity-1 derivations (a domain tag, then fixed-width -//! components, then the variable-length tail) as one explicit concatenation hashed through -//! [`Sha256Digest::of`], so a drifted update order, a dropped tag, or a swapped domain fails -//! against an independent statement of the contract. +//! The formula oracles concatenate the domain tag, fixed-width components and variable-length tail +//! before hashing with [`Sha256Digest::of`]. This compares the incremental derivation with an +//! independently assembled byte sequence. use core::str; @@ -19,7 +18,11 @@ use crate::{ }, }; -/// Writes the fixture inputs' preimage bytes. +/// Serializes the fixture inputs' preimage bytes. +/// +/// # Panics +/// +/// Panics if the fixture projection does not serialize. fn preimage() -> Vec { let mut preimage = Vec::new(); rule() @@ -80,6 +83,7 @@ fn preimage_is_the_ordered_two_field_projection() { #[test] fn preimage_round_trips_through_the_document_serde_paths() { + /// The `snapshot` and `reproducibility` metadata documents a preimage decodes to. #[derive(serde::Deserialize)] struct Projection { snapshot: crate::file::salt::metadata::Snapshot, @@ -93,18 +97,10 @@ fn preimage_round_trips_through_the_document_serde_paths() { assert_eq!(decoded.reproducibility, reproducibility()); } -/// Pins identity 1's preimage bytes for one fixed input, by length, digest, and derived salt. +/// Pins the fixture's preimage length, digest and derived salt. /// -/// The rule identity owns its preimage encoder, and a recorded identity's conventions never -/// move, so every published document's recorded salt stays re-derivable byte for byte. A red -/// run here means the encoder's output moved, and re-pinning alone is never the repair. When a -/// fixture input changed value (a `FitConfig` default, say, or a new serialized field), show -/// that value change and re-pin, because the pin freezes one input's bytes rather than the -/// input itself. When the inputs stand and the serializer or its formatting moved, identity 1 -/// no longer replays older documents, so the change mints identity 2 for new draws while -/// identity 1 keeps its exact derivation. Once a second identity exists, this test is the -/// control that keeps the earlier identity's bytes unchanged while the current encoder -/// changes. +/// The fixture includes [`FitConfig`](crate::salt::fit::FitConfig) defaults in its serialized +/// inputs. #[test] fn identity_one_preimage_bytes_stay_frozen() { let preimage = preimage(); @@ -203,6 +199,9 @@ fn the_order_key_is_the_tagged_digest_of_salt_then_subject() { assert_eq!(key.0, expected); } +/// Compares equal inputs and distinct fixed subject/salt combinations. +/// +/// The unequal fixtures check these encodings, not a general absence of SHA-256 collisions. #[test] fn order_keys_separate_subjects_and_salts() { let salt = salt(); diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/measure/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/measure/mod.rs index 3207b03de82..5743e376a63 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/measure/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/measure/mod.rs @@ -1,11 +1,8 @@ -//! The whole readout of one generation, from salt to evidence body. +//! Draw selection, movement measurement and assembly of paired evidence. //! -//! [`measure`] is the C2 evidence writer's core. It derives the draw salt and takes the census -//! over the attraction index's regions, then reads every drawn subject between the aligned -//! step frames and assembles the persisted evidence body. It is a pure function of its inputs, so -//! the acceptance fixtures drive the exact production path over constructed index regions and -//! frames, injected failures included, while the fit's writer wraps it around the staged -//! artifacts. +//! [`measure`] derives the salt, takes the attraction-index census and reads the selected subjects +//! between aligned frames. Per-subject work runs in parallel, while collected readings retain draw +//! order for the serial aggregation. #[cfg(test)] mod tests; @@ -35,16 +32,24 @@ use crate::{ /// Measures the paired-movement readout of one generation. /// -/// `groups` and `edges` are the attraction index's regions in file order, and `zero` and -/// `canonical` are the ladder's aligned step frames. The zero frame's row count is the corpus -/// row domain the census walks, and the fit's writer asserts it against the staged index. The -/// salt derives under the initial rule identity from the same `snapshot` and `reproducibility` -/// values the seal serializes, so the draw replays from the published document's input -/// sections alone. +/// `groups` and `edges` must satisfy the attraction-index contract and use the same row identities +/// as `zero` and `canonical`. Both frames must already share the baseline basis. The census uses +/// `zero.len()` as its row domain. The metadata supplies the salt preimage but is not checked +/// against the regions or frames. /// -/// A census or movement refusal lands as [`MovementOutcome::Failed`] beside the completed draw -/// counts, and an empty pair domain as [`MovementOutcome::Vacuous`]. Every readout resolution -/// is an evidence body, so the readout never blocks publication. +/// A census refusal returns [`MovementOutcome::Failed`] with zero draw counts. A frame-length +/// refusal preserves the completed draw counts in a failed outcome. An empty pair domain returns +/// [`MovementOutcome::Vacuous`] before checking the canonical frame's length. These outcomes carry +/// no publication veto, while salt-encoding failure remains a returned error. +/// +/// # Complexity +/// +/// Beyond [`Draw::over`], this builds indexes for both N-row frames and the sampled endpoints. It +/// reads every sampled pair and control, then queries and sorts all Q nonparticipants' anchor +/// distances to define strata. This full candidate sweep costs O(Q) retained distances and O(Q log +/// Q) sorting work even for a small control sample. Distance ties can make individual tree queries +/// require full-domain sorting. Per-worker scratch is additional to the indexes and collected +/// readings. /// /// # Errors /// @@ -104,10 +109,10 @@ pub(crate) fn measure( } }; - // The reading sweeps are parallel over read-only frames and trees. Each worker allocates - // its readouts in a scratch arena and resets it between readings, so a reading bump-allocates - // into warm memory and frees in bulk. Collection preserves draw order, so the readings are - // the serial loop's, whatever the schedule. + // Indexed parallel collection preserves the draw's order. Each reading depends only on its + // subject and the shared frames and trees. Therefore the serial aggregate receives the same + // ordered readings independently of the worker schedule. Arena reset reclaims each worker's + // result buffers between readings. let pairs: Vec<_> = draw .pairs() .par_iter() @@ -118,8 +123,8 @@ pub(crate) fn measure( }) .collect(); - // The anchor index holds the drawn pairs' endpoints at their zero-step positions. A gather - // from the proven zero field stays proven. + // the nonempty draw supplies at least one in-domain endpoint. Gathering those zero-step + // positions preserves finiteness for the anchor index. let anchor_rows = draw.anchors(); let anchor_frame = zero.gather(IdSlice::::from_raw(&anchor_rows)); let anchor_tree = KdTree::build(&anchor_frame); @@ -134,9 +139,8 @@ pub(crate) fn measure( }) .collect(); - // The collateral strata boundaries stand on the full candidate population, so the sweep - // reads every nonparticipant row's anchor distance through the same readout as the drawn - // controls' readings. + // define boundaries from every control candidate's proximity to the sampled anchors. Sorting + // later makes the parallel collection order irrelevant. let participants = participants(rows, edges); let mut candidates: Vec<_> = (0..rows) .into_par_iter() diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/measure/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/measure/tests.rs index 83fc4906ccd..a70e64344c0 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/measure/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/measure/tests.rs @@ -1,8 +1,7 @@ -//! Production-writer expectations, salt to serialized evidence body. +//! Complete paired-evidence values from constructed index regions and aligned frames. //! -//! Each pin runs [`measure`] over constructed index regions and frames and asserts the whole -//! serialized body at once, so the writer's counts, outcome dispatch, and aggregate wiring are -//! checked on the exact path the fit's writer drives. +//! Each case checks [`measure`]'s serialized result, including draw counts, outcome dispatch and +//! aggregate fields. use serde_json::json; @@ -17,12 +16,12 @@ use crate::{ }, }; -/// Builds the writer fixture's attraction regions, four Proximal pairs over ten corpus rows. +/// Builds attraction regions with four Proximal pairs over ten rows. /// -/// Edges `(0,1)`, `(2,3)`, `(4,5)`, and `(6,7)` sit in one force-bearing group, so the pair -/// domain is the four oriented pairs, the participant set is rows `0..=7`, and rows 8 and 9 -/// are the control candidates. The whole domain sits far under both draw bounds, so every -/// candidate is drawn and no pinned aggregate depends on the salt-keyed order. +/// Edges `(0,1)`, `(2,3)`, `(4,5)` and `(6,7)` form the pair domain in one force-bearing group. The +/// participant set is rows `0..=7`, leaving rows 8 and 9 as the control candidates. Both domains +/// fit within their quotas and every candidate is drawn. The fixture's bounded integer readings sum +/// exactly in any draw order. fn readout_index() -> (Vec, Vec>) { ( vec![group(3, 0, 1.0)], @@ -30,7 +29,7 @@ fn readout_index() -> (Vec, Vec>) ) } -/// The writer fixture's aligned frames, exact in every drawn reading. +/// Builds aligned frames with integer pair distances and control displacements. /// /// The pair clusters sit far apart, and each partner is its source's nearest row at both steps /// except one designed movement: row 4 moves ahead of partner 3 at the canonical step, where @@ -74,12 +73,10 @@ fn readout_frames() -> (Vec, Vec) { (zero, canonical) } -/// The `P = 0` reading through the production writer: a present vacuous body. +/// Records an empty pair population as a present vacuous outcome. /// -/// The C1 `--vacuous-placement` fixture cannot stand in for this one. A vacuous placement -/// publishes no ladder record at all, while this generation measures its ladder and records a -/// present body whose outcome is vacuous, with the recognized rule and derived salt beside -/// zero counts on both domains. +/// This measures a ladder with no Proximal pairs. The outcome keeps the recognized rule and derived +/// salt beside zero counts, distinguishing it from absent paired evidence. #[test] fn the_writer_reads_an_empty_pair_domain_as_a_present_vacuous_body() { let (_, edges) = readout_index(); @@ -111,13 +108,10 @@ fn the_writer_reads_an_empty_pair_domain_as_a_present_vacuous_body() { ); } -/// An injected typed post-census movement refusal through the production writer. +/// Retains completed draw counts when the canonical frame is one row short. /// -/// The canonical frame arrives one row short, so the census completes both domains and the -/// movement readout refuses the frame pair. A non-finite frame cannot enter here at all: the -/// frames arrive as proven-finite fields. The serialized equality pins the whole body: the -/// completed draw counts persist, the reason names both row counts, and no partial aggregate -/// key exists beside them. +/// The census completes before the frame-length check. The failed outcome names both row counts and +/// carries no partial aggregates. #[test] fn an_injected_movement_refusal_keeps_its_counts_and_no_partial_aggregates() { let (groups, edges) = readout_index(); @@ -154,7 +148,7 @@ fn an_injected_movement_refusal_keeps_its_counts_and_no_partial_aggregates() { ); } -/// The aggregation fixture on the production path: exact decimal literals, pinned bytes. +/// Repeats the complete readout over binary-representable fixture values. /// /// The readout computes twice with byte-identical serialized output. The pinned aggregate /// derives by hand from [`readout_frames`]'s table: distance differences `{-6, +4, -1, 0}` @@ -248,9 +242,8 @@ fn the_readout_reproduces_its_bytes_and_pins_exact_decimal_aggregates() { }), ); - // The forbidden shortcut over the same readings: subtracting the step medians reads - // 2 - 4 = -2 where the pair-first median reads -1, so the pinned aggregate cannot have - // come from step-aggregate subtraction. + // The step medians are 2 and 4, giving difference −2. The pair-difference median is −1. + // Therefore the pair-first aggregate distinguishes these two calculations. let typed = |readings: [f64; 4]| { readings.map(|reading| DFinite::new(reading).expect("the fixture readings are finite")) }; diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/mod.rs index c4d4bbf3e87..f4f61439a11 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/mod.rs @@ -1,4 +1,4 @@ -//! The paired-movement readout. +//! Pair convergence and control displacement between ladder conditions. //! //! The ladder's step evidence ([`measure_ladder`](super::measure_ladder)) reads the [relation //! lens](crate::salt::projector) through the whole layout: alignment residual and relation loss per @@ -9,16 +9,16 @@ //! step](super::Conditions) and the [canonical step](super::select_canonical), and for a control //! sample bounded by the pair count, drawn from [nonparticipant rows](mod@census), it measures //! displacement between the same steps, stratified by zero-step distance to the nearest sampled -//! endpoint. The control separates pair convergence from drift of the layout around it. The readout -//! persists as [ladder evidence](crate::file::salt::metadata::LadderEvidence) beside the step -//! measurements and blocks nothing. +//! endpoint. These populations show pair convergence beside surrounding displacement, without +//! isolating a causal effect. The readout persists as [ladder +//! evidence](crate::file::salt::metadata::LadderEvidence) beside the step measurements. Census and +//! frame-count failures become typed evidence outcomes. Salt-encoding failure remains an error. //! -//! The draw is a pure function of the [generation](crate::file::generation::Generation)'s declared -//! inputs. Deriving the sample rather than storing it keeps the evidence body to its aggregates and -//! every sample replayable from its generation. [`identity`] names the rule and derives the salt, -//! [`census`] draws both samples, [`movement`] reads each drawn subject at both steps, [`evidence`] -//! aggregates the readings into the persisted body, and [`measure`](mod@measure) runs the whole -//! readout for one generation. +//! Deriving the sample from metadata and the attraction index avoids persisting selected +//! identities. Replay requires the recorded rule's serialization conventions and the same index and +//! corpus row identities. [`identity`] derives the salt and subject order, and [`census`] selects +//! both samples. [`movement`] reads each drawn subject at both steps. [`evidence`] aggregates those +//! readings, and [`measure`](mod@measure) runs the readout for one generation. mod census; mod evidence; @@ -28,11 +28,6 @@ mod identity; mod measure; mod movement; -// The lib consumes the sibling modules inside this module only, so these re-exports are the -// module's whole production API. The test-only names are the inputs of the two external -// acceptance suites: the fit writer's replay assertions and the salt document's wire pins. -// The projector's evidence reading shares the aggregate family, so its gauge displacement -// quantiles keep the control evidence's exact shape by construction. #[cfg(test)] pub(crate) use self::{census::Draw, evidence::MovementOutcome, identity::RuleIdentity}; pub(crate) use self::{ diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/movement/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/movement/mod.rs index 11231e092cd..05302e69f47 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/movement/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/movement/mod.rs @@ -1,23 +1,31 @@ -//! The per-subject movement readings between two aligned steps. +//! Per-pair local-rank changes and control displacement between aligned steps. //! //! [`Movement`] holds the zero-condition and canonical frames of one generation, already aligned //! by the ladder, and reads each drawn subject. A pair reading ([`Movement::pair`]) carries the //! source-to-partner distance and the partner's local rank at both steps. A control reading //! ([`Movement::control`]) carries the row's displacement between the steps and its zero-step -//! distance to the nearest sampled anchor, the stratification key of the collateral deciles. +//! nearest-anchor reading, the stratification key of the collateral deciles. //! -//! The local rank follows the union domain. With `K_r(u)` the exact `k` nearest other rows -//! of `u` at step `r`, the partner's rank at step `r` is one plus the number of rows in -//! `K_0(u) ∪ K_c(u)` whose `(distance at step r, row)` orders before the partner's own reading. -//! The union is the candidate domain at both steps. A row that leaves the neighbourhood at one -//! step is still rank-relevant at the other, and a rank read over a single step's `k`-set alone -//! would miscount it. -//! Every distance is [`Vec2::distance_squared_wide`](crate::math::Vec2::distance_squared_wide), -//! the one metric, whether the row came from a tree readout or enters the comparison directly, -//! so tie sets never depend on the path that produced a reading. +//! The ideal local rank follows the union domain. Let Kᵣ(u) be the first k other rows of source u +//! at step r ∈ {zero,canonical}, ordered by (squared distance, row). A frame with fewer than k +//! other rows uses them all. Let U(u) be the union of both Kᵣ(u). For partner v, rankᵣ(u,v) = 1 + +//! |{w ∈ U(u) : (dᵣ²(u,w),w) < (dᵣ²(u,v),v)}|, where dᵣ² is the squared-distance metric. The union +//! is the candidate domain at both steps. A row that leaves the neighbourhood at one step is still +//! rank-relevant at the other, and a rank read over a single step's `k`-set alone would miscount +//! it. //! -//! Readings are exact and deterministic: the trees only accelerate neighbour selection, and the -//! module's tests hold every reading to a full-scan restatement of the same definitions. +//! The readout applies this equation to the union actually returned by the two [`KdTree`] queries. +//! Returned candidates are ordered by (squared distance, row) and limited to `k` per step. These +//! sets agree with Kᵣ(u) when the probed boundary is correct and radius selection is complete. The +//! readout inherits the index's selection-precision limit, including for nearest-anchor queries. +//! +//! Squared distances use [`Vec2::distance_squared_wide`](crate::math::Vec2::distance_squared_wide), +//! including direct comparisons and tree readouts. Finite f32 coordinates have magnitude below +//! 2¹²⁸. Widening before subtraction bounds the two-dimensional squared distance below 2²⁵⁹, within +//! f64's finite range. Pair distances and control displacements take their finite square roots. +//! +//! Equal frame bytes and query inputs give equal ranks under the same index implementation and +//! arithmetic semantics. Square roots can limit cross-platform bitwise replay of distances. #![expect( clippy::min_ident_chars, reason = "`k` is the k-nearest-neighbour count's literature name" @@ -46,10 +54,9 @@ pub(super) const RANK_WINDOW: NonZero = NonZero::new(256).expect("256 is /// The reading of one drawn pair, its distance and local rank at both steps. /// -/// Distances are world-unit Euclidean readings, finite and non-negative by construction. Ranks -/// are one-based over the union domain, so a rank never exceeds `1 + 2k`. Downstream -/// aggregation forms the per-pair differences from these fields directly and never subtracts -/// step aggregates. +/// Distances are world-unit Euclidean readings, finite and non-negative for valid input fields. +/// Ranks are one-based over the union domain and bounded by 1 + 2k under [`Movement::new`]'s window +/// requirement. Aggregation forms differences per pair and never subtracts step aggregates. #[derive(Debug, Copy, Clone, PartialEq)] pub(super) struct PairMovement { /// The source-to-partner distance at the zero step. @@ -67,17 +74,14 @@ pub(super) struct PairMovement { pub(super) struct ControlMovement { /// The row's displacement between the aligned zero and canonical steps. pub displacement: DNonNegative, - /// The row's zero-step distance to the nearest sampled anchor. + /// The row's zero-step nearest-anchor reading from [`Movement::anchor_distance`]. pub anchor_distance: DNonNegative, } hashql_core::id::newtype! { /// A position in the sampled anchor frame. /// - /// The anchor frame restates the drawn pairs' deduplicated endpoints as a compact frame of its own, - /// so a position in it is not a corpus row, and confusing the two domains is the wiring defect this - /// type prevents. The `u32` width holds the domain whole: the anchor count is bounded by twice the - /// pair sample cap. + /// Anchor positions have their own row domain, distinct from corpus rows. The drawn pairs contribute at most twice [`SAMPLE_CAP`](super::census::SAMPLE_CAP), or 145,110 anchors, within the `u32` domain. #[id(const)] pub(super) struct AnchorRowId(u32) } @@ -125,12 +129,20 @@ pub(super) struct Movement<'frame> { impl<'frame> Movement<'frame> { /// Builds the readout over both aligned frames. /// - /// The frames share the corpus row domain, so row `r` is one corpus row's position at each - /// step. Alignment between the steps is the caller's and arrives already applied. + /// Both frames must identify the same corpus row at each position, with alignment already + /// applied. Only their lengths are checked. To keep every possible union rank representable, + /// `k` must satisfy 2k + 1 ≤ `u32::MAX`. This window requirement is not checked here. + /// + /// Builds and retains a [`KdTree`] for each frame, with O(N) index storage for N rows. Each + /// frame must satisfy the index's capacity and typed row-domain requirements. /// /// # Errors /// /// [`MovementError::Rows`] when the frames disagree on the row count. + /// + /// # Panics + /// + /// Panics under [`KdTree::build`]'s capacity or typed-domain failure conditions. pub(super) fn new( zero: &'frame FinitePointField, canonical: &'frame FinitePointField, @@ -154,12 +166,23 @@ impl<'frame> Movement<'frame> { /// Reads one drawn pair: distances and union-domain local ranks at both steps. /// - /// The readouts and the union buffer allocate in `scratch`, and the reading returns none of - /// them, so the caller resets the arena between readings to reclaim them in bulk. + /// The source and partner may be equal. Only the source is excluded from neighbour selection. + /// The partner itself never counts ahead of its own reading, although a lower row at the same + /// position can. + /// + /// Reset `scratch` between readings to reclaim the neighbour results and union buffer. + /// Tree-query scratch can also allocate outside this arena. + /// + /// # Complexity + /// + /// Worst-case O(N log N) work and O(N) temporary storage for N frame rows. A boundary tie can + /// include every row before the neighbour selection sorts and truncates, even when `k` is + /// small. /// /// # Panics /// - /// This panics when `source` or `partner` is not a frame row. + /// This panics when `source` or `partner` is not a frame row, or query or union-buffer capacity + /// is exceeded. Queries also inherit [`KdTree::nearest_in`]'s typed-domain requirements. pub(super) fn pair( &self, source: NodeRowId, @@ -180,12 +203,10 @@ impl<'frame> Movement<'frame> { let partner_zero = source_zero.distance_squared_wide(self.zero[partner]); let partner_canonical = source_canonical.distance_squared_wide(self.canonical[partner]); - // One plus the union rows ordering before the partner under `(distance, row)`. The - // partner itself never counts: its own reading compares equal, and equality is not - // before. Recomputing every union member's reading through the one metric keeps the - // comparison independent of which step's readout supplied the row. The union rows are - // scattered by construction, so the sweep gathers four at a time through the lane - // metric, whose readings equal the scalar metric's bit for bit. + // Both steps rank against the same union, irrespective of which step supplied a row. The + // lane metric widens before subtraction and uses the scalar metric's separate square and + // addition roundings. The four-row gathers therefore preserve the scalar comparison and its + // ties. let mut rank_zero: u32 = 1; let mut rank_canonical: u32 = 1; let source_zero_batch = Vec2x4T::splat(source_zero); @@ -234,14 +255,14 @@ impl<'frame> Movement<'frame> { /// Reads one drawn control row against the sampled anchor index. /// - /// `anchors` indexes the anchor positions at the zero step, and the reading's - /// `anchor_distance` is [`Self::anchor_distance`], the nearest of them to this row's - /// zero-step position. + /// `anchors` indexes the anchor positions at the zero step. The reading's `anchor_distance` is + /// [`Self::anchor_distance`], subject to the index's selection precision and typed-domain + /// requirements. /// /// # Panics /// - /// This panics when `row` is not a frame row or when `anchors` indexes an empty frame - /// ([`Self::anchor_distance`]). + /// This panics when `row` is not a frame row, the anchor query returns no candidate, or query + /// capacity is exceeded (see [`Self::anchor_distance`]). pub(super) fn control( &self, row: NodeRowId, @@ -256,17 +277,17 @@ impl<'frame> Movement<'frame> { } } - /// Reads one frame row's zero-step distance to the nearest sampled anchor. + /// Reads one frame row's zero-step distance to the index-selected anchor. /// - /// The one anchor readout. The drawn controls ([`Self::control`]) and the evidence writer's - /// candidate sweep both read through it, so a drawn control's reading is bit-identical to - /// its own candidate reading and the collateral strata always cover the draw. + /// The nearest-anchor query inherits [`KdTree`]'s selection precision and typed-domain + /// requirements. With the same frame and anchors, repeated calls read the same distance under + /// the same index implementation and arithmetic semantics. This gives sampling and the + /// candidate census a common stratification key. /// /// # Panics /// - /// This panics when `row` is not a frame row or when `anchors` indexes an empty frame. An - /// empty anchor frame is unreachable from the draw: readings run only under a nonempty - /// draw, and every drawn pair contributes an endpoint. + /// Panics when `row` is not a frame row, the anchor query returns no candidate, or query + /// capacity is exceeded. An empty anchor frame returns no candidate. pub(super) fn anchor_distance( &self, row: NodeRowId, diff --git a/libs/@local/graph/atlas/src/salt/ladder/paired/movement/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/paired/movement/tests.rs index a4a4831eb10..baa1a91856f 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/paired/movement/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/paired/movement/tests.rs @@ -1,8 +1,7 @@ //! Movement readout expectations. //! -//! The oracle restates one pair reading from full scans over both frames: k-sets by sort, the -//! union domain, then the counting rule, so the tree-backed readout is checked against a plain -//! statement of the same tie semantics. +//! The oracle constructs both neighbour sets by full sort, forms their union and counts rows ahead +//! of the partner. Its row tie-break supplies an independent comparison for tree-based selection. use alloc::collections::BTreeSet; use core::{iter, num::NonZero}; @@ -21,7 +20,7 @@ use crate::{ salt::ladder::paired::fixtures::frame, }; -/// A seeded frame on a coarse integer lattice, so exact distance ties are common. +/// Generates a seeded integer-lattice frame with frequent exact distance ties. fn lattice_frame(seed: u64, rows: usize) -> Vec { let mut rng = Xoshiro256PlusPlus::seed_from_u64(seed); iter::repeat_with(|| { @@ -34,7 +33,14 @@ fn lattice_frame(seed: u64, rows: usize) -> Vec { .collect() } -/// Restates one pair reading from full scans: k-sets by sort, union, then the counting rule. +/// Measures one pair using full-scan neighbour sets and union-domain ranks. +/// +/// Fixture frames must have fewer than 2³² rows. +/// +/// # Panics +/// +/// Panics when `source`, `partner` or a union row is outside a compared frame, or the rank cannot +/// fit `u32`. #[expect( clippy::cast_possible_truncation, reason = "test frames stay far below u32::MAX rows, so row conversions are exact" @@ -141,8 +147,8 @@ fn an_exact_distance_tie_resolves_by_row_identity() { .expect("the frames are finite and equal"); let scratch = Scratch::new(); - // Row 1 orders before the tied partner 2, so it counts against partner 2 and not the - // other way round. + // row 1 breaks the distance tie ahead of partner 2. Reversing the partners excludes that + // contribution. assert_eq!( movement.pair(NodeRowId::new(0), NodeRowId::new(2), &scratch), PairMovement { @@ -165,10 +171,10 @@ fn an_exact_distance_tie_resolves_by_row_identity() { #[test] fn a_partner_outside_one_step_ranks_over_the_union_domain() { - // At the zero step the k = 2 set of row 0 is {1, 2}, and at the canonical step it is - // {4, 3}. Partner 4 enters only the canonical set and partner 1 only the zero set, and each - // rank at the partner's absent step needs a union row its own k-set does not carry, so a - // single-step candidate domain reads 3 where the union reads 4. + // At k = 2, row 0's neighbour sets are {1, 2} at zero and {4, 3} at canonical. Partner 4's + // zero-step rank needs row 3, and partner 1's canonical-step rank needs row 2. Each extra row + // belongs only to the other step's set. Therefore the union reads rank 4 where a single-step + // domain would read 3. let zero = [ Vec2::new(0.0, 0.0), Vec2::new(1.0, 0.0), diff --git a/libs/@local/graph/atlas/src/salt/ladder/report/mod.rs b/libs/@local/graph/atlas/src/salt/ladder/report/mod.rs index d5c3555cd66..2e8e13066ff 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/report/mod.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/report/mod.rs @@ -1,52 +1,48 @@ -//! The relation-effect report over one published generation's condition ladder. +//! Endpoint contraction and corpus displacement across a published condition ladder. //! -//! The ladder's evidence records how far each step moved and what the frozen relation loss read, -//! both in the trainer's own vocabulary of RMS movement after alignment and loss in local-scale -//! units. Neither states the product's named claim - that the relation lens ends related entities -//! closer on the published map. This report reads that claim in the map's own units. For every -//! force-bearing relation instance it takes the distance between its endpoints at the -//! zero-condition step against the same distance at every other step, aggregated with the -//! trainer's own engagement mass. The bridge between the two vocabularies is the point of the -//! bundle: the manifest's loss column rides beside the measured contraction so a reader can see -//! where they disagree. +//! Frozen relation loss measures the [projector](crate::salt::projector)'s locally normalized +//! objective, while this report asks whether retained relation endpoints become closer in map +//! units. It compares every stored attraction instance's endpoint distance at the [zero-condition +//! step](super::Conditions) with its distance at each later step. Loss and contraction appear +//! together to expose disagreement between the objective and map geometry. //! //! # What the report reads //! -//! Everything comes from the published generation directory; the report dials no store and mutates -//! nothing. The trained checkpoint projects the representation matrix at every step condition of -//! the recorded schedule (rows project independently, so this reproduces the fit's own frames), -//! each raw frame maps into the baseline frame through its recorded manifest alignment, and the -//! attraction index supplies the engaged pairs with their weights. Frames are rebuilt rather than -//! read because the fit persists them only as scratch. The checkpoint, the schedule, and the -//! alignments together determine them exactly. +//! The published generation supplies the checkpoint, representations, recorded schedule and +//! alignments, plus the attraction index. The report reconstructs all step frames on the supplied +//! device and applies each recorded alignment into the baseline frame. Per-step raw frames are fit +//! scratch artifacts and are not available as published columns. Row-independent projection +//! establishes the mathematical reconstruction, but backend kernels and floating-point behavior can +//! change its bits. No database or embedding provider is contacted. //! //! # The certificate //! -//! Before the report takes any reading, the canonical step's rebuilt aligned frame must reproduce -//! the published coordinate column within [`CERTIFICATE_TOLERANCE`] world units per component. A -//! report whose forward pass does not reproduce the published bytes would describe a lookalike, -//! so failure panics instead of reporting. The tolerance derives from measurement. Independent -//! full-corpus reproductions of two prior generations reached a maximum component error near -//! `1e-4`, and the bound stands one order above that floor. The bundle carries the measured -//! residual, so drift toward the bound is visible long before it fails. +//! The canonical reconstruction must have absolute component error strictly below +//! [`CERTIFICATE_TOLERANCE`] (0.001 world units) against the published coordinate column. Failure +//! panics. This establishes numerical agreement at the canonical step, not byte equality or +//! reconstruction accuracy at every other step. The bound uses a calibration of full-corpus +//! reconstructions of two generations with maximum component error near 0.0001. The report includes +//! the measured residual for comparison with that bound. //! //! # Bases, stated //! -//! - Distances are Euclidean world units in the baseline frame after each step's own manifest -//! alignment. The alignment quotients the global similarity freedom the projector never promises -//! to pin down, exactly as the ladder's published movement evidence does; a uniform rescale of a -//! whole frame is therefore not read as contraction. -//! - The engagement mass of an instance is the trainer's own loss factor, computed identically: -//! `confidence * degree_normalization * strength`. Mass-weighted and unweighted aggregates ride -//! together, and every group row carries its channel weights, so no single convention hides the -//! other. -//! - Populations are exact. Every retained instance and every corpus row enters, with no sampling -//! anywhere. Byte-identical duplicate rows project identically and enter with their multiplicity, -//! the same convention the trainer's loss uses. Confidence machinery over these point readings is -//! a later concern. Nothing here is an estimate. -//! - The contracted fraction counts instances whose endpoint distance strictly shrank against the -//! baseline; ties - including coincident duplicates at distance zero on both steps - do not -//! count. +//! - Distances use the baseline frame's world units after each step's recorded alignment. A pure +//! source similarity is removed by an ideal exact alignment, up to the implementation's rounding. +//! Changing the baseline's unit scale changes the reported magnitudes. +//! - For instance i, engagement mass is mᵢ = confidenceᵢ · normalizationᵢ · strengthᵢ. This uses +//! the loss readout's per-instance factor expression. Channel weights and locally normalized +//! class energies are separate parts of the loss, and the report shows the channel weights per +//! group. +//! - With contraction Δᵢ = distanceᵢ,zero − distanceᵢ,step, the mass-weighted mean is Σᵢ mᵢΔᵢ/Σᵢ +//! mᵢ, or zero at zero mass. The unweighted mean is Σᵢ Δᵢ/E for E instances, or zero at E = 0. +//! Positive values mean contraction. The contracted fraction counts only Δᵢ > 0, excluding ties. +//! - The report measures every stored instance and every corpus row. It preserves stored +//! multiplicity and does not deduplicate oriented pairs or sample them. This population differs +//! from the distinct-row training domain and its capped draws. Complete coverage removes sampling +//! error, not numerical error or uncertainty about causal effects. +//! - Endpoint differences and lengths compute in `f32` before widening to `f64` for the serial +//! aggregates. Even finite fields can overflow this distance path. The report refuses non-finite +//! aggregate results instead of treating them as measurements. #[cfg(test)] mod tests; @@ -103,14 +99,18 @@ pub(crate) struct LadderReport { pub canonical_condition: NonNegative, /// The schedule index with the largest mass-weighted mean contraction. /// - /// Ties keep the first. Index `0` states that no step contracts the engaged pairs against the - /// baseline at all. + /// Ties keep the first. With the baseline's zero contraction at index 0, a selected index of 0 + /// means no step has a strictly positive mass-weighted mean. Individual pairs can still + /// contract. pub contraction_argmax_index: usize, /// Whether the published step is the contraction argmax. pub canonical_is_argmax: bool, /// The canonical step's reproduction residual against the published coordinate column. pub certificate: Certificate, - /// One reading per step, in schedule order. The baseline step is the all-zero row. + /// One reading per step, in schedule order. + /// + /// The baseline has zero contraction and displacement, while retaining its counts, masses and + /// recorded loss. pub steps: Vec, } @@ -132,7 +132,7 @@ pub(crate) struct Certificate { pub(crate) struct StepReading { /// The step's condition value. pub condition: NonNegative, - /// The manifest's frozen relation loss at projection time, local-scale units. + /// The manifest's frozen relation loss using locally normalized distances. pub relation_loss: DNonNegative, /// The manifest's RMS movement against the baseline field after alignment. pub baseline_movement: DNonNegative, @@ -158,7 +158,7 @@ pub(crate) struct ContractionReading { pub edge_count: usize, /// Total engagement mass of those instances. pub total_mass: DNonNegative, - /// Mass-weighted mean of `baseline distance - step distance`, world units. + /// Mass-weighted mean of baseline distance minus step distance, in world units. /// /// Zero when no mass entered. pub mass_weighted_mean: DFinite, @@ -216,8 +216,15 @@ impl LadderReport { /// no measured ladder, when an artifact fails to open or disagrees with another about the row /// domain, when the echoed schedule and the ladder evidence describe different ladders, or /// when the rebuilt canonical frame does not reproduce the published coordinate column within - /// [`CERTIFICATE_TOLERANCE`]. A report run has no recovery path, and the error is the - /// diagnosis. + /// [`CERTIFICATE_TOLERANCE`]. Non-finite projection, alignment, distance or aggregate results + /// also panic. + /// + /// # Complexity + /// + /// For S steps, N rows, G groups and E stored instances, the reading retains O(S(N + G) + E) + /// host data for frames, terms and results, in addition to model and device working memory. + /// Group extraction repeatedly validates archive regions, costing O(G(G + E)). Reading all + /// steps costs O(S(N + G + E)) after frame reconstruction. #[tracing::instrument(skip_all)] pub(crate) fn compile( root: &GenerationRoot, @@ -270,7 +277,8 @@ impl LadderReport { "the attraction index and the representation matrix disagree on the row domain", ); - // The model, opened on the placement backend the fit trained on. + // reconstruct on the supplied inference device. The certificate checks its canonical output + // against the published column. let model: Projector = artifact::open_model( File::open(generation.path_of(&checkpoint.name())).expect("the checkpoint opens"), options.architecture, @@ -328,7 +336,7 @@ struct GroupTerms { terms: Vec, } -/// The echoed options beside the measured ladder evidence, certified to describe one ladder. +/// Echoed projector options with a matching recorded condition sequence. struct LadderSources<'source> { /// The echoed projector options. options: &'source ProjectorOptions, @@ -341,9 +349,12 @@ impl<'source> LadderSources<'source> { /// /// # Panics /// - /// This panics when the generation placed rows by landmark baseline, when it published no - /// ladder, or when the echoed schedule and the evidence describe different ladders: a - /// reading over disagreeing sources would describe neither. + /// Panics for landmark-baseline placement, missing projector or ladder evidence, a + /// condition-sequence mismatch, an out-of-range recorded canonical index, or disagreement + /// between that entry and the recorded canonical condition. + /// + /// This does not compare the echoed canonical option with the evidence's canonical condition or + /// validate the recorded alignments. pub(crate) fn new(repository: &'source SaltRepository) -> Self { let PlacementOptions::Projector(options) = &repository.metadata.reproducibility.config.placement @@ -385,12 +396,15 @@ impl<'source> LadderSources<'source> { } } -/// Rebuilds every step: the production forward pass at the step's condition, then the recorded -/// manifest alignment into the baseline frame. +/// Reprojects each recorded condition and applies its recorded baseline alignment. +/// +/// The representation and role columns must cover the same rows. All aligned frames remain +/// allocated in the result. /// /// # Panics /// -/// This panics when a forward slice reads back a non-finite coordinate. +/// Panics when the columns do not cover a requested row range, projection fails, or applying an +/// alignment produces a non-finite coordinate. fn rebuild_frames>( model: &Projector, columns: NodeColumns<'_, NodeRowId>, @@ -424,7 +438,10 @@ fn rebuild_frames>( .collect() } -/// Materializes every group's instances with the trainer's own loss factor as their mass. +/// Materializes every stored instance with its confidence-normalization-strength mass. +/// +/// Each group lookup revalidates archive regions. Extracting G groups over E edges costs O(G(G + +/// E)) work and O(G + E) result storage. fn materialize_terms(attraction: &AttractionArchive) -> Vec { (0..attraction.group_count()) .map(|index| { @@ -438,7 +455,7 @@ fn materialize_terms(attraction: &AttractionArchive) -> Ve .map(|edge| EdgeTerm { source: edge.source, target: edge.target, - // The trainer's loss factor, computed identically (`relation_loss`). + // preserve the loss readout's factor expression and multiplication order. mass: (edge.confidence.value() * edge.normalization) * weights.strength.widen(), }) @@ -448,8 +465,15 @@ fn materialize_terms(attraction: &AttractionArchive) -> Ve .collect() } -/// Reads every step against the baseline: contraction per group and in aggregate, and the point -/// displacement of both row populations. +/// Reads per-step contraction and participant/nonparticipant displacement. +/// +/// `aligned` must follow the evidence's step order. The frames, endpoints and participant mask must +/// describe the same row domain. +/// +/// # Panics +/// +/// Panics when the baseline or a recorded step is missing, an indexed row is outside its frame, or +/// a contraction or displacement aggregate is non-finite. fn read_steps( evidence: &LadderEvidence, aligned: &[Box>], @@ -498,9 +522,11 @@ fn read_steps( /// /// # Panics /// -/// This panics when the largest absolute component error reaches [`CERTIFICATE_TOLERANCE`] and -/// when any rebuilt component is non-finite: the rebuilt frames would describe a lookalike of the -/// published generation, and no reading over them is evidence about it. +/// Panics when frame lengths differ, a measured residual is non-finite, or the largest absolute +/// component error is at least [`CERTIFICATE_TOLERANCE`]. +/// +/// Component subtraction and point lengths use `f32` before widening. Empty paired fields yield +/// zero residuals and pass the bound. #[expect( clippy::cast_precision_loss, reason = "corpus row counts sit orders of magnitude below the f64 mantissa" @@ -528,8 +554,8 @@ fn certify( } let components = (rebuilt.len() * 2) as f64; - // `f64::max` skips NaN, so the max fold alone cannot certify finiteness. Any NaN component - // poisons the component sum, and the mean's constructor is the refusal. + // finite input coordinates can still overflow f32 subtraction or length calculation. Validate + // the residual summaries before comparing the component bound. let non_finite = "the rebuilt canonical frame carries a non-finite coordinate and reproduces nothing"; let certificate = Certificate { @@ -553,6 +579,14 @@ fn certify( } /// Aggregates endpoint-distance contraction over one instance population. +/// +/// Distances compute through the `f32` vector-length path. Endpoint differences, their squared +/// lengths and the subsequent mass and difference folds must remain finite. A [`FinitePointField`] +/// alone does not establish those arithmetic bounds. +/// +/// # Panics +/// +/// Panics for an out-of-frame endpoint or a non-finite completed weighted or unweighted mean. #[expect( clippy::cast_precision_loss, reason = "instance counts sit orders of magnitude below the f64 mantissa" @@ -565,10 +599,9 @@ fn contract( let mut edge_count = 0_usize; let mut contracted = 0_usize; let mut total_mass = DNonNegative::ZERO; - // The mass-weighted fold multiplies two unbounded operands, so it runs as a derivation and - // makes its one claim at the mean's construction. The unweighted fold keeps the typed - // escape op by its totality theorem: every difference is bounded by 2·f32::MAX and the - // population by the corpus rows, so the serial sum stays far inside `f64`. + // the weighted mean validates its completed derivation. If the f32 distances remain finite, + // each unweighted difference has magnitude below 2¹²⁸. Fewer than 2⁶⁴ terms then keep the sum + // below 2¹⁹². Finiteness of the input coordinates alone does not establish the first premise. let mut weighted_sum = Derivation::::ZERO; let mut unweighted_sum = DFinite::ZERO; @@ -610,7 +643,14 @@ fn contract( /// Summarizes point displacement between two aligned frames over one row population. /// -/// `engaged` selects which side of the participant mask enters. +/// `engaged` selects which side of the participant mask enters. Only matching mask positions are +/// read. Rows beyond the mask's length do not enter either population. Differences and lengths +/// compute in `f32` before widening. +/// +/// # Panics +/// +/// Panics when a selected mask position exceeds either frame or a resulting displacement summary is +/// non-finite. #[expect( clippy::cast_precision_loss, reason = "row counts sit orders of magnitude below the f64 mantissa" diff --git a/libs/@local/graph/atlas/src/salt/ladder/report/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/report/tests.rs index ad87db85541..467220cadc6 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/report/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/report/tests.rs @@ -1,8 +1,4 @@ -//! Unit tests of the report's aggregation core. -//! -//! The compile path needs a published generation and stays with the integration suites; the -//! aggregates, the displacement summaries, the argmax rule, and the certificate bound are pure and -//! verify here on hand-derived values. +//! Hand-derived contraction, displacement and certificate fixtures. use hashql_core::id::{Id as _, IdSlice}; @@ -12,7 +8,9 @@ use crate::{ math::{FinitePointField, Vec2, d_finite, d_non_negative}, }; -/// Wraps fixture points every test states as finite literals. +/// Interprets fixture points as a corpus-row field. +/// +/// Every coordinate must be finite. fn field(points: &[Vec2]) -> &FinitePointField { FinitePointField::new_unchecked(IdSlice::from_raw(points)) } @@ -63,7 +61,6 @@ fn contraction_weighs_the_trainer_mass() { assert!((f64::from(reading.contracted_fraction) - 0.5).abs() < 1e-12); } -/// An unchanged distance is not a contraction: ties stay out of the contracted count. #[test] fn contraction_ties_do_not_count() { let baseline = [Vec2::new(0.0, 0.0), Vec2::new(1.0, 0.0)]; @@ -82,7 +79,6 @@ fn contraction_ties_do_not_count() { assert!(f64::from(reading.unweighted_mean).abs() < 1e-12); } -/// An empty population reads as zeros, never as a division artifact. #[test] fn contraction_of_nothing_is_zero() { let baseline = [Vec2::new(0.0, 0.0)]; @@ -103,8 +99,8 @@ fn contraction_of_nothing_is_zero() { /// The participant mask splits the displacement populations exactly. /// -/// Hand-derived: rows move by 0, 1, and 5; participants are rows 0 and 2, so the engaged side -/// reads mean 2.5 and RMS `sqrt(12.5)`, and the other side reads exactly 1 everywhere. +/// Rows move by 0, 1 and 5. Participants are rows 0 and 2, with mean (0 + 5)/2 = 2.5 and RMS √((0² +/// + 5²)/2) = √12.5. The nonparticipant row has mean, RMS and maximum displacement 1. #[test] fn displacement_splits_on_the_participant_mask() { let baseline = [ @@ -132,8 +128,6 @@ fn displacement_splits_on_the_participant_mask() { assert!((f64::from(bystanders.max) - 1.0).abs() < 1e-12); } -/// Ties keep the first index, an empty series reads the baseline, and an all-negative series -/// names the baseline as well: index zero states that no step beats doing nothing. #[test] fn argmax_keeps_the_first_and_defaults_to_the_baseline() { assert_eq!( @@ -167,7 +161,7 @@ fn certificate_reports_the_measured_residual() { assert!((f64::from(certificate.max_point_distance) - expected).abs() < 1e-12); } -/// A residual at the bound refuses: the frames would describe a lookalike. +/// Rejects a 0.002 component residual, above the 0.001 certificate bound. #[test] #[should_panic(expected = "does not reproduce the published coordinate column")] fn certificate_refuses_at_the_bound() { diff --git a/libs/@local/graph/atlas/src/salt/ladder/tests.rs b/libs/@local/graph/atlas/src/salt/ladder/tests.rs index a2a4a3cea49..4c6c5297414 100644 --- a/libs/@local/graph/atlas/src/salt/ladder/tests.rs +++ b/libs/@local/graph/atlas/src/salt/ladder/tests.rs @@ -14,15 +14,17 @@ hashql_core::id::newtype! { struct StepRow(u32) } -/// Proves a fixture's points finite over the tests' row domain. +/// Validates fixture coordinates over the tests' row domain. +/// +/// # Panics +/// +/// Panics when a coordinate is non-finite. #[track_caller] fn field(points: &[Vec2]) -> &FinitePointField { FinitePointField::new(IdSlice::from_raw(points)).expect("the fixture points are finite") } -/// A sixteen-point deterministic field. -/// -/// Spread radii and no symmetry a similarity could exploit. +/// Builds sixteen points with varied radii and angular spacing. fn base_field() -> Vec { (0..16_u8) .map(|index| { @@ -39,8 +41,8 @@ fn base_field() -> Vec { /// Swaps the axes of every even-indexed point. /// -/// A deformation no similarity can explain (axis swap alone is a reflection, which the -/// orientation-preserving family excludes). +/// On [`base_field`]'s asymmetric fixture this creates a residual deformation. An axis swap is a +/// reflection, outside the orientation-preserving fit family. fn deformed_field(base: &[Vec2]) -> Vec { base.iter() .enumerate() @@ -123,9 +125,10 @@ fn conditions_reject_each_violated_invariant() { ); } -/// A negative-zero input is `+0.0` before validation sees it ([`NonNegative`] canonicalizes the -/// sign of zero at construction), so no `-0.0` alias can reach the baseline check or condition -/// the projector with different bits. +/// Accepts the same baseline after canonicalizing either zero sign. +/// +/// [`NonNegative`](crate::math::NonNegative) converts negative zero to positive zero before +/// schedule validation. #[test] fn conditions_accept_a_canonicalized_negative_zero_baseline() { assert_eq!( @@ -215,9 +218,8 @@ fn pure_similarity_step_measures_negligible_movement() { .expect("scale 2.0 is normal and positive"); let moved: Vec = base.iter().map(|&point| transform.apply(point)).collect(); - // The raw coordinates moved far - more than one unit per point on - // average over the sixteen points; only alignment reveals that - // nothing about the layout's shape changed. + // more than one unit of raw displacement per point, although the transform preserves shape up + // to rounding. Only the aligned residual distinguishes that motion from deformation. let raw_displacement = base .iter() .zip(&moved) @@ -246,15 +248,14 @@ fn pure_similarity_step_measures_negligible_movement() { "a pure similarity image leaves no residual movement, moved {}", step.adjacent_movement.get() ); - // The fitted alignment inverts the transform: its scale undoes the - // doubling. + // the inverse alignment's scale undoes the doubling. assert!( (step.alignment.scale().get() - 0.5).abs() < 1e-4, "the alignment must recover the inverse scale, got {}", step.alignment.scale() ); - // Both comparands are the baseline field here, and the two fits run - // over identical slices, so the movements agree exactly. + // Both targets are the baseline. The sixteen rows fit within one parallel chunk, giving each + // call the same arithmetic order. Therefore both residuals agree exactly. assert_eq!(step.baseline_movement, step.adjacent_movement); } @@ -320,8 +321,8 @@ fn adjacent_and_baseline_movements_use_their_own_comparands() { "the repeat still differs from the baseline, got {}", repeat.baseline_movement.get() ); - // The repeated step's baseline alignment matches its predecessor's: - // identical fields fit identical alignments. + // the identical source and target slices fit in one chunk, preserving the same arithmetic order + // for both baseline alignments. assert_eq!(repeat.alignment, measurements[1].alignment); } @@ -455,9 +456,7 @@ fn canonical_selection_requires_an_exact_member() { assert_eq!(selected.index, 1); assert_eq!(selected.measurement.alignment, measurements[1].alignment); - // The step whose loss rose and whose movement collapsed onto its - // predecessor publishes like any other member: the measurements - // are diagnostics, and they block nothing. + // a rising loss and negligible adjacent movement do not prevent canonical selection. let risen = select_canonical(&measurements, non_negative!(1.0)).expect("the last step is a member"); assert_eq!(risen.index, 2); diff --git a/libs/@local/graph/atlas/src/salt/landmark/artifact.rs b/libs/@local/graph/atlas/src/salt/landmark/artifact.rs index d182098dbcc..ac3e4462c2a 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/artifact.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/artifact.rs @@ -1,10 +1,9 @@ //! The landmark skeleton's published form, one combined file and its mapped reader. //! -//! A fitted skeleton - selection, assignment, and layout coordinates - publishes as one -//! [`crate::file::landmark`] file, so the three parts that share the ordinal vocabulary cannot fall -//! out of sync. [`LandmarkSkeletonArchive`] reopens the file over a whole-file mapping and -//! validates the skeleton invariants once, so training and serving read landmark data from the page -//! cache without holding it on the heap. +//! A fitted skeleton publishes its selection, assignment and coordinates in one +//! [`crate::file::landmark`] file under a shared ordinal domain. [`LandmarkSkeletonArchive`] +//! validates row ordering, assignment bounds and coordinate finiteness over the mapped regions. +//! Accessors then borrow those regions without a heap copy. use core::{error::Error, fmt}; use std::io; @@ -27,7 +26,7 @@ use crate::{ math::Vec2, }; -/// An opened landmark file does not hold a valid skeleton. +/// A violation of the landmark archive's local invariants. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum InvalidLandmarkFile { /// The selected rows break the strictly ascending order. @@ -68,9 +67,9 @@ impl Error for InvalidLandmarkFile {} /// A fitted landmark skeleton, assembled for publication. /// -/// Selection, assignment, and coordinates share one ordinal vocabulary by construction. The -/// assignment stage works from the selection, so the assignment's ordinal domain is the selection's -/// length, and the constructor pins the coordinates to the same domain. +/// Selection, assignment and coordinates must derive from the same selected rows. Construction +/// checks the common landmark count and coordinate finiteness. Equal counts alone do not establish +/// that the parts use the same ordinal meanings. #[derive(PartialEq)] pub(crate) struct LandmarkSkeleton { selection: LandmarkSelection, @@ -87,7 +86,6 @@ where /// # Panics /// /// This panics when the parts disagree on the landmark count or a coordinate is not finite. - /// Both cases violate the contracts of the stages that produced them. #[must_use] pub(crate) fn new( selection: LandmarkSelection, @@ -153,14 +151,18 @@ where { type Error = io::Error; - /// Writes the skeleton as a landmark file. + /// Writes a skeleton whose selected rows encode as little-endian `u64` values. /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. + /// `N` must encode each selected row as one little-endian `u64`. The [`ByteStable`] bound alone + /// supplies no width or byte-order guarantee. /// /// # Errors /// - /// Returns an error when the underlying writer fails. + /// Returns the underlying I/O error when writing fails. + /// + /// # Panics + /// + /// This panics when the selected-row bytes cannot form exactly one `u64` per coordinate. fn write_into(&self, write: impl io::Write) -> io::Result { let mut writer = Writer { accumulator: Sha256::new(), @@ -179,16 +181,17 @@ where } } -// The skeleton publishes over the corpus row domain alone: the constructor mapped selection and -// assignment onto first corpus rows before any write. +// the published landmark artifact uses corpus row ids. impl WriteAs for LandmarkSkeleton {} /// A published landmark skeleton opened over its mapped file. /// -/// Construction checks the skeleton invariants once. Node rows are strictly ascending, every -/// assignment ordinal lies inside the landmark domain, and every coordinate is finite. An open -/// skeleton therefore serves only valid views, and consumers re-validate nothing. The regions stay -/// in the page cache under memory pressure and off the heap. +/// Selected node rows are strictly ascending, every assignment ordinal lies inside the landmark +/// domain, and every coordinate is finite. Construction validates these properties once. Accessors +/// borrow the mapped regions without repeating validation. +/// +/// Validation covers these local invariants. It does not establish that selected rows lie inside +/// the assignment's corpus domain or that assignments identify the corresponding selected rows. #[derive(Debug)] pub(crate) struct LandmarkSkeletonArchive { file: LandmarkFile, @@ -199,7 +202,7 @@ impl LandmarkSkeletonArchive { /// /// # Errors /// - /// Returns an error when the file violates a skeleton invariant. + /// Returns [`InvalidLandmarkFile`] when the file violates a local skeleton invariant. #[tracing::instrument(skip_all)] pub(crate) fn new(file: LandmarkFile) -> Result { let landmarks = file.landmarks(); diff --git a/libs/@local/graph/atlas/src/salt/landmark/assignment.rs b/libs/@local/graph/atlas/src/salt/landmark/assignment.rs index eba2ff7b77b..02c29e023e4 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/assignment.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/assignment.rs @@ -1,10 +1,9 @@ -//! Nearest-landmark assignment over the search backend. +//! Corpus-to-landmark assignment through nearest-neighbour search. //! -//! Every corpus row maps to the ordinal of its nearest selected landmark by cosine distance over -//! the projector representations. Landmarks map to themselves without a search; every other row -//! asks a backend built over exactly the landmark rows for its single nearest neighbour. The -//! backend keys landmarks by their corpus node row, and ordinals fall out of the selection's -//! ascending order. +//! Landmarks map to themselves without a search. Every other row takes the first result from a +//! backend built over exactly the selected landmarks, inheriting that backend's approximation +//! quality. Search keys are input row ids. The selection translates them to [`LandmarkOrdinal`] +//! positions shared with the quotient and layout. use core::{error::Error, fmt}; @@ -24,7 +23,7 @@ use crate::{ /// Dense corpus-to-landmark assignment in node-row order. /// /// Every stored ordinal lies below [`landmarks`](Self::landmarks), the length of the selection -/// whose ordinals it stores, so consumers index landmark-domain tables without re-validation. +/// whose ordinals it stores. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct LandmarkAssignment { landmark_by_row: Box>, @@ -35,7 +34,7 @@ impl LandmarkAssignment where N: Id, { - /// Wraps precomputed ordinals, for fixtures. + /// Validates precomputed fixture ordinals against `landmarks`. /// /// # Panics /// @@ -74,8 +73,8 @@ where /// Groups the assigned corpus rows by landmark ordinal. /// - /// Each landmark's run ascends in row order, because the enumeration ascends over rows: - /// consumers that fold a run in order fold it as a serial row pass would. + /// Each run preserves ascending row order. Folding it therefore uses the same order as a serial + /// pass over the assigned row domain. #[must_use] pub(super) fn runs(&self) -> Runs { Runs::from_pairs( @@ -86,11 +85,11 @@ where ) } - /// Re-indexes the assignment through `rows`: entry `i` of the result is this assignment's entry - /// at the `i`-th yielded row. + /// Re-indexes the assignment through the rows yielded in order. /// - /// This expands an assignment built over a quotient domain onto the domain `rows` maps from: - /// every row of the wider domain takes its representative's landmark, under the unchanged + /// Result entry `i` takes this assignment's entry at the `i`-th yielded row. This expands an + /// assignment built over a quotient domain onto the domain `rows` maps from: every row of + /// the wider domain takes its representative's landmark, under the unchanged /// ordinal vocabulary. /// /// # Panics @@ -186,17 +185,18 @@ impl LandmarkSelection where N: Id, { - /// Assigns every corpus row to its nearest selected landmark. + /// Assigns every input row to a selected landmark using `index`. /// - /// `embeddings` holds the projector representations in node-row order; a mapped `f32[N, 512]` - /// artifact yields the slice directly. The empty backend ingests exactly the landmark rows, - /// links under `rng`, and answers one nearest-neighbour query per non-landmark row, in - /// parallel and deterministically for a deterministic backend. + /// `embeddings` must hold l2-normalized projector representations in row order, and `index` + /// must be empty. The backend ingests exactly the selected rows and builds under `rng`. + /// Landmarks map to themselves. Non-landmark rows use the backend's first vector-search result, + /// with no independent check of nearestness. A deterministic backend gives deterministic + /// assignments despite parallel queries. /// /// # Errors /// - /// Returns an error when a selected row lies outside the corpus, the backend fails, or a - /// search returns nothing or a non-landmark row. + /// Returns [`AssignmentError`] for an out-of-domain selected row, a backend failure, or a + /// missing or non-landmark search result. #[tracing::instrument(skip_all)] pub(crate) fn assign( &self, diff --git a/libs/@local/graph/atlas/src/salt/landmark/layout.rs b/libs/@local/graph/atlas/src/salt/landmark/layout.rs index 88f2718433a..a75396df331 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/layout.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/layout.rs @@ -1,34 +1,48 @@ //! Semantic-graph layout by UMAP's negative-sampling update rule. //! -//! [`layout_landmarks`] places one 2D point per graph row. Sampled edges pull their endpoints -//! together and uniformly drawn vertices push the sampled endpoint away, each step the gradient of -//! its own pair energy under the [`AffinityCurve`]. The expected update is a field with no scalar -//! objective behind it (negatives move the anchor only); its expected per-pair repulsion is vertex -//! degree times the negative rate over the vertex count, so the layout reproduces the graph's -//! near-binary neighbourhood structure (the scaffold the skeleton stages consume) and the fuzzy -//! weights act through the edge schedule rather than as calibrated similarity targets. +//! [`layout_landmarks`] places one 2D point per graph row using [`AffinityCurve`]'s clipped and +//! regularized pair kernels. Attraction moves both endpoints. Negative sampling moves only the +//! anchor, leaving the sampled vertex unchanged. Fuzzy edge weights determine sampling frequency +//! rather than calibrated low-dimensional similarity targets. In general these updates neither +//! descend one scalar objective for the whole graph nor guarantee recovery of its neighbourhoods. //! -//! Sampling follows the edge weights. An edge is first due one full period of `maximum_weight / -//! weight` epochs in, so the strongest edge applies every epoch after the first and weaker edges -//! proportionally less often. An edge never due within the epoch budget drops out up front. Every -//! sampled edge additionally repels [`negative_sample_rate`](LayoutOptions::negative_sample_rate) -//! uniformly drawn vertices, and the learning rate decays linearly toward zero across the epoch -//! budget (the final epoch steps at `initial / epochs`). +//! # Schedule //! -//! Due edges apply in [`Vec2x4T`] batches of four. Gradients within one batch evaluate at the -//! batch's entry coordinates, and the four negative-sample gradients of one chunk accumulate -//! against one anchor position, which gives mini-batch semantics rather than strictly sequential -//! updates. +//! For a stored edge of weight w > 0, let P = wₘₐₓ / w be its period. It is first due at P and +//! subsequently advances by P on each visit. With E epochs numbered 0 through E − 1, periods +//! greater than E − 1 drop out before optimization. The strongest edges have P = 1 and apply once +//! per epoch after epoch zero. An epoch budget of one returns the initialization even when the +//! graph has edges. //! -//! Points start on a jittered circle of diameter ten, an extent that puts the per-axis [gradient -//! clip](AffinityCurve::GRADIENT_CLIP) at 0.40 of the initial span. Every draw comes from the -//! caller-seeded generator, so a rerun over an equal graph, curve, options, and seed reproduces the -//! layout exactly. Rows without edges keep their initial placement: no attraction schedules them, -//! and repulsion moves only the sampled endpoint. +//! Each due directed edge draws r = [`negative_sample_rate`](LayoutOptions::negative_sample_rate) +//! vertices uniformly with replacement. If dᵢ(e) scheduled edges have anchor i in epoch e and the +//! graph has N vertices, the expected number of draws from i to any fixed j is r · dᵢ(e) / N. +//! Different anchor frequencies can produce unequal opposite repulsive updates. This is the +//! sampling model, not a symmetric cross-entropy gradient. Every draw remains in the schedule, +//! including self draws, whose coincident-pair gradient is zero. //! -//! The optimizer is serial by design: each gradient step reads coordinates the previous step wrote, -//! and the bit-reproducible layout is the property the serial order buys. Parallelism belongs to -//! the stages around it, not inside the epoch loop. +//! The learning rate is η(e) = η₀ · (1 − e / E). In real arithmetic the final epoch uses η₀ / E. +//! Periods, due times and learning rates use `f32`, with rounded division and repeated period +//! additions. Large epoch counts can lose integer precision and alter that ideal schedule. +//! +//! Due edges apply in [`Vec2x4T`] batches of four. Their gradients use the batch's entry +//! coordinates, and updates for shared vertices accumulate. Negative draws likewise apply in chunks +//! of four against one anchor position, followed by a scalar remainder. Changing this batch +//! structure changes the numerical method. +//! +//! # Initialization and reproducibility +//! +//! Vertex i starts at angle 2π · i / N with radius 5 · (1 + 0.01 · Uᵢ). +//! +//! Uᵢ is uniform in `[0, 1)` before floating-point rounding. The base diameter is ten, placing the +//! per-axis [gradient clip](AffinityCurve::GRADIENT_CLIP) of four at 0.40 of that base span before +//! learning-rate scaling. Rows with no scheduled edge keep their initial placement because negative +//! sampling never moves its target. +//! +//! Serial batch order fixes which coordinates each update reads. Equal graphs, curves, options and +//! random streams repeat the layout with the same floating-point behavior. Different targets or +//! kernels can change rounded results. The schedule and clipping provide no general convergence +//! guarantee. use core::{ array, error::Error, f32::consts::TAU, fmt, iter::Step, num::NonZero, simd::num::SimdFloat as _, @@ -43,9 +57,16 @@ use crate::{ salt::semantic::SemanticGraphView, }; +// The defaults are the UMAP reference defaults, carried as unvalidated starting points. The +// release evaluation's layout criteria (trustworthiness, landmark rank correlation) revise them +// from evidence. +/// The default epoch budget. const DEFAULT_EPOCHS: NonZero = const { NonZero::new(500).unwrap() }; +/// The default learning rate at epoch zero. const DEFAULT_INITIAL_LEARNING_RATE: Positive = positive!(1.0); +/// The default weight of repulsive updates. const DEFAULT_REPULSION_STRENGTH: NonNegative = non_negative!(1.0); +/// The default number of vertices repelled per sampled edge. const DEFAULT_NEGATIVE_SAMPLE_RATE: NonZero = const { NonZero::new(5).unwrap() }; /// Schedule settings for one layout, valid by construction. @@ -55,13 +76,13 @@ const DEFAULT_NEGATIVE_SAMPLE_RATE: NonZero = const { NonZero::new(5).unwra // evidence. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct LayoutOptions { - /// Optimization epochs. + /// Optimization epochs, 500 by default. pub epochs: NonZero = DEFAULT_EPOCHS, - /// Learning rate at epoch zero, decaying linearly toward zero across the epoch budget. + /// Initial learning rate, 1.0 by default, decaying across the epoch budget. pub initial_learning_rate: Positive = DEFAULT_INITIAL_LEARNING_RATE, - /// Weight of repulsive updates. Zero disables repulsion. + /// Weight of repulsive updates, 1.0 by default. Zero disables repulsion. pub repulsion_strength: NonNegative = DEFAULT_REPULSION_STRENGTH, - /// Vertices repelled per sampled edge. + /// Uniform vertex draws per sampled edge, 5 by default. pub negative_sample_rate: NonZero = DEFAULT_NEGATIVE_SAMPLE_RATE, } @@ -84,15 +105,21 @@ impl fmt::Display for EdgelessGraphError { impl Error for EdgelessGraphError {} hashql_core::id::newtype! { + /// The index of one directed edge in the layout's edge schedule. #[id(derive(Step), const)] struct LandmarkEdgeId(u32) } /// Lays out one point per graph row, in row order. /// -/// `graph` is the attraction structure - for the landmark skeleton, the quotient over the landmark -/// domain, indexed by ordinal - and `curve` the fitted low-dimensional kernel -/// ([`AffinityCurve::fit`]). `rng` drives the initial placement and the negative draws. +/// `graph` supplies the weighted attraction structure, and `curve` supplies the low-dimensional +/// kernels ([`AffinityCurve::fit`]). `rng` drives initialization and negative draws. The schedule +/// uses the graph's stored edge order. +/// +/// # Panics +/// +/// This panics when the retained schedule exceeds its `u32` edge-id domain or a raw CSR pointer +/// lies outside the graph's stored entries. /// /// # Errors /// @@ -127,19 +154,14 @@ where /// Radius of the initial circle. /// -/// The resulting diameter of ten puts the per-axis [`GRADIENT_CLIP`](AffinityCurve::GRADIENT_CLIP) -/// at 0.40 of the initial extent. -// Pinning the radius and the clip fixes one ratio: how far a single sample can move a point -// relative to the layout's extent. A knob on one side changes that ratio with nothing to signal the -// change. If the frame ever moves, both move together. The clip is the UMAP reference constant, -// whose frame spans twenty (ratio 0.20), and ten carries no recorded derivation, so the 0.40 here -// is an unvalidated starting point that the release evaluation's layout criteria revise from -// evidence. +/// The base diameter of ten gives a per-axis [`GRADIENT_CLIP`](AffinityCurve::GRADIENT_CLIP) ratio +/// of 4/10 = 0.40 before jitter and learning-rate scaling. +// the radius is an unvalidated starting point. Assess changes together with the clip-to-span ratio +// using trustworthiness and landmark rank correlation. const INITIAL_RADIUS: f32 = 5.0; /// Relative radial jitter of the initial circle, breaking the regular polygon's symmetry. -// Pinned, not configurable: any small positive value serves; the only -// distinguishable settings are zero (restores the symmetric saddle) -// and large (distorts the circle for nothing). +// one-percent radial variation breaks exact regular-polygon symmetry before rounding. Assess its +// scale through the layout quality measurements. const RADIAL_JITTER: f32 = 0.01; /// Places every vertex on the jittered initial circle, by vertex order. @@ -179,8 +201,13 @@ where { /// Extracts the edges due at least once within the epoch budget. /// - /// Returns [`None`] when the graph stores no edges. Weights are finite in `(0, 1]` by the - /// graph's invariants, so the schedule re-validates nothing. + /// Returns [`None`] only when the graph stores no edges. A nonempty graph can yield an empty + /// schedule if no edge is due within the budget. + /// + /// # Panics + /// + /// This panics when a raw CSR pointer lies outside the stored entries or the retained edge + /// count exceeds `E`'s id domain. #[expect( clippy::cast_precision_loss, reason = "the matrix's u32 column index type bounds the square row domain, and epoch \ @@ -298,7 +325,7 @@ where /// Applies the symmetric attraction update of four edges. /// - /// Edges sharing a vertex within one batch see the batch's entry coordinates; their updates + /// Edges sharing a vertex within one batch see the batch's entry coordinates. Their updates /// accumulate. fn attract_x4(&mut self, edges: [E; 4], learning_rate: f32) { let heads = edges.map(|edge| self.schedule.heads[edge]); @@ -331,8 +358,8 @@ where /// Repels the anchor from `negative_sample_rate` drawn vertices. /// /// Draws apply in chunks of four against the anchor's position at chunk entry, with a scalar - /// remainder. A draw of the anchor itself is a coincident pair and contributes no gradient, so - /// the loop keeps every draw. + /// remainder. Every draw remains in the schedule, including a draw of the anchor itself, whose + /// coincident-pair gradient is zero. fn repel(&mut self, anchor: N, learning_rate: f32) { let mut remaining = self.options.negative_sample_rate.get(); diff --git a/libs/@local/graph/atlas/src/salt/landmark/mod.rs b/libs/@local/graph/atlas/src/salt/landmark/mod.rs index 47dad50421a..2353dbb0ec8 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/mod.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/mod.rs @@ -1,31 +1,34 @@ -//! The bounded landmark skeleton. +//! Bounded nonlinear layout over selected landmark rows. //! -//! The skeleton caps the nonlinear layout problem at a configured landmark count `M` independent of -//! the corpus size `N`: +//! A [`LandmarkSkeleton`](artifact::LandmarkSkeleton) combines selected rows, an assignment of +//! every input row to a selected landmark, and the landmarks' 2D coordinates. A capacity `M` bounds +//! the nonlinear layout independently of the input row count `N`: //! -//! 1. [`select_landmarks`](select::select_landmarks) draws `M` representative node rows by weighted -//! sampling without replacement, honoring subgroup minimums and a retained fraction of the prior -//! generation's landmarks. -//! 2. [`LandmarkSelection::assign`](select::LandmarkSelection::assign) maps every corpus row to its -//! nearest selected landmark through the generation's search backend, and a landmark assigns to -//! itself. +//! 1. [`select_landmarks`](select::select_landmarks) selects at most `M` rows by weighted +//! priorities. Subgroup minimums take precedence over a retention target for prior landmarks, +//! followed by a fill to capacity. Minimums use a greedy procedure that can reject jointly +//! feasible overlapping requirements. +//! 2. [`LandmarkSelection::assign`](select::LandmarkSelection::assign) maps every input row to a +//! selected landmark through a nearest-neighbour backend. Landmarks assign to themselves. Other +//! assignments inherit the backend's approximation quality. //! 3. [`LandmarkAssignment::quotient`](assignment::LandmarkAssignment::quotient) contracts the -//! corpus [`SemanticGraph`](super::semantic::SemanticGraph) through the assignment into a -//! semantic graph over the landmark domain: the structure the nonlinear layout optimizes over, -//! `M x M` instead of `N x N`. -//! 4. [`layout_landmarks`](layout::layout_landmarks) places the landmarks in 2D by stochastic -//! gradient descent of the UMAP objective over the quotient graph, on the -//! [`AffinityCurve`](crate::math::AffinityCurve) gradient kernels. +//! input [`SemanticGraph`](super::semantic::SemanticGraph) through this assignment. The quotient +//! has the same fuzzy-weight semantics over at most `M` rows, keeping the layout graph bounded +//! while assignment still covers all `N` rows. +//! 4. [`layout_landmarks`](layout::layout_landmarks) places landmarks by UMAP-style +//! negative-sampling updates using [`AffinityCurve`](crate::math::AffinityCurve) kernels. Its +//! asymmetric negative updates do not generally descend one scalar objective for the whole +//! graph. //! -//! A fitted skeleton publishes as one combined landmark file ([`artifact`]): selection, assignment, -//! and coordinates share the ordinal vocabulary, so they live in one artifact and cannot fall out -//! of sync. +//! Selection, assignment and coordinates share [`LandmarkOrdinal`](select::LandmarkOrdinal) +//! positions and publish in one combined landmark file ([`artifact`]). Assembly checks their +//! landmark counts and coordinate finiteness. Supply parts derived from the same selection, since +//! equal counts alone do not establish that relationship. //! -//! The quotient is a [`SemanticGraph`] like the corpus graph it contracts, so the layout consumes -//! one graph type at either scale. Every stage draws its randomness from a caller-seeded generator; -//! rerunning with equal inputs, options, and seed reproduces the skeleton exactly. -//! -//! [`SemanticGraph`]: super::semantic::SemanticGraph +//! Selection and contraction preserve their operation order across thread counts. Assignment +//! requires a deterministic backend for reproducible results. The serial layout repeats under equal +//! inputs and random streams with the same floating-point behavior, without a cross-platform +//! bit-equality guarantee. pub(crate) mod artifact; pub(crate) mod assignment; pub(crate) mod layout; diff --git a/libs/@local/graph/atlas/src/salt/landmark/quotient.rs b/libs/@local/graph/atlas/src/salt/landmark/quotient.rs index 1518b88d049..03430fa0804 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/quotient.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/quotient.rs @@ -1,28 +1,29 @@ //! Quotient contraction of the semantic graph. //! -//! The corpus semantic graph contracts through the nearest-landmark assignment. Every corpus edge -//! whose endpoints map to distinct landmarks contributes its weight to the directed landmark pair, -//! in double precision. Each landmark row then normalizes by its largest inflow and keeps its -//! strongest [`maximum_neighbours`](QuotientOptions::maximum_neighbours). Both directions of a pair -//! combine by the probabilistic union `a + b - a · b`, the same symmetrization the corpus-scale -//! [`SemanticGraph`](crate::salt::semantic) uses. The result is a symmetric graph over the landmark -//! domain with weights in `(0, 1]` and optimization memory proportional to the landmark count. +//! Let A(i) assign input row i to one of M landmarks, and let wᵢⱼ ∈ (0, 1] be a stored input edge +//! weight. For distinct landmarks a and b, the directed flow is F(a, b) = Σ_{i: A(i) = a} Σ_{j: +//! A(j) = b} wᵢⱼ, summing only stored edges. Edges inside one landmark contribute nothing. Each row +//! normalizes by its largest flow, `p(a, b) = F(a, b) / max_c F(a, c)`, and keeps its strongest +//! [`maximum_neighbours`](QuotientOptions::maximum_neighbours) entries, breaking ties by ordinal. +//! Missing and discarded directions have p = 0. //! -//! The union treats the two directions as fuzzy memberships rather than as two measurements of one -//! quantity. Each direction is the pair's flow normalized by a *different* denominator (its own -//! row's largest inflow), so the two values carry mismatched scales. A plain sum double-counts the -//! shared corpus edges and a difference reports asymmetry instead of affinity. The probabilistic -//! union keeps a one-sided edge at its directed value (a hub's weak judgement of a satellite never -//! erases the satellite's strong judgement of the hub) and reinforces edges both sides claim. -//! Corpus scale behaves the same way, so the layout optimizer sees one weight semantics at either -//! scale. +//! The quotient weight is q(a, b) = p(a, b) + p(b, a) − p(a, b) · p(b, a), the probabilistic union +//! also used by [`SemanticGraph`]. Row-specific maxima put +//! the directions on different scales. Union preserves either direction's support and never lowers +//! its real-arithmetic membership. In particular, a landmark's weak normalized flow to another +//! never erases that other's strong flow back. This preserves fuzzy-membership semantics at both +//! graph scales instead of summing the shared corpus edges twice. //! -//! The corpus graph stores every edge in both rows, so each undirected corpus edge feeds both -//! directions of its landmark pair; the per-row normalization is what keeps the contraction from -//! being a plain doubling. +//! A successful quotient is symmetric with weights in (0, 1]. Mirroring retained directions can +//! give one row more neighbours than its directed cap. With cap K, the result has at most 2 · M · K +//! stored directed entries. Contraction also needs the corpus-to-landmark grouping and dense +//! landmark-domain scratch columns. //! -//! The contraction accumulates in parallel, one task per landmark over that landmark's corpus rows -//! in ascending order, so the sums are bit-equal to a serial pass at any thread count. +//! Accumulation uses `f64`. Every landmark's task visits its assigned corpus rows and their edges +//! in ascending order, preserving the per-pair addition order of a serial pass at any thread count. +//! Normalized weights narrow to `f32`, and equal-weight ordering makes both mirrored unions +//! bit-equal under the same floating-point behavior. Extreme flow ratios can underflow to zero on +//! narrowing, in which case final graph validation fails. use core::{error::Error, fmt, num::NonZero}; @@ -34,13 +35,14 @@ use crate::salt::semantic::{ SemanticGraph, SemanticGraphView, SemanticMatrix, SemanticValidationError, }; +/// The default per-landmark directed-edge cap. const MAXIMUM_NEIGHBOURS: NonZero = const { NonZero::new(64).unwrap() }; /// Contraction settings. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct QuotientOptions { - /// Strongest directed edges each landmark row keeps before the symmetric union. - // The default is an unvalidated starting point (legacy required the value as config, setting no precedent). It bounds quotient memory at about `M · 64` directed edges before the union, and the layout quality criteria (trustworthiness, landmark rank correlation) revise it from evidence. + /// Strongest directed edges each landmark row keeps before union, 64 by default. + // the unvalidated default bounds retained directions at M · 64 before union. Trustworthiness and landmark rank correlation supply the measurements for revising it. pub maximum_neighbours: NonZero = MAXIMUM_NEIGHBOURS, } @@ -97,11 +99,15 @@ where { /// Accumulates each landmark's directed inflows and keeps its strongest normalized neighbours. /// - /// One task per landmark accumulates into a dense per-thread scratch column - the touched list - /// records which slots to read and reset, so a task costs its own edges, not the landmark - /// domain. Rows ascend within each task ([`runs`](Self::runs) yields ascending runs), so the - /// per-pair addition order (and the sum, bit for bit) matches a serial pass at any thread - /// count. + /// Each task state has a dense landmark-domain scratch column. The touched list limits + /// extraction and reset to nonzero flows after allocation. [`runs`](Self::runs) preserves + /// ascending corpus-row order within each landmark, matching a serial pass's per-pair + /// additions. + /// + /// # Panics + /// + /// This panics when a visited row lies outside the graph or a neighbour lies outside the + /// assignment. fn strongest_neighbours( &self, semantic: &SemanticGraphView<'_, N>, @@ -177,8 +183,8 @@ where /// /// # Errors /// - /// Returns an error when this assignment does not cover the graph's rows or when no edge - /// crosses landmark boundaries. + /// Returns [`QuotientError`] for inconsistent row domains, an edgeless quotient, or a + /// contracted matrix that fails graph validation. #[tracing::instrument(skip_all)] pub(crate) fn quotient( &self, @@ -211,10 +217,10 @@ where return Err(QuotientError::EmptyQuotient); } - // Weight-descending within a key, so a pair's two mirror - // positions fold in one order and compute bit-equal unions. Each - // key folds at most two memberships in (0, 1], whose union is ≤ 1, - // and the clamp is the unit fraction's ceiling. + // A pair has at most one retained membership from each direction. Descending weight order + // makes its two mirrored positions fold those same operands in the same order. Their unions + // are bit-equal, and the clamp enforces the upper bound of one. Final validation rejects a + // zero from earlier underflow. edges.sort_unstable_by( |&(row_a, column_a, weight_a), &(row_b, column_b, weight_b)| { (row_a, column_a) @@ -248,10 +254,8 @@ where indptr.push(indices.len() as u64); } - // The sort, dedup and fill above give the pairs their compressed-sparse-row shape: - // ascending and unique by (row, column), every column a landmark ordinal, `indptr` - // monotone through the entry count. Dropping any one of the three breaks that - // shape. + // the sorted, deduplicated pairs give ascending unique columns in every row. The fill + // starts indptr at zero and closes it at the entry count, including empty landmark rows. let matrix = SemanticMatrix::try_new((landmarks, landmarks), indptr, indices, weights) .map_err(|(_, _, _, error)| error) .expect("mirrored sorted pairs form a compressed sparse row structure"); diff --git a/libs/@local/graph/atlas/src/salt/landmark/select.rs b/libs/@local/graph/atlas/src/salt/landmark/select.rs index a3545355264..7a00aca8614 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/select.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/select.rs @@ -1,22 +1,28 @@ //! Weighted stratified landmark selection. //! -//! Selection uses weighted sampling without replacement by exponential clocks: candidate `i` -//! receives the priority +//! For candidate `i`, let wᵢ > 0 be its sampling weight and Uᵢ a uniform draw in (0, 1]. The +//! exponential-clock model assigns tᵢ = −ln(Uᵢ) / wᵢ and selects the smallest priorities without +//! replacement. In continuous arithmetic each clock has rate wᵢ. The implementation uses discrete +//! `f64` draws and rounded logarithms and division. Extreme weights can overflow priorities to +//! infinity or underflow them to zero. Candidate index breaks every priority tie. //! -//! ```text -//! t_i = -ln(U_i) / w_i, -//! ``` +//! One shared priority set serves the subgroup minimums, prior-landmark retention, and the free +//! fill. Minimums run in [`Subgroup`] order. Earlier selections count toward every later subgroup +//! they belong to, and later phases never evict them. Every successful selection therefore +//! satisfies all minimums. This greedy procedure can exhaust capacity even when another set could +//! satisfy overlapping minimums. //! -//! `U_i` drawn uniformly from `(0, 1]` and `w_i` the candidate's sampling weight, and the smallest -//! priorities win. Selection runs in three phases over one shared set of priorities: subgroup -//! minimums first, then prior landmarks up to the retained target, then a free fill to capacity. -//! Later phases never evict earlier picks, so every minimum still holds in the final selection. +//! Retention targets the ceiling of the computed f64 product C · f, with C = min(`maximum_count`, +//! candidate count) and f = `retained_fraction`. The u32 capacity limit keeps C exactly +//! representable in f64. Multiplication can round before the ceiling, giving a target different +//! from ceil(C · f) in real arithmetic. Prior rows already selected for minimums count toward the +//! target. Remaining capacity limits additional retention, and the free fill can select more prior +//! rows than the target. //! -//! The corpus-scale passes run in parallel and deterministically. Priorities come from one -//! generator per fixed-size candidate chunk, each seeded by the caller's generator, and every phase -//! reduces thread-local top-`k` heaps into the unique best set under the (priority, index) total -//! order. A rerun over equal candidates with an equally seeded generator selects identical rows at -//! any thread count. +//! Priorities come from one seeded generator per fixed-size candidate chunk. Each phase collects +//! eligible candidates in parallel and selects the unique best set under the `(priority, index)` +//! total order. Equal candidates, options and generator streams select identical rows at any thread +//! count with the same floating-point behavior. The returned rows ascend by row id. #![expect(clippy::empty_enums, reason = "zerocopy uses them in the derive")] use core::{ @@ -137,8 +143,9 @@ pub(crate) struct SubgroupMinimum { #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct LandmarkCandidate { pub row: N, - /// The candidate's relative selection propensity. One is the neutral weight, giving every - /// row equal likelihood. + /// Relative sampling weight. + /// + /// Equal weights give equal priority distributions before subgroup and retention constraints. pub sampling_weight: DPositive, /// The candidate's value on every stratification axis. pub axes: SubgroupAxes, @@ -159,9 +166,9 @@ impl LandmarkCandidate { pub(crate) struct SelectionOptions { /// The landmark capacity `M`. /// - /// Selection returns at most this many rows, fewer only when the corpus is smaller. The `u32` width is the [`LandmarkOrdinal`] encoding's contract: every selection position fits the persisted ordinal form. + /// Set this explicitly, as it has no default. Selection returns at most this many rows, fewer only when the corpus is smaller. Every selection position fits the persisted [`LandmarkOrdinal`] encoding. pub maximum_count: NonZero, - /// Fraction of the capacity reserved for prior landmarks when enough are on offer. + /// Target fraction of prior landmarks, 0.25 by default. /// /// Retention stabilizes generation-to-generation orientation. // The default is an unvalidated starting point; the temporal-drift @@ -169,14 +176,14 @@ pub(crate) struct SelectionOptions { pub retained_fraction: UnitFraction = const { UnitFraction::new(0.25).unwrap() }, /// Candidates per generator stream: the priority pass's seeding and parallel work unit. /// - /// This value fixes which stream draws for which candidate, so equal seeds reproduce equal selections only under an equal chunk. The manifest echo records the chunk beside the seed. The chunk holds enough candidates that per-task overhead disappears against the scans. + /// By default, uses 4,096 candidates. This value fixes which stream draws for each candidate. Equal-seed replay requires the same chunk size. pub parallel_chunk: NonZero = PARALLEL_CHUNK, } hashql_core::id::newtype! { /// A reference to a landmark by its position in a [`LandmarkSelection`]. /// - /// Ordinals are dense and zero-based: the value is the position of the landmark's node row in the selection's ascending row order. The little-endian representation is the persisted form, so a column of these ordinals moves to and from artifact files without conversion. + /// Ordinals are dense and zero-based: each value is the position of a selected row in ascending row order. Its little-endian bytes are also its persisted representation. #[id(endian = little, unaligned, derive(Step), const)] pub(crate) struct LandmarkOrdinal(u32) } @@ -215,14 +222,12 @@ where /// Maps every selected row through `map`, preserving ordinals and the retained count. /// - /// The selection's vocabulary is positional, where ordinal `i` names the `i`-th selected row. A - /// row translation therefore composes without touching the assignment or the layout built - /// against it. + /// `map` must preserve strictly ascending row order. Ordinal `i` continues to name the `i`-th + /// selected row, preserving assignments and coordinates indexed by those ordinals. /// /// # Panics /// - /// This panics when the mapped rows break the strictly ascending row order. A strictly - /// increasing `map` preserves it. + /// This panics when the mapped rows decrease. #[must_use] pub(crate) fn map_rows(&self, map: impl FnMut(N) -> M) -> LandmarkSelection where @@ -251,7 +256,7 @@ where } } -/// The selection inputs are unsatisfiable or malformed. +/// A malformed input or a constraint the greedy selection cannot satisfy. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum SelectionError { /// The corpus offers no candidates. @@ -260,7 +265,7 @@ pub(crate) enum SelectionError { UnorderedCandidates { index: usize }, /// A subgroup carries more than one minimum. DuplicateMinimum { subgroup: Subgroup }, - /// The minimums together demand more rows than the capacity. + /// Earlier picks plus the next minimum's unmet count exceed the capacity. MinimumExceedsCapacity { requested: usize, capacity: usize }, /// A subgroup offers fewer candidates than its minimum demands. InsufficientSubgroup { @@ -312,7 +317,6 @@ hashql_core::id::newtype! { hashql_core::id::newtype! { /// A minimum's position in the subgroup-ordered minimums. /// - /// Candidates and minimums index unrelated domains, so mixing a [`MinimumId`] with a [`CandidateId`] is a type error rather than a silent off-by-everything. #[id(derive(Step))] pub(crate) struct MinimumId(u64) } @@ -379,7 +383,8 @@ where let mut rng = R::seed_from_u64(seed); for (priority, candidate) in priorities.iter_mut().zip(candidates) { - // 1 - U maps the generator's [0, 1) onto (0, 1], keeping the logarithm finite. + // 1 − U maps the generator's [0, 1) onto (0, 1], keeping the logarithm finite. + // Division by an extreme weight can still overflow. *priority = -(1.0 - rng.random::()).ln() / candidate.sampling_weight; } }); @@ -387,6 +392,12 @@ where priorities } +/// Checks the selection inputs before any sampling. +/// +/// # Errors +/// +/// Returns [`SelectionError`] for empty or unordered candidates or duplicate minimums, checked in +/// that order. An ordering error names the first row that does not exceed its predecessor. fn validate( candidates: &IdSlice>, minimums: &IdSlice, @@ -420,7 +431,10 @@ where Ok(()) } -/// Returns `ceil(capacity · retained_fraction)`. +/// Returns the ceiling of the computed f64 retention product. +/// +/// A capacity at most `u32::MAX` converts exactly to f64. The product with `retained_fraction` can +/// still round before ceil. The result converts to usize with saturation. #[expect( clippy::cast_possible_truncation, clippy::cast_precision_loss, @@ -433,12 +447,16 @@ const fn retained_target(capacity: usize, retained_fraction: UnitFraction) -> us (capacity as f64 * retained_fraction).ceil() as usize } -/// Returns the indices of the `count` smallest-priority unselected candidates. +/// Replaces `output` with up to `count` smallest-priority unselected candidate indices. /// -/// Only candidates satisfying `predicate` qualify, and fewer return when the pool is smaller. +/// Only candidates satisfying `predicate` qualify. Output order is unspecified, but the selected +/// set is unique under the `(priority, index)` total order, independent of how the parallel scan +/// splits. /// -/// Workers filter in parallel and one exact selection cuts the (priority, index) total order at -/// `count`: the result is the unique best set, independent of how the scan splits across threads. +/// # Panics +/// +/// For a nonzero `count`, this panics when `selected` does not cover the candidate domain or an +/// eligible candidate has no entry in `priorities`. fn best_indices( candidates: &IdSlice>, priorities: &IdSlice, @@ -473,7 +491,11 @@ fn best_indices( output.extend(ranked.into_iter().map(|ranked| ranked.id)); } -/// Inserts the chosen indices and returns how many carried the prior landmark flag. +/// Marks the chosen indices and counts their prior-landmark flags. +/// +/// # Panics +/// +/// This panics when an index lies outside `candidates` or the selected-set domain. fn mark( selected: &mut DenseBitSet, candidates: &IdSlice>, @@ -495,21 +517,26 @@ fn mark( /// Candidates per parallel work item. /// -/// 4096 candidates give each task tens of microseconds of work, large enough that per-task overhead -/// disappears against the scans. +/// A 4,096-candidate chunk amortizes task setup across a block of candidates. Its size is part of +/// the seeded selection's replay inputs. pub(crate) const PARALLEL_CHUNK: NonZero = const { NonZero::new(4096).unwrap() }; -/// Selects at most the configured capacity, honoring minimums and retention. +/// Selects weighted landmark rows subject to greedy minimums and a retention target. +/// +/// `candidates` must have strictly ascending row ids. On success, the selection satisfies every +/// subgroup minimum and fills min(`maximum_count`, candidate count) positions. Minimums take +/// precedence over retention, as described in the [selection model](super::select). +/// +/// # Complexity /// -/// `candidates` arrive in strictly ascending row order, and `rng` seeds the priority streams. The -/// selection satisfies every subgroup minimum, then retains prior landmarks up to `ceil(capacity * -/// retained_fraction)` when enough are on offer, then fills to capacity, all by ascending -/// exponential-clock priority. +/// Every minimum, the retention phase and the free fill scan the candidate domain. Priorities and +/// eligible-candidate storage require O(N) space for N candidates, even when the selected capacity +/// is small. Updating overlapping minimum counts additionally visits later minimums for each pick. /// /// # Errors /// -/// Returns an error for an empty corpus, unordered candidate rows, duplicate minimums, or minimums -/// the corpus or capacity cannot satisfy. +/// Returns [`SelectionError`] for malformed inputs or a minimum that the greedy procedure cannot +/// satisfy. Capacity failure describes the current picks, not global infeasibility. #[tracing::instrument(skip_all)] pub(crate) fn select_landmarks( candidates: &IdSlice>, diff --git a/libs/@local/graph/atlas/src/salt/landmark/tests.rs b/libs/@local/graph/atlas/src/salt/landmark/tests.rs index e9b602b4430..03d858dd2fd 100644 --- a/libs/@local/graph/atlas/src/salt/landmark/tests.rs +++ b/libs/@local/graph/atlas/src/salt/landmark/tests.rs @@ -36,11 +36,16 @@ use crate::{ }, }; +/// Creates the seed-42 generator shared by deterministic fixtures. fn rng() -> Xoshiro256PlusPlus { Xoshiro256PlusPlus::seed_from_u64(42) } /// Runs the task in a dedicated rayon pool of `threads` workers. +/// +/// # Panics +/// +/// This panics when the pool cannot create its workers. fn in_pool(threads: usize, task: impl FnOnce() -> T + Send) -> T { ThreadPoolBuilder::new() .num_threads(threads) @@ -49,6 +54,7 @@ fn in_pool(threads: usize, task: impl FnOnce() -> T + Send) -> T { .install(task) } +/// Creates a unit-weight, non-prior candidate at `row` with zero-valued axes. fn candidate(row: u64) -> LandmarkCandidate { LandmarkCandidate { row: NodeRowId::new(row), @@ -58,10 +64,16 @@ fn candidate(row: u64) -> LandmarkCandidate { } } +/// Creates unit-weight candidates at every row in `0..count`. fn candidates(count: u64) -> Vec> { (0..count).map(candidate).collect() } +/// Sets `maximum` as the landmark capacity, retaining the remaining defaults. +/// +/// # Panics +/// +/// This panics when `maximum` is zero. fn options(maximum: u32) -> SelectionOptions { SelectionOptions { maximum_count: NonZero::new(maximum).expect("test capacities are nonzero"), @@ -69,7 +81,11 @@ fn options(maximum: u32) -> SelectionOptions { } } -/// Calls [`select_landmarks`] over plain fixture slices. +/// Selects landmarks from plain fixture slices. +/// +/// # Errors +/// +/// Returns [`SelectionError`] when [`select_landmarks`] rejects the fixture. fn select( candidates: &[LandmarkCandidate], minimums: &[SubgroupMinimum], @@ -281,10 +297,8 @@ fn selection_rejects_unsatisfiable_minimums() { #[test] fn selection_counts_rows_toward_every_minimum_they_satisfy() { - // Rows 0..4 carry both marked axes, so the same three rows can - // satisfy both three-row minimums at once; a capacity of three - // suffices. Counting the overlap per minimum would demand six rows - // and fail with MinimumExceedsCapacity. + // rows 0..4 carry both marked axes. Selecting any three satisfies both three-row minimums in a + // capacity of three. let mut candidates = candidates(50); for candidate in &mut candidates[..4] { candidate.axes[SubgroupDimension::Language] = 7; @@ -347,11 +361,15 @@ fn selection_is_invariant_across_thread_counts() { } /// A brute-force cosine backend over resident rows. +/// +/// Id searches panic when the row is absent, or when `limit` is [`usize::MAX`] and overflow checks +/// are enabled. struct ExactIndex { rows: Vec<(NodeRowId, BoxedVecN)>, } impl ExactIndex { + /// Creates an empty fixture index. fn new() -> Self { Self { rows: Vec::new() } } @@ -432,7 +450,7 @@ impl NearestNeighboursIndex for ExactIndex { } } -/// A fixture of six rows in three well-separated directions. +/// Creates six normalized rows in three separated directions. /// /// Rows 0 and 1 point one way, 2 and 3 another, 4 and 5 a third. fn clustered_embeddings() -> Vec> { @@ -463,6 +481,11 @@ struct Matrix { } impl Matrix { + /// Copies `rows` into the head of the aligned storage. + /// + /// # Panics + /// + /// Panics when more than eight rows are given, the fixture's capacity. fn new(rows: &[BoxedVecN]) -> Self { let mut storage = BoxedVecN::zero(); let (chunks, _) = storage @@ -478,6 +501,7 @@ impl Matrix { } } + /// Borrows the initialized rows as an aligned row-indexed slice. fn view(&self) -> &IdSlice> { IdSlice::from_raw( AlignedVecN::from_slice(&self.storage.as_array()[..self.rows * PROJECTOR_DIMENSIONS]) @@ -486,6 +510,11 @@ impl Matrix { } } +/// Selects exactly `rows` by setting capacity to the candidate count. +/// +/// # Panics +/// +/// This panics when `rows` is empty, not strictly ascending, or longer than the `u32` capacity. fn selection_of(rows: &[u64]) -> super::select::LandmarkSelection { let candidates: Vec> = rows.iter().map(|&row| candidate(row)).collect(); @@ -493,7 +522,7 @@ fn selection_of(rows: &[u64]) -> super::select::LandmarkSelection { select(&candidates, &[], options(capacity), rng()).expect("selecting every candidate succeeds") } -/// Ordinals from bare positions, for fixtures. +/// Converts literal positions into landmark ordinals. fn ordinals(positions: &[u32]) -> Box<[LandmarkOrdinal]> { positions .iter() @@ -536,12 +565,21 @@ fn assignment_rejects_landmarks_outside_the_corpus() { ); } -/// An assignment straight from ordinals, for quotient fixtures. +/// Builds a fixture assignment from literal ordinals. +/// +/// # Panics +/// +/// This panics when an ordinal lies outside `landmarks`. fn assignment_of(positions: &[u32], landmarks: usize) -> LandmarkAssignment { LandmarkAssignment::from_ordinals(IdSlice::from_boxed_slice(ordinals(positions)), landmarks) } -/// A semantic graph over `count` rows from undirected weighted edges. +/// Builds a semantic graph over `count` rows from undirected weighted edges. +/// +/// # Panics +/// +/// This panics for an out-of-domain endpoint, repeated edge, self edge, a weight outside finite (0, +/// 1], or fewer than two rows. fn semantic_from_edges(count: usize, edges: &[(u32, u32, f32)]) -> SemanticGraph { let mut rows: Vec> = vec![Vec::new(); count]; for &(left, right, weight) in edges { @@ -564,7 +602,7 @@ fn semantic_from_edges(count: usize, edges: &[(u32, u32, f32)]) -> SemanticGraph SemanticGraph::new(matrix).expect("the fixture satisfies the graph invariants") } -/// A corpus semantic graph over six rows. +/// Builds a six-row graph with strong clusters and weaker inter-cluster edges. /// /// Edges within clusters have weight 1.0, one bridge edge (1, 2) has weight 0.5, and one weaker /// bridge (3, 4) has weight 0.25. @@ -590,9 +628,9 @@ fn quotient_contracts_cross_landmark_edges() { .quotient(&graph.view(), QuotientOptions { .. }) .expect("the fixture quotient has edges"); - // The directed inflows are L0 <- 0.5 (edge 1-2), L1 <- 0.5 + 0.25, and L2 <- 0.25. Every row - // max-normalizes, then pairs combine by the probabilistic union: (L0, L1) = 1 + 1 - 1 = 1.0 and - // (L1, L2) = 0.5 + 1 - 0.5 = 1.0. + // cross-landmark flows are F(0, 1) = F(1, 0) = 0.5 and F(1, 2) = F(2, 1) = 0.25. Row maxima are + // 0.5, 0.5 and 0.25. The union gives q(0, 1) = 1 + 1 − 1 · 1 = 1 and q(1, 2) = 0.5 + 1 − 0.5 · + // 1 = 1. let view = quotient.view(); assert_eq!(view.rows(), 3); let row0: Vec<(u64, f64)> = view @@ -631,9 +669,8 @@ fn quotient_keeps_only_the_strongest_neighbours() { ) .expect("the fixture quotient has edges"); - // L0 keeps only L1; L2 and L3 keep their single inflow (normalized - // to 1.0 within their own rows), so their edges to L0 survive from - // the other direction. + // L0 keeps only L1 before mirroring. L2 and L3 each normalize their single flow to 1.0, + // preserving the remaining edges to L0 from the other direction. assert!( quotient .view() @@ -704,11 +741,21 @@ fn quotient_is_invariant_across_thread_counts() { ); } +/// Fits the fixture affinity curve at spread `1.0` and floor `0.1`. +/// +/// # Panics +/// +/// This panics if fitting the fixed reference parameters fails. fn curve() -> AffinityCurve { AffinityCurve::fit(positive!(1.0), positive!(0.1)) .expect("the reference inputs are well-conditioned") } +/// Sets `epochs` as the layout budget, retaining the remaining defaults. +/// +/// # Panics +/// +/// This panics when `epochs` is zero. fn layout_options(epochs: u32) -> LayoutOptions { LayoutOptions { epochs: NonZero::new(epochs).expect("test epoch budgets are nonzero"), @@ -736,7 +783,7 @@ fn layout_is_deterministic_under_a_seed() { assert_ne!(first, third, "a different seed draws a different layout"); } -/// Two 3-cliques with no edge between them. +/// Builds two disjoint 3-cliques. fn clique_pair() -> SemanticGraph { semantic_from_edges( 6, @@ -754,7 +801,11 @@ fn clique_pair() -> SemanticGraph { /// The clique-pair vertex groups. const CLIQUES: [&[usize]; 2] = [&[0, 1, 2], &[3, 4, 5]]; -/// Longest pairwise distance inside either clique. +/// Finds the longest pairwise distance inside either fixture clique. +/// +/// # Panics +/// +/// This panics when `coordinates` has fewer than six rows. fn widest_within(coordinates: &IdSlice) -> f32 where N: Id, @@ -774,7 +825,11 @@ where widest } -/// Shortest distance between the two cliques. +/// Finds the shortest distance across the fixture cliques. +/// +/// # Panics +/// +/// This panics when `coordinates` has fewer than six rows. fn narrowest_across(coordinates: &IdSlice) -> f32 where N: Id, @@ -817,8 +872,7 @@ fn layout_separates_clusters() { #[test] fn repulsion_widens_the_gap_between_disconnected_components() { - // Same graph, same seed, one knob: with repulsion off the cliques only contract in place, so - // the gap the default schedule opens must exceed the unrepelled one. + // equal graph, seed and epoch budget isolate the effect of repulsion in this fixture. let graph = clique_pair(); let repelled = layout_landmarks(&graph.view(), curve(), layout_options(200), rng()) .expect("the fixture graph lays out"); @@ -844,9 +898,9 @@ fn repulsion_widens_the_gap_between_disconnected_components() { #[test] fn layout_drops_edges_weaker_than_the_epoch_budget() { - // The (2, 3) weight needs a hundred epochs between samples, beyond the fifty-epoch budget. The - // schedule never samples the pair, so it gets no attraction and keeps its initial separation - // while the full-weight pair gathers. + // edge (2, 3) has period 1/0.01 ≈ 100, beyond the last deadline 49. Neither endpoint has + // another edge. They never anchor an update, and negative draws move only the anchor, + // preserving this pair's initialization. let graph = semantic_from_edges(4, &[(0, 1, 1.0), (2, 3, 0.01)]); let coordinates = layout_landmarks(&graph.view(), curve(), layout_options(50), rng()) @@ -871,9 +925,9 @@ fn layout_leaves_edgeless_rows_on_the_initial_circle() { .expect("the fixture graph lays out"); assert_eq!(coordinates.len(), 4); - // The initial circle has radius 5 with up to 1% radial jitter; no - // force acts on an edgeless row, so it stays in that annulus (the - // bounds carry rounding slop from the trigonometric placement). + // initial radii lie between 5 and 5 · 1.01 = 5.05 before rounding. Edgeless rows never anchor + // updates, and repulsion moves only the anchor. The assertion includes rounding tolerance for + // trigonometric placement. for &isolated in &[2_usize, 3] { let radius = coordinates[NodeRowId::from_usize(isolated)].length(); assert!( @@ -882,8 +936,7 @@ fn layout_leaves_edgeless_rows_on_the_initial_circle() { ); } - // The connected pair starts half a circle apart (distance ~10) and - // attraction draws it in. + // rows 0 and 1 start a quarter turn apart: base separation 5√2 ≈ 7.07 before jitter. assert!( coordinates[NodeRowId::new(0)].distance(coordinates[NodeRowId::new(1)]) < 2.0, "the connected pair gathers", @@ -899,7 +952,11 @@ fn layout_rejects_an_edgeless_graph() { ); } -/// A per-test scratch file path under the system temp directory. +/// Creates the fixture directory and returns a per-test scratch path. +/// +/// # Panics +/// +/// This panics when the directory cannot be created. fn scratch(name: &str) -> PathBuf { let dir = std::env::temp_dir().join(format!( "hash-graph-atlas-landmark-skeleton-{}", @@ -909,9 +966,11 @@ fn scratch(name: &str) -> PathBuf { dir.join(name) } -/// A skeleton from real stage outputs over the clustered fixture. +/// Builds a skeleton and its assignment and coordinates from the clustered fixture. +/// +/// # Panics /// -/// The tuple also includes the stage outputs that produced the skeleton. +/// This panics when fixture selection, assignment, contraction or layout fails. fn fixture_skeleton() -> ( LandmarkSkeleton, LandmarkAssignment, @@ -969,8 +1028,9 @@ fn mapped_skeleton_rejects_violated_invariants() { .write_into(&mut bytes) .expect("writing into a vector cannot fail"); - // The fixture's geometry has rows at 4096, assignment at 8192, and coordinates at 12288, with - // each region padded to one 4096 unit. + // after the 4,096-byte header, three u64 selected rows occupy 24 bytes padded to 4,096. Six u32 + // assignments occupy another 24 bytes padded to 4,096. Coordinates start at 12,288 and occupy + // 24 final, unpadded bytes. let open = |name: &str, bytes: &[u8]| { let path = scratch(name); fs::write(&path, bytes).expect("the scratch file is writable"); diff --git a/libs/@local/graph/atlas/src/salt/lod/bench.rs b/libs/@local/graph/atlas/src/salt/lod/bench.rs index aa2c49abe34..d2854b3b248 100644 --- a/libs/@local/graph/atlas/src/salt/lod/bench.rs +++ b/libs/@local/graph/atlas/src/salt/lod/bench.rs @@ -1,43 +1,48 @@ -//! Measurement hooks for the restricted-view backfill walk. +//! Delivery probes for comparing masked backfill and visible-cell selection. //! -//! A masked tile delivery fills its budget by pulling visible points up from deeper importance -//! buckets. Two candidate-selection strategies produce the same response shape and differ only in -//! how they treat points an ancestor tile already pulled up: +//! A masked tile delivery can fill its budget by pulling visible points from deeper importance +//! buckets. [`WalkBench::independent`] walks each tile in isolation and can repeat an ancestor's +//! delivery. [`WalkBench::chained`] recomputes the ancestor chain under the same mask and excludes +//! every position it already delivered. //! -//! - [`WalkBench::independent`] walks the tile's extent in isolation and fills to its budget; -//! re-deliveries across the zoom ladder are the client's to skip. -//! - [`WalkBench::chained`] first re-derives every ancestor's delivery over the same predicate and -//! starts its own fill where the chain left off, so no point crosses the wire twice. +//! The base column is bucket-major, following the cascade's coarse-to-fine assignment. A tile at +//! zoom z with span exponent m schedules bucket z + m inside its extent. The root schedules buckets +//! `0..=m` whole. The unmasked budget is the scheduled count before masking. A bucket walk admits +//! visible, untaken scheduled points, then fills from deeper buckets in bucket order and Morton +//! order within a bucket, stopping at its target or exhaustion. Scheduled admissions can already +//! exceed a coverage-derived target. //! -//! Both variants share one delivery model. The base column is bucket-major (the cascade's -//! coarse-to-fine assignment), a tile at zoom `z` with span exponent `m` schedules bucket `z + m` -//! within its extent (the root schedules buckets `0..=m` whole), and the budget is the scheduled -//! count before masking. A masked walk delivers the scheduled points the predicate admits, then -//! fills the shortfall from buckets below the cut, in bucket order and in morton order within a -//! bucket, until the delivery meets the budget or exhausts the extent. +//! [`FillRule`] chooses the target or a representative-selection rule. [`FillRule::Coverage`] +//! subtracts inherited deliveries from the visible cut-cell count. [`FillRule::CoverageCells`] +//! instead tracks represented cells and fills only uncovered ones. [`VisibleCellPyramid`] supplies +//! these cell counts, while [`WalkBench::visible_cascade`] supplies the visible-only cascade's +//! schedule for comparison. At the cascade's deepest cut, its catch-all can deliver more than one +//! point per occupied cell. [`WalkBench::audit`] separates the target, delivered count and occupied +//! cells. //! -//! [`WalkBench::deliver`] runs that same chain against a [`FillRule`], the count each level fills -//! to: [`FillRule::Unmasked`] is the scheduled count before masking, [`FillRule::Coverage`] the -//! depth-`z + m` cells of the tile cell holding a visible point less the chain's own deliveries -//! inside it, [`FillRule::Visible`] the visible scheduled count alone, and -//! [`FillRule::CoverageCells`] the same cell count with the fill restricted to cells no delivered -//! point occupies. The coverage targets read a [`VisibleCellPyramid`], one cell census per cut -//! depth over the visible view; [`WalkBench::visible_cascade`] runs the production cascade over the -//! visible points alone, the schedule a coverage target claims to reproduce. [`WalkBench::audit`] -//! reports both alongside the cells a chain's delivery actually occupies. +//! The rank-representative rules resolve each unrepresented grid cell to its best-ranked visible +//! point. [`WalkBench::deliver`] scans each selected cell for that point. +//! [`WalkBench::served_deliver`] reads a [`ServedGeneration`]: a visible-only cascade at +//! [`Depth::MAX`], sorted into bucket-major order. Below the catch-all, a depth-d occupied cell has +//! exactly one point in buckets at or below d. Range lengths then count cells without scanning +//! every point. Refinement adds grid-planning work to either engine. //! -//! The rank-representative rules have two engines over one rule. [`WalkBench::deliver`] finds each -//! cell's representative by scanning the cell, `O(points in the extent)` per level. -//! [`WalkBench::served_deliver`] reads it out of a [`ServedGeneration`] instead: the visible view -//! published as its own generation, bucket-major, where a cell of depth `d` holds exactly one point -//! whose bucket lies at or below `d`, so the buckets-at-or-below-`d` ranges of an extent are its -//! depth-`d` representatives and their lengths count its occupied cells. Both engines deliver the -//! same rows in the same order, tile for tile, and the scanning one is the oracle that says so. +//! The scanning engine provides a comparison oracle, but the served engine's refinement counts +//! treat catch-all entries as distinct cells. Exact-key duplicates can make the engines choose +//! different grids and deliveries when refinement reaches [`Depth::MAX`]. Direct representative +//! extraction deduplicates exact keys. Uniform-grid delivery deliberately keeps every catch-all +//! entry for terminal completeness. //! -//! [`WalkBench::build`] synthesizes a clustered corpus and runs the production cascade over it; -//! [`WalkBench::mask_uniform`] hides rows independently and [`WalkBench::mask_clustered`] hides -//! whole spatial blocks, the adversarial shape for walk lengths. Selections return plain counts; -//! wall time belongs to the bench target. Nothing here is API for consumers of the crate. +//! Noninterference comparisons hold visible keys, relative ranks and the schedule fixed. Under +//! these conditions, [`FillRule::CoverageRank`] and constant-budget refinements depend on the +//! visible view alone. [`DotBudget::Scheduled`] reads the unmasked corpus and does not have that +//! property. [`WalkBench::visible_only`] preserves existing keys and ranks rather than fitting the +//! visible rows again. +//! +//! [`WalkBench::build`] synthesizes a clustered corpus. [`WalkBench::mask_uniform`] hides rows by +//! independent draws, while [`WalkBench::mask_clustered`] hides spatial blocks to exercise long +//! fills. The probes return counts and delivered positions through the benchmark facade. Their work +//! counters are engine-specific, and wall time belongs to the benchmark target. use alloc::collections::BinaryHeap; use core::{cmp::Reverse, f64::consts::TAU, num::NonZero, ops::Range}; @@ -62,20 +67,23 @@ use crate::{ /// The extent's positions inside each bucket segment, one range per bucket. type Ranges = [Range; SEGMENTS]; -/// One tile delivery's outcome, as plain counts. +/// One tile's target, delivered counts and engine-specific work count. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub struct Selection { /// The count the fill runs to, which [`FillRule`] chooses. /// /// Under [`FillRule::Unmasked`] it is the scheduled count before masking. pub budget: usize, - /// Scheduled points the predicate admitted. + /// Visible, untaken scheduled points, or all new representatives under a rank rule. pub natural: usize, - /// Points pulled up from deeper buckets to cover the shortfall. + /// Points pulled from deeper buckets, zero under a rank rule. pub tail: usize, - /// Candidate positions examined, hidden and taken ones included. + /// The engine's work count, summed over the chain for chained deliveries. /// - /// [`WalkBench::chained`] sums the whole ancestor chain's scans into this count. + /// Bucket walks count examined positions, including hidden and taken ones. Scanning rank rules + /// count cell ranges visited plus points scanned for representatives. Served rank rules count + /// selected range operations, merge entries and population-search probes. This is not a + /// complete count of memory reads or directly comparable work across engines. pub scanned: usize, } @@ -86,15 +94,15 @@ pub enum FillRule { Unmasked, /// The level cut's cells holding a visible point, less the chain's deliveries inside the cell. /// - /// The target is a function of the visible view and the chain's own output: it reads a - /// [`VisibleCellPyramid`] and the delivered positions, never a hidden row. + /// The target uses [`VisibleCellPyramid`] counts and inherited delivery counts. The inherited + /// output can itself depend on hidden rows through the unmasked bucket assignment. Coverage, /// The level's visible scheduled count, which its own admissions meet. Visible, /// The level cut's cells holding a visible point, less the cells the chain already represents. /// - /// The fill takes a point only where its cut cell holds no delivered point, so the delivery - /// occupies one cell per covered cell. + /// The fill admits only points whose cut cell has no delivered point. Scheduled admissions + /// still deliver whole, including co-located points in the catch-all. CoverageCells, /// One point per level-cut cell holding a visible point, the best-ranked visible point in it. /// @@ -111,10 +119,11 @@ pub enum FillRule { /// A dot budget and the order a refinement spends the remainder of it in. /// -/// [`FillRule::Refined`] refines the whole level while the level's delivered count stays at or -/// below [`budget`](Self::budget), then refines individual cells one level further in -/// [`order`](Self::order) while each one still fits. Both inputs are public constants: a delivery -/// reads the visible view and these two numbers, never a hidden row. +/// [`FillRule::Refined`] tests successively finer whole grids until the next grid exceeds +/// [`budget`](Self::budget), then tries individual cells one level further in +/// [`order`](Self::order). The cut grid is always the minimum resolution, even when its delivery +/// exceeds the budget. Choose both fields explicitly. [`DotBudget::Constant`] fixes the budget +/// independently of the corpus, while [`DotBudget::Scheduled`] derives it from unmasked rows. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub struct Refinement { /// The count one level's own delivery aims at, `K_z`. @@ -129,23 +138,23 @@ pub struct Refinement { /// The count one level of a refinement aims at. /// -/// Both forms are functions of public data (a constant, or the corpus before masking), so neither -/// carries a hidden row into the delivered count. +/// A constant is independent of hidden rows. The scheduled form uses the corpus before masking and +/// can change the delivered count when hidden rows change. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum DotBudget { /// One count for every tile. Constant(usize), - /// The tile's own scheduled count before masking, which is today's per-tile budget. + /// The tile's own scheduled count before masking. /// - /// A tile delivers what today's rule would have delivered had the mask hidden nothing, so a - /// view hiding nothing sees today's density exactly. + /// Refinement can undershoot this count when no finer split fits or exceed it at the cut-depth + /// floor. It does not promise the unmasked rule's density. Scheduled, } /// The order a partial refinement visits one level's cells in. /// -/// Every order is a function of the visible view, so the delivered set stays independent of the -/// hidden rows whichever one a delivery picks. +/// Cell index and visible population determine these visit orders. The order adds no dependence on +/// hidden rows, but a [`DotBudget::Scheduled`] refinement still reads the unmasked corpus. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum RefineOrder { /// Whole levels alone: the delivered grid stays uniform at depth `z + m + k`. @@ -193,7 +202,7 @@ enum FillTarget<'cells> { pub struct ChainAudit { /// The tile's own fill target. pub target: usize, - /// Cells of the tile's cut depth inside the tile cell holding a visible point. + /// Occupied cut-cell count, or all visible entries at a served catch-all cut. pub covered: usize, /// Chain deliveries inside the tile cell, the levels above the tile alone. pub inherited: usize, @@ -215,14 +224,14 @@ pub struct ChainAudit { pub spent: bool, /// Whether the tile's delivery ended below its target. pub dry: bool, - /// Candidate positions examined across the chain. + /// The engine-specific work count across the chain, as in [`Selection::scanned`]. pub scanned: usize, } /// One cell census per delivery cut depth over the visible view. /// -/// Level `d` holds, ascending, every depth-`d` cell index containing at least one visible point; -/// the levels span the cut depths `m` through `z_max + m` a tile schedule reads. +/// Level `d` holds, ascending, every depth-`d` cell index containing at least one visible point. +/// The levels span the cut depths `m` through `z_max + m` a tile schedule reads. /// [`VisibleCellPyramid::count`] answers how many of a cell's depth-`d` cells hold a visible point. /// /// One pyramid describes one `(corpus, mask)` pair: a mask replacement invalidates it. @@ -251,9 +260,9 @@ struct VisiblePoint { /// any depth, and which visible point represents each one. Sixteen bytes per visible row, one /// column for one `(corpus, mask)` pair, invalidated by a mask replacement. /// -/// The rank column carries the corpus-wide importance rank, whose order restricted to the visible -/// rows is the same order a visible-only generation would rank them in, so the representative a -/// cell resolves to does not move when rows outside the view appear or vanish. +/// The rank column retains the corpus-wide ranks. Restricting that order to visible rows preserves +/// each cell's representative when hidden rows disappear, provided the visible keys and relative +/// ranks remain unchanged. A fresh fit or ranking of changed inputs is outside this comparison. #[derive(Debug)] pub struct VisibleColumn { /// Visible points ascending by key. @@ -262,7 +271,7 @@ pub struct VisibleColumn { /// The visible-view artifacts a fill rule's target reads. /// -/// [`FillRule::Coverage`] and [`FillRule::CoverageCells`] read cell counts out of the pyramid; +/// [`FillRule::Coverage`] and [`FillRule::CoverageCells`] read cell counts out of the pyramid. /// [`FillRule::CoverageRank`] and [`FillRule::Refined`] read cells and representatives out of the /// column. Both describe the same `(corpus, mask)` pair. #[derive(Debug, Copy, Clone)] @@ -274,7 +283,9 @@ pub struct VisibleView<'view> { } impl<'view> VisibleView<'view> { - /// Pairs a pyramid with a column over the same visible view. + /// Pairs cell counts and representatives for one visible view. + /// + /// Both artifacts must describe the same corpus and mask. Construction does not compare them. #[must_use] pub const fn new(pyramid: &'view VisibleCellPyramid, column: &'view VisibleColumn) -> Self { Self { pyramid, column } @@ -283,9 +294,9 @@ impl<'view> VisibleView<'view> { /// Where a served generation keeps its key column. /// -/// The keys are the only part of the artifact recoverable from elsewhere: a visible entry names a -/// base position, and the corpus-wide base column already holds that position's key. The choice is -/// therefore eight bytes per visible row against one indirect read per key comparison. +/// A visible entry's base position also identifies its key in the corpus-wide column. Inline keys +/// cost eight extra bytes per visible row. Shared keys require an indirect read through the base +/// position at each key comparison. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum GenerationLayout { /// Keys beside the positions, eight further bytes per visible row. @@ -294,21 +305,23 @@ pub enum GenerationLayout { Shared, } -/// The visible view published as its own generation: one bucket-major column plus a key index. +/// A visible-only cascade in bucket-major order with a population index. /// -/// The visible points cascade alone at the finest grid and then sort as a published generation -/// does, bucket-major and ascending by key inside a bucket. The cascade's contract makes this the -/// whole input of a rank-representative delivery, without a scan. A cell of depth `d` holds exactly -/// one point whose bucket lies at or below `d`, its representative, so the buckets-at-or-below-`d` -/// ranges of an extent hold one representative per occupied depth-`d` cell and their lengths count -/// those cells. +/// The visible points cascade alone at the finest grid, then sort by bucket, key and rank. For a +/// cell extent no deeper than d, with d below [`Depth::MAX`], the buckets-at-or-below-d ranges +/// contain exactly one best-ranked representative per occupied depth-d cell. Their lengths count +/// cells. At the maximum depth, exact-key duplicates also enter the catch-all and these lengths +/// count points instead. /// -/// The key index carries the same entries ascending by key, which is what answers a cell's visible -/// population - the one quantity the bucket-major order cannot count in sublinear time. +/// The separate key-sorted index permits cell-population queries without searching each bucket. A +/// cell occupies one contiguous interval in this index. /// -/// The cascade runs to [`Depth::MAX`] rather than to the schedule's deepest cut, because a -/// refinement addresses grids below that cut and a cascade's own deepest bucket is a catch-all -/// holding every co-located point rather than one per cell. +/// Cascading to [`Depth::MAX`] supports refinement below the schedule's deepest cut. The catch-all +/// retains every point that never claimed a distinct cell. +/// +/// Delivery methods require the generation's original corpus and mask. Even the inline-key layout +/// contains base positions belonging to that corpus. No generation identity check enforces this +/// pairing. /// /// One generation describes one `(corpus, mask)` pair: a mask replacement invalidates it. #[derive(Debug, PartialEq, Eq)] @@ -335,9 +348,9 @@ pub enum VisibleRankOrder { /// The visible subcorpus's own cascade, the schedule a visible-only generation publishes. /// /// The production first-occupant cascade over the visible points alone, at the corpus's deepest -/// grid. [`VisibleCascade::schedule`] returns one tile's scheduled count under that assignment and -/// [`VisibleCascade::covered`] the count the tile's cut reaches, so a delivery target derived from -/// the visible view compares against the schedule it stands in for. +/// grid. [`VisibleCascade::schedule`] returns one tile's scheduled count under that assignment, and +/// [`VisibleCascade::covered`] returns the cumulative point count at its cut. Below the catch-all, +/// this cumulative count equals occupied-cell coverage. At the catch-all it can exceed coverage. #[derive(Debug)] pub struct VisibleCascade { /// Visible points as key bits paired with their bucket depth, ascending by key. @@ -357,7 +370,36 @@ pub struct Crowding { pub duplicates: usize, } -/// A synthetic corpus with its cascade output, a visibility mask, and the walk variants. +/// A corpus and visibility mask for comparing tile-delivery rules. +/// +/// [`Self::build`] produces a clustered fixture, and [`Self::from_parts`] accepts existing cascade +/// artifacts. Per-view pyramids, columns and served generations must come from this corpus under +/// its current mask. Rebuild them after replacing the mask. These methods check no artifact +/// provenance. +/// +/// Tile-delivery addresses must be on-grid and within [`Self::max_zoom`], with cuts inside the key +/// width. Some early returns precede validation. For a root gather or uniform-grid delivery, the +/// implementation ignores x and y. +/// +/// [`Self::visible_only`] retains original row identities and can leave a sparse row domain. Mask +/// replacement methods require row ids dense in the resident row count. Use a sparse visible-only +/// corpus without remasking. +/// +/// # Example +/// +/// With the `bench` feature, compare independent and chained delivery on co-located points. Row +/// zero represents the root, and row one occupies the depth-one catch-all. Hiding row zero makes +/// the root fill with row one: +/// +/// ```rust +/// use hash_graph_atlas::bench::lod::WalkBench; +/// +/// let mut bench = WalkBench::from_parts(&[0, 0], &[1, 1], vec![0, 1], 0, 1); +/// bench.mask_rows([1]); +/// assert_eq!(bench.independent_delivery(0, 0, 0), vec![1]); +/// assert_eq!(bench.independent_delivery(1, 0, 0), vec![1]); +/// assert!(bench.chained_delivery(1, 0, 0).is_empty()); +/// ``` #[derive(Debug)] pub struct WalkBench { /// Morton codes in base order, bucket-segmented. @@ -383,6 +425,10 @@ pub struct WalkBench { } /// Builds both directions of the corpus-wide `(key, rank)` order. +/// +/// # Panics +/// +/// Panics when the columns differ in length or their count exceeds `u32::MAX`. fn key_order( codes: &[MortonKey], ranks: &[u32], @@ -407,6 +453,13 @@ fn key_order( } /// Inverts the row column over its mask domain. +/// +/// Absent rows receive [`BasePosition::MAX`]. The rows must be distinct, lie below `domain` and +/// number at most `u32::MAX`. +/// +/// # Panics +/// +/// Panics on a repeated row, a row outside `domain` or a position beyond the base-position domain. fn positions_of_rows(rows: &[u32], domain: usize) -> Box> { let mut positions: IdVec = IdVec::from_elem(BasePosition::MAX, domain); for (position, &row) in rows.iter().enumerate() { @@ -419,7 +472,13 @@ fn positions_of_rows(rows: &[u32], domain: usize) -> Box, key_order_of_position: &[KeyOrdinal], @@ -458,12 +517,13 @@ fn radix_key_order( source } +#[expect(clippy::missing_panics_doc)] impl WalkBench { /// Builds the corpus and runs the production cascade over it. /// - /// The corpus is eight gaussian clusters over a uniform background, so dense cells stay - /// populated down to the deepest zooms and descent paths are real. Equal `(points, seed)` pairs - /// build identical fixtures. The mask starts all-visible. + /// The corpus mixes eight Gaussian clusters with a uniform background to populate deep cells + /// and exercise long descent paths. Equal `(points, seed)` pairs repeat within the same + /// floating-point and sorting implementation. The mask starts all-visible. /// /// # Panics /// @@ -533,9 +593,7 @@ impl WalkBench { ) .expect("finite synthetic coordinates admit a world frame"); - // The column values cross from the typed lod domains into this probe's raw u32 - // vocabulary once, at this boundary; the scan machinery below reads the raw form, while - // storage indexed by an id keeps its index domain. + // the scan machinery uses raw u32 values, while id-indexed storage retains its domain let row_of_position: Box<[u32]> = lod .row_of_position .as_raw() @@ -569,21 +627,23 @@ impl WalkBench { } } - /// Builds the instrument over externally supplied cascade artifacts. + /// Builds a delivery probe over supplied cascade artifacts. /// - /// `code_bits` is the bucket-segmented base-order key column as raw key bits; `lengths` the - /// per-bucket segment lengths in depth order (fewer entries than the bucket table reads as - /// trailing empty buckets); `row_of_position` the base permutation; `span` and `max_zoom` the - /// delivery schedule. The mask starts all-visible. Feeding one corpus's real artifacts to both - /// this instrument and the serving path is what a set-agreement comparison rides. + /// `code_bits` contains base-order keys. `lengths` gives each bucket's length in depth order, + /// with omitted trailing buckets empty. `row_of_position` must permute `0..code_bits.len()`. + /// The mask starts all-visible. /// - /// The row identity stands in for the importance rank a rank-representative rule reads, so a - /// delivery over these parts represents each cell by its lowest row id. + /// Row ids stand in for importance ranks: rank-representative rules select the lowest row id in + /// a cell. Keys within each bucket must ascend by `(key, row id)`. The bucket assignment must + /// be the cascade for these ranks and the supplied schedule. The schedule requires `span + + /// max_zoom ≤ 32`. Construction checks lengths, the row permutation and the individual + /// parameter domains, but not sortedness, cascade agreement or the depth sum. /// /// # Panics /// - /// This panics when the lengths overrun the bucket table or disagree with the code count, or - /// when the columns disagree on length. + /// Panics when lengths overrun the bucket table or fail to cover the code column, the columns + /// differ in length, the row count exceeds `u32::MAX`, a row is repeated or out of range, `span + /// ≥ 64`, or `max_zoom > 32`. #[must_use] pub fn from_parts( code_bits: &[u64], @@ -660,8 +720,10 @@ impl WalkBench { /// Replaces the mask, hiding each row independently. /// - /// Each row stays visible with probability `visible`; equal `(visible, seed)` pairs reproduce - /// the same mask. + /// A row is visible when its discrete uniform draw in `[0, 1)` is below `visible`. Values at + /// least one show every row, and nonpositive values or NaN hide every row. Equal `(visible, + /// seed)` pairs reproduce the mask for the same row count. Row ids must be dense in the + /// resident row count. pub fn mask_uniform(&mut self, visible: f64, seed: u64) { let rows = self.row_of_position.len(); let mut rng = keyed_rng(seed, 0x0DD5_EED5, 1); @@ -674,14 +736,20 @@ impl WalkBench { self.visible = mask; } - /// Replaces the mask, hiding whole spatial blocks until the hidden rows meet the quota. + /// Replaces the mask by hiding rows in randomly drawn spatial blocks. + /// + /// `visible` must lie in `[0, 1]`, and row ids must be dense in the resident row count. The + /// quota is ⌊(1 − visible) · rows⌋ after `f64` arithmetic. Drawn cells have depths 4 through 7. + /// The last block can be partially hidden to meet the quota exactly. Spatially contiguous + /// hiding exercises fills whose scheduled runs contain no visible rows. + /// + /// The loop has no iteration bound. A negative `visible` can request more hidden rows than + /// exist and prevent termination. + /// + /// # Panics /// - /// Blocks are cells drawn at depths 4 through 7; every row inside a drawn cell hides until the - /// hidden rows reach the `1 - visible` share of the corpus, rounded down to whole rows. - /// Spatially contiguous hiding is the adversarial mask shape: whole scheduled runs vanish and - /// fills walk deep. + /// Panics when a retained row id lies outside the resident row count. #[expect( - clippy::missing_panics_doc, clippy::cast_possible_truncation, clippy::cast_precision_loss, clippy::cast_sign_loss, @@ -738,8 +806,8 @@ impl WalkBench { /// Replaces the mask with an explicit visible row set. /// - /// The set is what a serving-side visibility proof pins, so one masked view can drive this - /// instrument and the serving path at once. + /// Repeated ids have no additional effect. Row ids must lie in the resident row count, + /// including after a [`Self::visible_only`] transformation. /// /// # Panics /// @@ -754,8 +822,8 @@ impl WalkBench { /// Returns the visible keys the tile's cut reaches inside the tile cell, ascending. /// - /// Every visible point whose bucket lies at or below `z + m`: what today's chain delivers - /// cumulatively for the tile's extent when the mask hides nothing. + /// Includes every visible point in buckets at or below `z + m`. With full visibility, this is + /// the unmasked chain's cumulative delivery inside the tile. /// /// # Panics /// @@ -790,11 +858,13 @@ impl WalkBench { keys } - /// Returns the tile's scheduled count before masking, which is today's per-tile budget. + /// Returns the tile's scheduled count before masking. + /// + /// The schedule's maximum zoom is not checked here. /// /// # Panics /// - /// This panics when the coordinate lies off the grid. + /// Panics when the coordinate lies off the grid, `z > 32`, or `z + span > 32`. #[must_use] pub fn scheduled(&self, z: u8, x: u32, y: u32) -> usize { self.budget_of(z, x, y) @@ -882,11 +952,11 @@ impl WalkBench { /// Delivers one tile behind its recomputed ancestor chain. /// - /// The walk re-derives every ancestor's delivery against the same mask, top down, and the - /// tile's own fill skips everything the chain took. An ancestor whose fill ends short of budget - /// spent its subtree's visible pool, so the chain stops there and the tile delivers nothing. - /// Every descendant extent is a subset of the spent one. [`Selection::scanned`] sums the - /// chain's scans. The other counts describe the tile itself. + /// The walk recomputes each ancestor's delivery against the same mask, top down, and excludes + /// everything it took. An ancestor that exhausts its visible, untaken candidates below budget + /// ends the chain. Every descendant extent is a subset of that exhausted extent and has no + /// eligible point left. The tile then delivers nothing. [`Selection::scanned`] sums the chain's + /// scans, while the other counts describe the tile itself. /// /// # Panics /// @@ -941,15 +1011,15 @@ impl WalkBench { delivered } - /// Delivers one tile behind its recomputed ancestor chain, returning the delivered positions in - /// delivery order. + /// Returns one tile's new positions after recomputing its ancestor chain. /// - /// The chain and its early exit follow [`Self::chained`] exactly; a spent chain returns the - /// empty delivery. + /// Positions follow delivery order. The chain and early exit follow [`Self::chained`], with an + /// empty delivery after ancestor exhaustion. /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// May panic on an off-grid coordinate, a zoom beyond the schedule or an unrepresentable cut. + /// Ancestor exhaustion can return before checking the target address. #[must_use] pub fn chained_delivery(&self, z: u8, x: u32, y: u32) -> Vec { let mut taken = DenseBitSet::new_empty(self.codes.len()); @@ -1023,8 +1093,7 @@ impl WalkBench { /// Builds the Morton-ordered visible column over the whole corpus. /// - /// One sort of the visible key column; the footprint is sixteen bytes per visible row, which - /// [`VisibleColumn::footprint`] reports. + /// The footprint is sixteen bytes per visible row, which [`VisibleColumn::footprint`] reports. #[must_use] pub fn column(&self) -> VisibleColumn { let mut points = Vec::with_capacity(self.visible.count()); @@ -1074,7 +1143,8 @@ impl WalkBench { /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// Panics when `z` exceeds the schedule's maximum zoom or a non-root coordinate lies off its + /// grid. The root ignores x and y. #[must_use] pub fn gather(&self, z: u8, x: u32, y: u32) -> VisibleColumn { assert!( @@ -1102,6 +1172,10 @@ impl WalkBench { } /// Appends the position's column entry when its row is visible. + /// + /// # Panics + /// + /// Panics when `position` lies outside the corpus or its row lies outside the mask domain. fn collect(&self, position: usize, points: &mut Vec) { if !self .visible @@ -1135,14 +1209,17 @@ impl WalkBench { /// Returns the visible view as its own corpus, with nothing hidden. /// - /// The visible keys enter the production cascade as a standalone generation at the same deepest - /// grid, ranked in the relative order they hold here, and sort into the base delivery order a - /// published generation would carry. The row column keeps this corpus's row identities, so - /// [`Self::rows`] over either corpus names the same rows. + /// The visible keys enter the production cascade at the same deepest grid, preserving their + /// relative rank order, then sort into base delivery order. Original row ids remain intact: + /// [`Self::rows`] over either corpus names the same rows. Keys are copied without fitting or + /// normalization. + /// + /// Comparing the two corpora tests whether a rule's delivered sequence depends on hidden rows + /// when visible keys, relative ranks and schedule are fixed. Reading a hidden quantity can + /// change a result, but need not do so for every fixture. /// - /// A rule whose delivery is a function of the visible view alone delivers equal rows over the - /// two corpora, tile for tile and in the same order; a rule reading any hidden quantity does - /// not. That comparison is what the corpus is for. + /// Retained row ids can be sparse. Mask replacement methods require dense row ids within the + /// resident count, which this transformation does not establish. /// /// # Panics /// @@ -1227,14 +1304,10 @@ impl WalkBench { /// The visible entries cascade alone at [`Depth::MAX`] under the corpus's own importance order /// restricted to them, then sort bucket-major and ascending by key inside a bucket. `layout` /// chooses whether each entry stores its key: [`GenerationLayout::Shared`] recovers it from the - /// corpus base column through the entry's position, for four bytes per visible row against - /// twelve. + /// corpus base column through the entry's position. Including the population index, the shared + /// columns use eight bytes per visible row and the inline columns sixteen. /// /// [`ServedGeneration::footprint`] reports the bytes. - /// - /// # Panics - /// - /// This panics when the visible rows overrun the `u32` row domain. #[must_use] pub fn generation(&self, layout: GenerationLayout) -> ServedGeneration { let (keys, positions, ranks) = self.visible_entries(); @@ -1252,19 +1325,12 @@ impl WalkBench { Self::assemble(layout, &keys, &positions, &ranks, buckets.as_raw()) } - /// Builds the same generation by neighbour separation instead of a per-depth cascade. + /// Assigns visible-generation buckets by nearest-better-neighbour deletion. /// - /// A point is its cell's representative from the depth at which the cell no longer holds a - /// better-ranked point on, so its bucket is one past the deepest grid it shares with any - /// better-ranked visible point, and the key-nearest better-ranked point on either side reaches - /// that deepest shared grid. Deleting the entries from a key-ordered list in worst-rank-first - /// order exposes exactly those two neighbours, so two sorts and one linear pass assign every - /// bucket. - /// - /// The assignment equals [`Self::generation`]'s entry for entry; equal keys share every grid, - /// so a point sharing its key with a better-ranked one takes the catch-all bucket. - /// - /// # Panics + /// Keys sharing a cell form a contiguous interval in key order. For any point, the nearest + /// better-ranked key on either side attains that side's deepest shared grid. Deleting points + /// worst-rank-first from a key-ordered list exposes exactly these neighbours. Therefore two + /// sorts and one linear deletion pass recover the first-separation bucket assignment. /// /// This panics when the visible rows overrun the `u32` row domain. #[must_use] @@ -1328,6 +1394,12 @@ impl WalkBench { } /// Builds a served generation from visible positions in `(key, rank)` order. + /// + /// Positions must be distinct and sorted by their corpus keys and ranks. + /// + /// # Panics + /// + /// Panics when a position lies outside the corpus or the entry count exceeds `u32::MAX`. fn generation_from_key_order( &self, layout: GenerationLayout, @@ -1381,10 +1453,14 @@ impl WalkBench { /// Builds the served generation by merging the production buckets' visible runs. /// - /// The base order is 33 `(key, rank)`-ordered runs, one per bucket. Transposing the row mask to - /// base positions restricts each run without sorting; a 33-way merge then produces visible key - /// order for the monotonic-stack assignment. For `N` corpus rows and `V` visible rows, this - /// costs `O(N / 64 + V log 33)` and needs no further corpus-wide index. + /// The base order has 33 `(key, rank)`-ordered runs, one per bucket. Transposing the row mask + /// to base positions restricts each run without sorting. A 33-way merge produces visible key + /// order for the monotonic-stack assignment. + /// + /// # Complexity + /// + /// For N resident rows, a mask domain of D rows and V visible rows, time is O((N + D)/64 + V + /// log 33). Temporary storage is O(N/64 + V), with no additional corpus-wide key index. /// /// # Panics /// @@ -1464,10 +1540,11 @@ impl WalkBench { /// Builds the served generation by transposing the visible mask into key order. /// - /// Each visible row sets its shared key ordinal in a temporary bit set; iterating that set is + /// Each visible row sets its shared key ordinal in a temporary bit set. Iterating that set is /// the visible restriction of `(key, rank)` order. One monotonic-stack pass then finds both /// nearest better-ranked neighbours, and one counting distribution produces bucket-major order. - /// For `N` corpus rows and `V` visible rows, this costs `O(N / 64 + V)`. + /// With N resident rows, a mask domain of D rows and V visible rows, time is O((N + D)/64 + V) + /// and temporary storage is O(N/64 + V). #[must_use] pub fn indexed_generation(&self, layout: GenerationLayout) -> ServedGeneration { let mut visible_by_key = DenseBitSet::new_empty(self.codes.len()); @@ -1484,8 +1561,9 @@ impl WalkBench { /// Builds the served generation by radix-ordering the visible key ordinals. /// /// This form needs the inverse key ordinal alone, rather than both directions of the shared key - /// order. Its mask iteration costs `O(N / 64 + V)` and its three radix passes, monotonic stack, - /// and bucket distribution each cost `O(V)`. + /// order. For a mask domain of D rows and V visible rows, mask iteration costs O(D/64 + V). Its + /// three fixed-width radix passes, monotonic stack and bucket distribution each cost O(V), with + /// O(V) temporary storage. #[must_use] pub fn radix_generation(&self, layout: GenerationLayout) -> ServedGeneration { let positions = self.visible.iter().map(|row| self.position_of_row[row]); @@ -1495,12 +1573,20 @@ impl WalkBench { /// Counts surviving entries that moved shallower or deeper under a mask. /// - /// `full` must cover this instrument's whole corpus and `masked` its current visible view. The - /// returned pair is `(shallower, deeper)`. + /// `full` must cover this whole corpus and `masked` its current visible view. The returned pair + /// is `(shallower, deeper)`. + /// + /// # Properties + /// + /// Removing rows restricts each survivor's set of better-ranked neighbours. With fixed keys and + /// relative ranks, the maximum shared depth cannot increase. Therefore no surviving bucket + /// moves deeper, and the second count is zero for matching full and masked generations. /// /// # Panics /// - /// This panics when either artifact covers a different corpus or mask. + /// Panics when artifact lengths differ from the expected full and visible counts, a position + /// lies outside the corpus, or a masked position is absent from `full`. Equal lengths establish + /// neither corpus nor mask agreement. #[must_use] pub fn bucket_movements( &self, @@ -1569,6 +1655,13 @@ impl WalkBench { } /// Orders the visible entries into a served generation under one bucket assignment. + /// + /// Columns must be entry-aligned, with distinct ranks and a valid bucket assignment. + /// + /// # Panics + /// + /// Panics when the key count exceeds `u32::MAX` or an entry has no corresponding position, rank + /// or bucket. fn assemble( layout: GenerationLayout, keys: &[MortonKey], @@ -1620,8 +1713,10 @@ impl WalkBench { /// Runs the production cascade over the visible points alone. /// /// The visible key column enters as its own corpus at the same deepest grid, ranked by base - /// position in `order`. The resulting counts are a function of the keys and the deepest grid: - /// both orders assign the same per-cell counts. + /// position in `order`. Both orders cover the same occupied cells. Cumulative counts below the + /// catch-all agree. With positive span, per-tile scheduled counts agree too. With zero span, + /// changing an ancestor's representative can change a child's own scheduled count. At the + /// catch-all, cumulative counts include all visible points. /// /// # Panics /// @@ -1665,13 +1760,15 @@ impl WalkBench { /// Delivers one tile behind its recomputed ancestor chain under a fill rule. /// - /// The chain, the per-level order, and the early exit match [`Self::chained`]. - /// [`FillRule::Unmasked`] reproduces that variant's counts exactly. [`Selection::budget`] holds - /// the tile's own target under `rule`. + /// [`FillRule::Unmasked`] reproduces [`Self::chained`]'s counts. Other bucket rules change the + /// target and exhaustion test. Rank rules resolve unrepresented cells in key order and do not + /// use the exhaustion exit. [`Selection::budget`] holds the tile's own target under `rule`. The + /// artifacts must satisfy [`VisibleView`]'s corpus and mask pairing. /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// May panic on an invalid tile address, an unrepresentable cut or mismatched view artifacts. + /// Ancestor exhaustion can return before checking the maximum zoom. #[must_use] pub fn deliver( &self, @@ -1698,11 +1795,13 @@ impl WalkBench { /// Delivers one tile under a fill rule, returning the delivered positions in delivery order. /// - /// A spent chain returns the empty delivery, as [`Self::chained_delivery`] does. + /// Exhausted bucket-walk chains return an empty delivery, as [`Self::chained_delivery`] does. + /// The view must describe this corpus and its current mask. /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// May panic on an invalid tile address, an unrepresentable cut or mismatched view artifacts. + /// Ancestor exhaustion can return before checking the maximum zoom. #[must_use] pub fn delivery( &self, @@ -1734,7 +1833,8 @@ impl WalkBench { /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// May panic on an invalid tile address, an unrepresentable cut or mismatched view artifacts. + /// Ancestor exhaustion can return before checking the maximum zoom. #[must_use] pub fn cumulative_delivery( &self, @@ -1764,12 +1864,13 @@ impl WalkBench { /// Returns the depth-`depth` cells inside the tile cell holding a visible point. /// - /// Built by one pass over the whole corpus under the current mask, independent of every - /// artifact a delivery reads. + /// Scans the corpus under its current mask, independently of cached view artifacts. Depths + /// shallower than the tile are permitted and identify containing cells. The schedule's maximum + /// zoom is not checked. /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// Panics when `z > 32` or the coordinate lies off its grid. #[must_use] pub fn occupied_cells(&self, z: u8, x: u32, y: u32, depth: Depth) -> HashSet { let cell = cell_of(z, x, y); @@ -1789,13 +1890,14 @@ impl WalkBench { /// Audits one tile's chain against the visible cells its cut resolves. /// - /// The delivery runs exactly as [`Self::deliver`] does; the audit additionally counts the - /// cut-depth cells the chain's deliveries inside the tile cell occupy, so a target's count and - /// the coverage it achieves are separate numbers. + /// The delivery follows [`Self::deliver`]. The audit additionally counts the cut-depth cells + /// occupied by the chain's deliveries inside the tile, separating the target from achieved + /// coverage. /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// May panic on an invalid tile address, an unrepresentable cut or mismatched view artifacts. + /// Ancestor exhaustion can return before checking the maximum zoom. #[must_use] pub fn audit( &self, @@ -1844,8 +1946,12 @@ impl WalkBench { /// Delivers one tile behind its chain, recording the chain's deliveries inside the tile cell. /// - /// The rank-representative rules run their own engine over the visible column; the count-based - /// rules run the bucket walk. + /// Rank-representative rules use the visible column. Count-based rules use the bucket walk. + /// + /// # Panics + /// + /// May panic on an invalid tile address, a cut beyond the key width or mismatched view + /// artifacts. fn chain( &self, rule: FillRule, @@ -1968,13 +2074,18 @@ impl WalkBench { } } - /// Delivers one tile behind its chain by representing cells, not by filling a count. + /// Delivers one tile's new cell representatives after recomputing its chain. + /// + /// Every level resolves its grid and delivers the best-ranked visible point of each cell that + /// no chain delivery represents, ascending by cell index. [`RankPlan::Coarse`] uses the level + /// cut. [`RankPlan::Refined`] chooses a finer grid, optionally deepening individual cells. The + /// representative choices use the visible column, but a scheduled refinement budget also reads + /// the unmasked bucket counts. /// - /// Every level resolves the cells of its own grid and delivers the best-ranked visible point of - /// each cell no chain delivery already sits in, ascending by cell index. The grid is the level - /// cut under [`RankPlan::Coarse`] and the finest grid the budget admits under - /// [`RankPlan::Refined`]. Nothing read here is a bucket, a hidden row, or a count derived from - /// one, so the delivered rows are a function of the visible view alone. + /// # Panics + /// + /// Panics on an invalid tile address or cut depth. A column from another corpus can also + /// contain out-of-range positions. fn rank_chain( &self, plan: RankPlan, @@ -2088,8 +2199,7 @@ impl WalkBench { DotBudget::Constant(budget) => budget, DotBudget::Scheduled => self.budget_of(z, x, y), }; - // Whole levels first: the coarsest grid stays the floor, so a level whose cut alone - // overruns the budget still delivers the cut. + // the cut grid is the minimum resolution, even when its target exceeds the budget while depth < Depth::MAX && cells.len() < range.len() { let finer = Depth::new(depth.get() + 1).expect("a depth below the maximum has a successor"); @@ -2153,8 +2263,9 @@ impl WalkBench { /// Delivers one tile under a rank-representative rule out of a served generation. /// - /// The rule, the chain, and the delivered sequence are [`Self::deliver`]'s; the work is range - /// reads over the generation rather than a scan of each cell. + /// Uses generation ranges to select representatives. See the module's catch-all limitation + /// before comparing this with [`Self::deliver`]: exact-key duplicates can change refinement + /// choices at the maximum depth. `generation` must describe this corpus and mask. /// /// # Panics /// @@ -2214,8 +2325,7 @@ impl WalkBench { delivered } - /// Delivers one tile out of a served generation and returns every chain delivery inside the - /// tile cell. + /// Returns a served chain's cumulative positions inside one tile. /// /// # Panics /// @@ -2250,8 +2360,10 @@ impl WalkBench { /// Audits one tile's served chain against the visible cells its cut resolves. /// - /// Every count is [`Self::audit`]'s; [`ChainAudit::scanned`] counts generation entries read in - /// place of candidate positions examined. + /// Compare with [`Self::audit`] using the same corpus and mask. [`ChainAudit::scanned`] uses + /// the served engine's work counter. At a cut of [`Depth::MAX`], `covered` counts all visible + /// entries, including exact-key duplicates, rather than distinct cells. Refinement has the + /// module's catch-all limitation. /// /// # Panics /// @@ -2305,9 +2417,9 @@ impl WalkBench { /// Reads one extent's depth-`depth` representatives out of a served generation. /// - /// The base positions of the buckets-at-or-below-`depth` entries inside the tile cell, - /// ascending by key: one point per depth-`depth` cell of the extent holding visible content, - /// each its cell's best-ranked visible point. + /// Returns positions in ascending key order, one best-ranked point per occupied cell. At + /// [`Depth::MAX`], exact-key deduplication removes catch-all duplicates. `generation` must + /// describe this corpus and mask. /// /// # Panics /// @@ -2347,12 +2459,12 @@ impl WalkBench { /// Returns one zoom's public uniform-grid depth. /// - /// The grid is `d(z) = z + m + k`, where `m` is the schedule span and `k` is - /// `additional_depth`. The result clamps to [`Depth::MAX`]. + /// The grid is d(z) = min(z + m + k, 32), where m is the schedule span and k is + /// `additional_depth`. /// /// # Panics /// - /// This panics when `z` lies beyond the schedule's deepest zoom. + /// Panics when `z` exceeds the schedule's maximum zoom or `additional_depth ≥ 64`. #[must_use] pub fn uniform_grid_depth(&self, z: u8, additional_depth: u8) -> Depth { assert!( @@ -2366,7 +2478,16 @@ impl WalkBench { .saturating_add(additional_depth) } - // Reads either one delta or the cumulative prefix directly in scope-bucket order. + /// Reads one delta or cumulative prefix in generation-bucket order. + /// + /// `previous` is the parent tile's grid depth. The read starts one bucket below it, or at the + /// shallowest bucket when a cumulative read or the root passes `None`. The root ignores x and + /// y. + /// + /// # Panics + /// + /// Panics on an off-grid non-root coordinate. Mismatched generation positions can also exceed + /// the corpus key column. fn uniform_positions( &self, address: (u8, u32, u32), @@ -2407,11 +2528,11 @@ impl WalkBench { delivered } - /// Delivers one tile from a public uniform grid in scope-bucket order. + /// Delivers one uniform-grid tile in generation-bucket order. /// - /// Every zoom uses the same `additional_depth`, so consecutive levels read consecutive buckets - /// of the visible-only generation. A non-root delta is one bucket; the root is the prefix - /// through [`Self::uniform_grid_depth`]. + /// A fixed `additional_depth` makes successive grids differ by one depth until saturation. The + /// root reads the prefix through [`Self::uniform_grid_depth`]. Each non-root delta reads the + /// next bucket, or nothing when both grids have saturated. /// /// Before [`Depth::MAX`], accumulating these deltas gives exactly one best-ranked visible point /// per occupied cell of the public grid. At [`Depth::MAX`], the catch-all bucket also carries @@ -2419,7 +2540,9 @@ impl WalkBench { /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// Panics when `z` exceeds the schedule's maximum zoom, `additional_depth ≥ 64`, or a non-root + /// coordinate lies off its grid. Mismatched generation positions can also exceed the corpus key + /// column. The root ignores x and y. #[must_use] pub fn uniform_delivery( &self, @@ -2437,7 +2560,7 @@ impl WalkBench { ) } - /// Accumulates a public uniform grid inside one tile in scope-bucket order. + /// Accumulates a uniform grid inside one tile in generation-bucket order. /// /// The result is the visible-only generation prefix through [`Self::uniform_grid_depth`], /// narrowed to the tile cell. Accumulating [`Self::uniform_delivery`] down the tile's ancestor @@ -2445,7 +2568,9 @@ impl WalkBench { /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// Panics when `z` exceeds the schedule's maximum zoom, `additional_depth ≥ 64`, or a non-root + /// coordinate lies off its grid. Mismatched generation positions can also exceed the corpus key + /// column. The root ignores x and y. #[must_use] pub fn uniform_cumulative_delivery( &self, @@ -2484,14 +2609,16 @@ impl WalkBench { /// Delivers one tile from a public one-level refinement step. /// /// Zooms below `refine_from_zoom` use the cut grid. That zoom and every later regular zoom use - /// one additional level globally. The transition delta reads two consecutive scope buckets; - /// later deltas read one. The deepest zoom reads through [`Depth::MAX`] so the terminal tile - /// keeps the cascade's catch-all completeness. Delivery remains scope-bucket ordered - /// throughout. + /// one additional level globally. Before saturation, a non-root transition delta reads two + /// consecutive generation buckets and later regular deltas read one. The deepest zoom reads + /// through [`Depth::MAX`] to include every remaining visible point. Delivery stays in + /// generation-bucket order. /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// Panics when `z` exceeds the schedule's maximum zoom or a non-root coordinate lies off its + /// grid. Mismatched generation positions can also exceed the corpus key column. The root + /// ignores x and y. #[must_use] pub fn uniform_step_delivery( &self, @@ -2523,7 +2650,9 @@ impl WalkBench { /// /// # Panics /// - /// This panics when the coordinate lies off the grid or beyond the schedule's deepest zoom. + /// Panics when `z` exceeds the schedule's maximum zoom or a non-root coordinate lies off its + /// grid. Mismatched generation positions can also exceed the corpus key column. The root + /// ignores x and y. #[must_use] pub fn uniform_step_cumulative_delivery( &self, @@ -2547,7 +2676,12 @@ impl WalkBench { /// /// The chain is [`Self::rank_chain`]'s: every level resolves its own grid and delivers the /// best-ranked visible point of each cell no chain delivery already sits in, ascending by cell - /// index. + /// index. The generation must describe this corpus and mask. + /// + /// # Panics + /// + /// Panics on an invalid tile address or cut depth. Mismatched generation positions can also + /// exceed the corpus key column. fn served_chain( &self, plan: RankPlan, @@ -2637,10 +2771,10 @@ impl WalkBench { /// Plans one level's delivered grid out of range reads and delivers its representatives. /// - /// The plan reads three quantities off the generation without scanning the extent. The lengths - /// of the buckets-at-or-below-depth ranges count the cells of a grid, the length of the one - /// bucket below a cell counts its children, and one search of the key index answers a cell's - /// population. The delivered points are the ranges themselves. + /// Below the catch-all, prefix range lengths count occupied cells. Within a cell, the next + /// bucket plus its existing representative counts occupied children. The key index supplies + /// visible populations. These range identities avoid scanning every point, but count exact-key + /// duplicates as separate cells at the catch-all. /// /// # Panics /// @@ -2718,13 +2852,17 @@ impl WalkBench { } } - /// Returns the finest grid one level's budget admits, that budget, and the entries the search - /// read. + /// Returns the chosen whole-grid depth, its budget and the search work count. + /// + /// The cut grid is the minimum resolution even when its target exceeds the budget. Each + /// candidate grid adds the next bucket's length and subtracts distinct represented cells. Below + /// the catch-all, this gives the new representative count without scanning points. Catch-all + /// duplicates can overestimate it. The work count includes passes over the inherited keys. + /// + /// # Panics /// - /// The coarsest grid stays the floor, so a level whose cut depth alone overruns the budget - /// still delivers the cut. Each candidate grid costs one range length and one pass over the - /// chain's keys inside the extent: the cells of a grid are the entries at or below its depth, - /// so the search never looks at a point. + /// May panic on an invalid address, an unrepresentable cut or inconsistent range and + /// inherited-key counts. fn served_grid( &self, plan: RankPlan, @@ -2763,11 +2901,14 @@ impl WalkBench { /// Marks the grid cells a partial refinement takes one level further. /// - /// Returns the count the deepening adds to the level's target, the cells it deepened, and the - /// entries it read. A cell's children are the one bucket below the grid plus the cell's own - /// representative, one range length; its population is one search of the key index, which a - /// level with nothing left to spend never makes - what remains to deepen there costs no budget - /// in any order. + /// Returns target growth, deepened-cell count and work count. Below the catch-all, the next + /// bucket plus the parent representative counts occupied children. With no remaining budget, + /// only zero-growth splits can fit. Selecting them is independent of visit order, which makes + /// population searches unnecessary. + /// + /// # Panics + /// + /// May panic when scratch columns, grid ranges or inherited keys disagree. fn served_deepen( &self, spending: (RefineOrder, usize), @@ -2779,9 +2920,9 @@ impl WalkBench { let (order_of, mut remaining) = spending; let (depth, finer) = grid; let cells = scratch.candidates.len(); - // A level with nothing left to spend can only deepen a cell whose every child a chain - // delivery already sits in, so a level whose grid the chain has not reached deepens - // nothing and needs neither the children nor the populations. + // Splitting an unrepresented cell into at least two children increases the target. If no + // grid cell is represented and no budget remains, every eligible split has positive cost. + // Therefore no child or population lookup can select a split. if remaining == 0 && !scratch.occupied.contains(&true) { return (0, 0, 0); } @@ -2836,7 +2977,12 @@ impl WalkBench { /// Reads the extent's depth-`depth` representatives into the scratch, ascending by key. /// - /// Returns the generation entries the read touched. + /// Deduplicates exact keys, retaining the best-ranked representative. Returns the work count + /// for range reads and merges. + /// + /// # Panics + /// + /// Panics when a range or a shared-layout position lies outside its column. fn represented_cells( &self, generation: &ServedGeneration, @@ -2860,8 +3006,9 @@ impl WalkBench { &self.codes, ); } - // Equal keys share every cell, so the finest grid's catch-all bucket can repeat a cell's - // representative; the smallest bucket sorts first, which is the best-ranked one. + // Equal keys share every cell. Merges retain shallower buckets first, and each bucket + // orders equal keys by rank. Therefore exact-key deduplication retains the best-ranked + // representative, including when all duplicates occupy the catch-all. scratch.candidates.dedup_by_key(|&mut (key, _)| key); reads @@ -2869,7 +3016,12 @@ impl WalkBench { /// Reads each depth-`depth` cell's visible population out of the generation's key index. /// - /// Returns the index entries the searches probed. + /// Returns the number of counted search probes. `cells` must match the occupied-cell count at + /// `depth` within `cell`, and the generation must describe this corpus. + /// + /// # Panics + /// + /// Panics when `cells` exceeds that count or an indexed position lies outside its column. fn populations( &self, generation: &ServedGeneration, @@ -2901,8 +3053,13 @@ impl WalkBench { /// Collects the cells the chain's deliveries inside the level's cell occupy at `cut`. /// - /// `history` groups the chain's deliveries by the deepest level whose cell holds them, so the - /// levels from `z` on are exactly the deliveries inside this level's cell. + /// `history` groups deliveries by the deepest chain cell containing them. Entries from `z` + /// onward are exactly the deliveries inside this level's cell. Other rules leave `represented` + /// unchanged. + /// + /// # Panics + /// + /// Panics when a selected history position lies outside the corpus. fn represent( &self, rule: FillRule, @@ -2923,6 +3080,10 @@ impl WalkBench { } /// Counts the distinct cells the positions occupy at `depth`. + /// + /// # Panics + /// + /// Panics when a position lies outside the corpus. fn distinct_cells(&self, positions: &[u32], depth: Depth) -> usize { let mut cells = HashSet::with_capacity(positions.len()); for &position in positions { @@ -2990,7 +3151,13 @@ impl WalkBench { /// Delivers one tile, scheduled points first, then the fill from deeper buckets. /// /// `taken` positions never deliver. The walk appends every delivered position to `out`. The - /// scheduled admissions deliver whole, and the fill runs while the delivery stays below `fill`. + /// scheduled admissions deliver whole, even if they exceed the target or repeat represented + /// cells. The tail fills a count or uncovered cells according to `fill`. + /// + /// # Panics + /// + /// Panics on an invalid tile address, a cut beyond the key width or a row outside the mask + /// domain. fn walk( &self, z: u8, @@ -3107,8 +3274,12 @@ impl WalkBench { /// Returns whether the position's row is visible and the position is untaken. /// - /// A `taken` set sized zero excludes nothing: positions beyond its domain are absent by - /// definition. + /// A zero-sized `taken` set excludes nothing. This method treats positions outside that set's + /// domain as untaken. + /// + /// # Panics + /// + /// Panics when `position` lies outside the corpus or its row lies outside the mask domain. fn admits(&self, taken: &DenseBitSet, position: usize) -> bool { let taken = position < taken.domain_size() && taken.contains(BasePosition::from_usize(position)); @@ -3130,6 +3301,7 @@ impl WalkBench { } } +#[expect(clippy::missing_panics_doc)] impl VisibleCellPyramid { /// Counts the depth's cells inside `cell` holding a visible point. /// @@ -3151,7 +3323,7 @@ impl VisibleCellPyramid { level.partition_point(|&index| index <= high) - level.partition_point(|&index| index < low) } - /// Returns the cells one depth holds. + /// Returns the number of occupied cells at one depth. /// /// # Panics /// @@ -3162,10 +3334,6 @@ impl VisibleCellPyramid { } /// Returns the pyramid's depths, shallowest first. - /// - /// # Panics - /// - /// This panics when a level's depth lies beyond the key width. #[must_use] pub fn depths(&self) -> impl IntoIterator { let shallowest = self.shallowest; @@ -3184,7 +3352,11 @@ impl VisibleCellPyramid { .sum() } - /// Returns one depth's cells. + /// Returns one depth's occupied cells in ascending order. + /// + /// # Panics + /// + /// Panics when `depth` lies outside the pyramid's levels. fn level(&self, depth: Depth) -> &[u64] { let offset = depth .get() @@ -3254,7 +3426,11 @@ impl VisibleColumn { start..end } - /// Splits the slice into its depth's cells, ascending by cell index. + /// Replaces `out` with the range's occupied cells in ascending cell order. + /// + /// # Panics + /// + /// Panics when a nonempty range extends beyond the column. fn split(&self, range: Range, depth: Depth, out: &mut Vec>) { out.clear(); let mut at = range.start; @@ -3265,7 +3441,11 @@ impl VisibleColumn { } } - /// Returns the end of the depth's cell the slice's first point lies in. + /// Returns the end of the first occupied cell in `at..end`. + /// + /// # Panics + /// + /// Panics when `at` lies outside the column, `end` exceeds it or `at > end`. fn cell_end(&self, at: usize, end: usize, depth: Depth) -> usize { let prefix = MortonKey::from_bits(self.points[at].key).prefix(depth); let slice = &self.points[at..end]; @@ -3273,15 +3453,23 @@ impl VisibleColumn { at + slice.partition_point(|point| MortonKey::from_bits(point.key).prefix(depth) <= prefix) } - /// Returns the cell of `depth` the slice's points share. + /// Returns the depth-`depth` cell containing the point at `at`. + /// + /// # Panics + /// + /// Panics when `at` lies outside the column. fn cell_of_slice(&self, at: usize, depth: Depth) -> MortonCell { MortonKey::from_bits(self.points[at].key).cell(depth) } /// Returns the base position of the slice's best-ranked point. /// - /// Equal ranks resolve to the smallest key, which the corpus rank's totality leaves - /// unreachable. + /// The range must be nonempty. A rank tie resolves to the earliest point in the range, although + /// valid corpus ranks are distinct. + /// + /// # Panics + /// + /// Panics when the range starts outside the column or is invalid for slicing. fn representative(&self, range: Range) -> u32 { let mut best = self.points[range.start]; for point in &self.points[range] { @@ -3321,7 +3509,12 @@ impl ServedGeneration { + size_of::() } - /// Returns the entry's key. + /// Returns an entry's inline key or resolves it through its base position. + /// + /// # Panics + /// + /// Panics when `index` lies outside the generation or a shared-layout position lies outside + /// `codes`. fn key(&self, index: usize, codes: &[MortonKey]) -> u64 { self.keys.as_ref().map_or_else( || codes[self.positions[index] as usize].to_bits(), @@ -3329,15 +3522,24 @@ impl ServedGeneration { ) } - /// Returns the key index entry's key. + /// Returns the key at one ordinal of the ascending population index. + /// + /// # Panics + /// + /// Panics when `at` lies outside the index or [`Self::key`] cannot resolve its entry. fn ascending_key(&self, at: usize, codes: &[MortonKey]) -> u64 { self.key(self.ascending[at] as usize, codes) } /// Narrows every bucket range of an enclosing extent to the entries inside `cell`. /// - /// A chain descends through nested cells, so each level searches its parent's ranges rather - /// than the whole segment. + /// Searching the enclosing extent's ranges restricts work to the current subtree. `within` must + /// contain `cell`'s entries, and shared-layout keys must come from this generation's original + /// corpus. + /// + /// # Panics + /// + /// Panics when a searched range or shared-layout position lies outside its column. fn narrowed(&self, cell: MortonCell, within: &Ranges, codes: &[MortonKey]) -> Ranges { let (low, high) = (cell.min_key().to_bits(), cell.max_key().to_bits()); @@ -3349,7 +3551,11 @@ impl ServedGeneration { }) } - /// Returns the key index's slice inside `cell`. + /// Returns the key index's range inside `cell`. + /// + /// # Panics + /// + /// Panics when a shared-layout position lies outside `codes`. fn ascending_range(&self, cell: MortonCell, codes: &[MortonKey]) -> Range { let (low, high) = (cell.min_key().to_bits(), cell.max_key().to_bits()); let start = self.ascending_partition(0..self.ascending.len(), codes, |key| key < low); @@ -3358,8 +3564,14 @@ impl ServedGeneration { start..end } - /// Returns the end of the depth's cell the key index's entry at `at` lies in, and the probes - /// the search took. + /// Returns the first cell's end in `at..end` and the counted search probes. + /// + /// Requires a nonempty range in the ascending index. The work count excludes the initial prefix + /// lookup. + /// + /// # Panics + /// + /// Panics when a probed ordinal or shared-layout position lies outside its column. fn ascending_cell_end( &self, at: usize, @@ -3372,9 +3584,8 @@ impl ServedGeneration { MortonKey::from_bits(self.ascending_key(index, codes)).prefix(depth) <= prefix }; - // A grid cell holds few entries beside its representative, so the search finds the end by - // doubling out from the start before it narrows: the probes stay logarithmic in the cell, - // not in the extent. + // doubling from the start before binary search makes probe count logarithmic in the current + // cell's population, rather than the enclosing extent let mut probes = 0_usize; let mut inside = at; let mut outside = end; @@ -3406,6 +3617,12 @@ impl ServedGeneration { } /// Returns the first index of `range` whose key fails `before`. + /// + /// `before` must be true on an initial prefix and false thereafter. + /// + /// # Panics + /// + /// Panics when a probed entry or shared-layout position lies outside its column. fn partition( &self, range: Range, @@ -3426,7 +3643,13 @@ impl ServedGeneration { low } - /// Returns the first index of the key index's `range` whose key fails `before`. + /// Returns the first ordinal of the key-index range whose key fails `before`. + /// + /// `before` must be true on an initial prefix and false thereafter. + /// + /// # Panics + /// + /// Panics when a probed ordinal or shared-layout position lies outside its column. fn ascending_partition( &self, range: Range, @@ -3448,6 +3671,7 @@ impl ServedGeneration { } } +#[expect(clippy::missing_panics_doc)] impl VisibleCascade { /// Returns the tile's scheduled count under the visible-only assignment. /// @@ -3456,7 +3680,7 @@ impl VisibleCascade { /// /// # Panics /// - /// This panics when the coordinate lies off the zoom's grid. + /// Panics when `z > 32`, the coordinate lies off its grid or `z + span > 32`. #[must_use] pub fn schedule(&self, z: u8, x: u32, y: u32) -> usize { let cut = z + self.span; @@ -3477,7 +3701,7 @@ impl VisibleCascade { /// /// # Panics /// - /// This panics when the coordinate lies off the zoom's grid. + /// Panics when `z > 32`, the coordinate lies off its grid or `z + span > 32`. #[must_use] pub fn covered(&self, z: u8, x: u32, y: u32) -> usize { let cut = z + self.span; @@ -3497,10 +3721,6 @@ impl VisibleCascade { /// /// Every occupied cell of every grid up to the deepest holds a point whose bucket lies at or /// below that grid's depth. - /// - /// # Panics - /// - /// This panics when a stored bucket lies beyond the key width. #[must_use] pub fn coverage_holds(&self) -> bool { let keys: Vec = self @@ -3547,7 +3767,7 @@ struct ChainBuffers<'buffers> { struct ChainOutcome { /// The tile's own delivery counts. own: Selection, - /// Cut-depth cells inside the tile cell holding a visible point. + /// Cut-cell count, including duplicates when a served cut reaches the catch-all. covered: usize, /// Chain deliveries inside the tile cell, the levels above the tile alone. inherited: usize, @@ -3570,7 +3790,7 @@ struct RankStep { refined: u8, /// Cells a partial refinement took one level further. deepened: usize, - /// Column entries examined. + /// The engine-specific work count, as in [`Selection::scanned`]. scanned: usize, } @@ -3664,7 +3884,12 @@ struct RankLevel<'level> { /// Delivers the slice's representative when no chain delivery already sits in its cell. /// -/// Returns `true` when the representative delivers. +/// Returns `true` when a point is appended. `range` must be nonempty, and `represented` must ascend +/// by key. +/// +/// # Panics +/// +/// Panics when the range starts outside the column or is invalid for slicing. fn represent( column: &VisibleColumn, range: Range, @@ -3682,9 +3907,15 @@ fn represent( /// Marks the grid cells a partial refinement takes one level further. /// -/// Returns the count the deepening adds to the level's target, the cells it deepened, and the -/// column entries it read. A cell of fewer than two points, and a cell whose finer split holds a -/// single child, stay whole: deepening either one adds no representative. +/// Returns target growth, deepened-cell count and the number of child ranges examined. A cell of +/// fewer than two points, and a cell whose finer split holds a single child, stay whole: deepening +/// either one adds no representative. `cells` must be nonempty ranges of depth-`depth` cells, +/// `finer` their next depth, and `represented` an ascending key list. +/// +/// # Panics +/// +/// Panics when a cell range lies outside the column. Inconsistent grids can also make the +/// target-growth subtraction underflow. fn rank_deepen( spending: (RefineOrder, usize), column: &VisibleColumn, @@ -3736,6 +3967,12 @@ fn rank_deepen( } /// Counts the cells no chain delivery lies inside. +/// +/// Cell ranges must be nonempty, and `represented` must ascend by key. +/// +/// # Panics +/// +/// Panics when a cell range starts outside the column. fn needing( column: &VisibleColumn, cells: &[Range], @@ -3750,8 +3987,9 @@ fn needing( /// Counts the entries of an extent whose bucket lies at or below `depth`. /// -/// The cells the extent's depth-`depth` grid holds: the cascade gives each occupied cell exactly -/// one point at or below the cell's own depth. +/// When the extent is a cell no deeper than `depth`, this is its occupied-cell count below the +/// catch-all: the cascade assigns exactly one cumulative representative per cell. At the catch-all, +/// it includes exact-key duplicates. fn reach(ranges: &Ranges, depth: Depth) -> usize { ranges[..=usize::from(depth.get())] .iter() @@ -3785,7 +4023,12 @@ fn distinct_prefixes(keys: &[u64], depth: Depth) -> usize { /// Merges one bucket range's entries into an ascending candidate list. /// -/// Equal keys keep the accumulated entry first, which is the one from the shallower bucket. +/// Equal keys keep the accumulated entry first. Merging buckets shallowest-first preserves the +/// shallower representative. +/// +/// # Panics +/// +/// Panics when a run entry or shared-layout position lies outside its column. fn merge_entries( candidates: &mut Vec<(u64, u32)>, merged: &mut Vec<(u64, u32)>, @@ -3866,6 +4109,13 @@ fn retain_cell(keys: &mut Vec, cell: MortonCell) { } /// Marks the grid cells a chain delivery sits in. +/// +/// Candidates and held keys must ascend. Each candidate must represent one distinct cell at +/// `depth`. +/// +/// # Panics +/// +/// Panics when `occupied` has fewer entries than `candidates`. fn mark_represented(candidates: &[(u64, u32)], held: &[u64], depth: Depth, occupied: &mut [bool]) { let mut mark = 0_usize; for (index, &(key, _)) in candidates.iter().enumerate() { @@ -3878,7 +4128,15 @@ fn mark_represented(candidates: &[(u64, u32)], held: &[u64], depth: Depth, occup } } -/// Counts each grid cell's occupied children and the ones no chain delivery sits in. +/// Computes child and unrepresented-child counts from one finer bucket. +/// +/// Candidates, finer entries and held keys must ascend. Each candidate represents one distinct +/// cell, and `finer` must be one depth below `depth`. Counts equal occupied children only before +/// the catch-all or when exact keys are distinct. +/// +/// # Panics +/// +/// May panic when inherited keys represent more children than the supplied entries count. fn children_of( candidates: &[(u64, u32)], finer_entries: &[(u64, u32)], @@ -3903,8 +4161,8 @@ fn children_of( while at < finer_entries.len() && finer_entries[at].0 <= high { at += 1; } - // The cell's own representative is its first child's, so the children are the one bucket - // below plus it. + // the parent representative supplies one child, and the next bucket supplies the others + // below the catch-all let count = 1 + (at - from); while mark < held.len() && held[mark] < low { @@ -3922,9 +4180,14 @@ fn children_of( /// Delivers the level's grid and returns the points it delivered. /// -/// A whole cell delivers its representative when no chain delivery sits in it. A deepened cell -/// delivers its own representative and its occupied children's, ascending by key, so the delivery -/// stays in key order across the depths one level mixes. +/// A whole cell delivers its representative unless already represented. For a deepened cell, merge +/// the parent representative with the finer bucket and omit represented children. The merge +/// preserves key order across the mixed-depth grid. Catch-all duplicates in the finer bucket are +/// not deduplicated here. +/// +/// # Panics +/// +/// Panics when `cells` exceeds a scratch column's length. fn deliver_grid( held: &[u64], grid: (Depth, Depth), @@ -3969,7 +4232,9 @@ fn deliver_grid( out.len() - start } -/// Delivers one child cell's representative when no chain delivery sits in the child. +/// Appends one child representative, checking inherited coverage when `tested` is true. +/// +/// If `tested` is false, the child must be unrepresented. `held` must ascend by key. fn deliver_child( key: u64, position: u32, @@ -4029,7 +4294,13 @@ const fn rank_plan(rule: FillRule) -> Option { } } -/// Returns the cells the extent covers at `cut`, for the rules deriving a target from them. +/// Returns the cells the extent covers at `cut`, for rules deriving a target from them. +/// +/// Other rules return zero. +/// +/// # Panics +/// +/// For a coverage rule, panics when the pyramid lacks `cut` or the cut is shallower than `cell`. fn covered_of(rule: FillRule, cell: MortonCell, cut: Depth, pyramid: &VisibleCellPyramid) -> usize { match rule { FillRule::Coverage | FillRule::CoverageCells => pyramid.count(cell, cut), @@ -4078,7 +4349,7 @@ const fn cell_of(z: u8, x: u32, y: u32) -> MortonCell { .expect("the coordinate lies on the zoom's grid") } -/// Every bucket's full segment, in the instrument's scan offsets. +/// Returns every bucket's full segment as scan offsets. fn segments(fenceposts: &Fenceposts) -> Ranges { fenceposts .segments() @@ -4121,13 +4392,13 @@ mod tests { /// The dot budget the refinement checks run under: the cells one tile's cut grid holds. const BUDGET: usize = 4096; - /// The dot budget the last round measured as the knee: the cells one zoom step coarser hold. + /// A secondary budget: 4⁵ = 1024 cells, one subdivision coarser than the default cut grid. const KNEE: usize = 1024; /// The corpus scales a property builds per case. /// - /// Small enough to build hundreds of times in one run; large enough that a gaussian cluster's - /// points share cut cells, so budgets bind and refinement has cells to resolve. + /// Bounds repeated fixture construction while providing clustered inputs for cell sharing and + /// refinement. const CORPUS: RangeInclusive = 64..=1_024; /// The visible fractions a property masks with, from everything hidden to nothing. @@ -4139,12 +4410,20 @@ mod tests { /// that bind, and budgets nothing reaches. const BUDGETS: RangeInclusive = 1..=2_048; - /// Builds the fixture under one mask. + /// Builds the fixed fixture under a uniform or clustered mask. + /// + /// A clustered mask requires `visible` in `[0, 1]` as in [`WalkBench::mask_clustered`]. fn masked(clustered: bool, visible: f64) -> WalkBench { corpus(POINTS, SEED, clustered, visible) } /// Builds a corpus of `points` rows from `seed` and masks it with the same seed. + /// + /// A clustered mask requires `visible` in `[0, 1]` as in [`WalkBench::mask_clustered`]. + /// + /// # Panics + /// + /// Panics when `points` is zero or exceeds `u32::MAX`. fn corpus(points: usize, seed: u64, clustered: bool, visible: f64) -> WalkBench { let mut bench = WalkBench::build(points, seed); if clustered { @@ -4156,7 +4435,7 @@ mod tests { bench } - /// Every refinement order. + /// Generates every refinement order. fn refine_order() -> impl Strategy { prop_oneof![ Just(RefineOrder::Whole), @@ -4165,7 +4444,7 @@ mod tests { ] } - /// A refined rule under a constant budget drawn from [`BUDGETS`]. + /// Generates constant-budget refinements over [`BUDGETS`]. fn constant_refinement() -> impl Strategy { (BUDGETS, refine_order()).prop_map(|(budget, order)| { FillRule::Refined(Refinement { @@ -4175,7 +4454,7 @@ mod tests { }) } - /// A refined rule under the scheduled budget. + /// Generates scheduled-budget refinements in every order. fn scheduled_refinement() -> impl Strategy { refine_order().prop_map(|order| { FillRule::Refined(Refinement { @@ -4185,10 +4464,10 @@ mod tests { }) } - /// The rank-representative rules whose delivery is a function of the visible view alone. + /// Generates rank rules whose delivery depends only on the fixed visible view. /// - /// The coarse rule and every constant-budget refinement. The scheduled budget reads the corpus - /// before masking, so it is the family's known leak and stays out. + /// Includes the coarse rule and constant-budget refinements. The scheduled budget reads + /// unmasked corpus counts and is excluded from this comparison. fn hidden_independent_rule() -> impl Strategy { prop_oneof![ 1 => Just(FillRule::CoverageRank), @@ -4196,8 +4475,7 @@ mod tests { ] } - /// Every rule the served engine serves: [`hidden_independent_rule`] plus the scheduled budget - /// in every refinement order, weighted as the exhaustive family is. + /// Generates the served engine's coarse and refined rule families. fn served_rule() -> impl Strategy { prop_oneof![ 1 => Just(FillRule::CoverageRank), @@ -4206,7 +4484,7 @@ mod tests { ] } - /// Every refinement order under one budget. + /// Returns every refinement order under one budget. fn refinements(budget: DotBudget) -> Vec { [ RefineOrder::Whole, @@ -4218,10 +4496,7 @@ mod tests { .collect() } - /// The rank-representative rules a check compares the served form against. - /// - /// The coarse rule, both constant budgets, and the scheduled budget in every refinement order: - /// the whole family the last round measured. + /// Returns the coarse rule and refinements under both constant budgets and the schedule. fn served_rules() -> Vec { let mut rules = vec![FillRule::CoverageRank]; rules.extend(refinements(DotBudget::Constant(BUDGET))); @@ -4242,7 +4517,13 @@ mod tests { } } - /// Expands a generation's bucket segments into one bucket per base position. + /// Expands bucket segments into one bucket per base position. + /// + /// Positions absent from the generation retain [`Depth::MAX`]. + /// + /// # Panics + /// + /// Panics when a generation position is at least `positions`. fn buckets_by_position(generation: &ServedGeneration, positions: usize) -> Vec { let mut buckets = vec![Depth::MAX; positions]; for (bucket, segment) in generation.segments.iter().enumerate() { @@ -4257,10 +4538,14 @@ mod tests { buckets } - /// The count a rule's budget bounds one tile's own delivery by. + /// Returns the configured bound before the caller applies actual cut-cell coverage. + /// + /// A tile cut contains at most 4ᵐ cells, where m is the span. Small constant budgets can lie + /// below this floor. Scheduled budgets include the occupied cut-cell count here. + /// + /// # Panics /// - /// A cut grid holds `4^m` cells, so the cut-depth floor never passes the constant budget; the - /// scheduled budget's floor is the tile's own coverage. + /// Panics on an invalid address or cut when resolving a scheduled budget. fn bound(bench: &WalkBench, rule: FillRule, z: u8, x: u32, y: u32) -> usize { match rule { FillRule::Refined(Refinement { @@ -4307,7 +4592,8 @@ mod tests { /// /// Corpus B is the masked fixture. Corpus A contains the same visible rows and nothing else, /// all visible. A rule reading the visible view alone delivers the same rows over both, in the - /// same order. Anything a hidden row reaches shows up here as a disagreement. + /// same order. A mismatch on a sampled tile demonstrates interference for that fixture. + /// Agreement on the sampled tiles alone is not a proof. fn interference(rule: FillRule, clustered: bool, visible: f64) -> Option<(u8, u32, u32)> { let hidden = masked(clustered, visible); let alone = hidden.visible_only(); @@ -4338,6 +4624,10 @@ mod tests { /// /// [`interference`]'s comparison over the served engine: corpus B is the masked fixture, corpus /// A contains the same visible rows and nothing else. + /// + /// # Panics + /// + /// Panics when `rule` is outside the rank-representative family. fn served_interference( rule: FillRule, clustered: bool, @@ -4367,6 +4657,10 @@ mod tests { } /// Returns the first tile whose uniform-grid rows differ between the two corpora. + /// + /// # Panics + /// + /// Panics when `additional_depth ≥ 64`. fn uniform_interference( additional_depth: u8, clustered: bool, @@ -4424,10 +4718,6 @@ mod tests { tiles[pick.index(tiles.len())] } - /// The served engine is the scanning engine at any tile of any corpus under any served rule. - /// - /// Delivery sequence, cumulative delivery and audit agree between the two, and the indexed - /// generation the served engine reads is the cascade oracle. #[property_test] fn served_matches_scanning( #[strategy = CORPUS] points: usize, @@ -4476,8 +4766,6 @@ mod tests { ); } - /// A generation prefix read at the cut and two depths below holds one representative per - /// occupied cell and agrees with the column's coverage. #[test] fn generation_prefix_one_per_occupied_cell() { for clustered in [false, true] { @@ -4523,8 +4811,6 @@ mod tests { } } - /// Neighbour deletion, the bucket merge, the shared-order filter, the indexed stack and the - /// radix stack each build the cascade's generation, in both layouts. #[test] fn separation_matches_cascade_buckets() { for clustered in [false, true] { @@ -4567,8 +4853,6 @@ mod tests { } } - /// Masking leaves a visible point's bucket or moves it shallower, never deeper, and some point - /// moves. #[test] fn masked_bucket_not_deeper() { let mut strict = 0_usize; @@ -4603,8 +4887,6 @@ mod tests { ); } - /// The shared layout serves the inline layout's delivery under every served rule, in fewer - /// bytes. #[test] fn shared_layout_matches_inline() { for clustered in [false, true] { @@ -4625,8 +4907,6 @@ mod tests { } } - /// The served engine delivers identical rows over the masked corpus and over its visible-only - /// twin, at any tile and in the same order, under any hidden-independent rule. #[property_test] fn served_noninterference( #[strategy = CORPUS] points: usize, @@ -4656,8 +4936,6 @@ mod tests { ); } - /// The served noninterference check still separates the scheduled budget, which reads the - /// corpus before masking. #[test] fn served_noninterference_rejects_scheduled() { for rule in refinements(DotBudget::Scheduled) { @@ -4669,8 +4947,6 @@ mod tests { } } - /// The public uniform grid occupies each occupied cell once, delivers in scope-bucket order, - /// selects the set the cell-order read selects, and stays within its geometric per-tile bound. #[test] fn uniform_grid_proportional_in_bucket_order() { for additional_depth in [0_u8, 1] { @@ -4749,8 +5025,6 @@ mod tests { } } - /// The public grid's per-level deltas inside a tile cell accumulate to its cumulative delivery, - /// plain and stepped, and the stepped prefix is the occupied-cell census. #[test] fn uniform_grid_deltas_accumulate() { for additional_depth in [0_u8, 1] { @@ -4835,11 +5109,6 @@ mod tests { } } - /// The stepped public grid's terminal step is the unmasked full cut. - /// - /// Below the refinement step a tile delivers no deeper tail and above it no natural row, the - /// step itself delivers a natural run followed by a tail, and the deepest tile's cumulative - /// delivery is every visible row it gathers. #[test] fn bucket_order_wire_split_and_full_cut() { let full = masked(false, 1.0); @@ -4902,8 +5171,6 @@ mod tests { ); } - /// The public grids, plain and stepped, deliver the same rows once hidden rows exist, while the - /// scheduled budget beside them still fails. #[test] fn uniform_grid_noninterference() { for additional_depth in [0_u8, 1] { @@ -4941,8 +5208,6 @@ mod tests { } } - /// The proportional-density metric accepts the coarse rank rule and the public grid, and - /// rejects today's unmasked rule and a per-tile budget. #[test] fn density_metric_accepts_grids_rejects_budgets() { let mut today_rejected = false; @@ -5008,6 +5273,7 @@ mod tests { } counts(delivered) }; + // cross multiplication compares normalized window histograms without division let proportional = |shown: &[usize], actual: &[usize]| { let shown_total = shown.iter().sum::() as u128; let actual_total = actual.iter().sum::() as u128; @@ -5046,11 +5312,6 @@ mod tests { ); } - /// The served engine leaves no cell holding visible content empty. - /// - /// At the tile's cut and at the grid its refinement resolved, the served cumulative delivery - /// occupies exactly the occupied cells. The audit's coverage is the cut's cell count, and the - /// tile's own delivery stays within its budget or its cut-depth floor. #[property_test] fn served_covers_visible_cells( #[strategy = CORPUS] points: usize, @@ -5100,8 +5361,6 @@ mod tests { ); } - /// The scanning engine delivers identical rows over the masked corpus and over its - /// visible-only twin, at any tile and in the same order, under any hidden-independent rule. #[property_test] fn rank_rule_noninterference( #[strategy = CORPUS] points: usize, @@ -5132,8 +5391,6 @@ mod tests { ); } - /// The noninterference check still separates each rule reading a hidden quantity: the unmasked, - /// coverage, visible and cell rules, and the scheduled budget. #[test] fn noninterference_rejects_hidden_readers() { let mut rules = vec![ @@ -5153,7 +5410,6 @@ mod tests { } } - /// The unmasked rule is the chained variant, selection and delivery, down the densest descent. #[test] fn unmasked_matches_chained() { for clustered in [false, true] { @@ -5179,8 +5435,6 @@ mod tests { } } - /// The pyramid and the column count what the visible cascade covers, and the cascade's reach - /// and schedule are independent of its rank order. #[test] fn pyramid_matches_visible_cascade() { for clustered in [false, true] { @@ -5222,8 +5476,6 @@ mod tests { } } - /// A chain's inherited count is the number of ancestor deliveries inside the tile cell, up to - /// the first ancestor whose audit reports `spent` or `dry`. #[test] fn chain_inherited_matches_ancestors() { for rule in [FillRule::Unmasked, FillRule::Coverage, FillRule::Visible] { @@ -5268,8 +5520,6 @@ mod tests { } } - /// With nothing hidden, the coverage, visible and cell rules audit as the unmasked rule does, - /// and the cumulative delivery is the coverage. #[test] fn full_visibility_rules_agree() { let bench = WalkBench::build(POINTS, SEED); @@ -5291,7 +5541,6 @@ mod tests { } } - /// The cell rule represents every covered cell without running its chain short. #[test] fn cell_rule_covers_cut_cells() { for clustered in [false, true] { @@ -5316,11 +5565,6 @@ mod tests { } } - /// A rank-representative rule leaves no cell holding visible content empty. - /// - /// At the tile's cut and at the grid its refinement resolved, the cumulative delivery occupies - /// exactly the occupied cells. The audit's coverage is the cut's cell count, and the tile's own - /// delivery stays within its budget or its cut-depth floor. #[property_test] fn rank_rule_covers_visible_cells( #[strategy = CORPUS] points: usize, @@ -5372,8 +5616,6 @@ mod tests { ); } - /// A budget below the cut grid's cell count still shows every cut cell: the cut-depth floor - /// overrides it, and the overrun happens. #[test] fn small_budget_covers_cut_cells() { /// A budget far below the 4096 cells a tile's cut grid holds. @@ -5413,8 +5655,6 @@ mod tests { ); } - /// The coarse rank rule delivers the visible-only schedule, and above the catch-all its - /// cumulative delivery is the visible-only generation's cut prefix. #[test] fn coverage_rank_matches_visible_only() { for clustered in [false, true] { @@ -5435,8 +5675,10 @@ mod tests { {z}/{x}/{y}, clustered {clustered}, visible {visible}", ); - // The deepest cut's bucket is the cascade's catch-all. It holds every - // co-located point, not one per cell, so the prefix is not a cell census there. + // The deepest bucket contains all points that never claimed a distinct cell. + // Its cumulative prefix includes co-located points in both this bucket and + // shallower buckets. Therefore the deepest prefix need not equal the + // occupied-cell count. if z == bench.max_zoom() { continue; } @@ -5460,10 +5702,6 @@ mod tests { } } - /// The pyramid's depths run from the cut span to the deepest cut. - /// - /// Its footprint is its levels' occupied cells in bytes, and at each depth the root's count is - /// that depth's occupancy. #[test] fn pyramid_holds_cut_depths() { let bench = WalkBench::build(POINTS, SEED); diff --git a/libs/@local/graph/atlas/src/salt/lod/cascade.rs b/libs/@local/graph/atlas/src/salt/lod/cascade.rs index 826e4ad841f..2c748bbd36a 100644 --- a/libs/@local/graph/atlas/src/salt/lod/cascade.rs +++ b/libs/@local/graph/atlas/src/salt/lod/cascade.rs @@ -1,4 +1,4 @@ -//! The first-occupant cascade, which gives every point a minimum-zoom bucket. +//! First-occupant depth assignments for progressive spatial coverage. //! //! [`buckets`] assigns by scanning the grids coarse to fine, one rank-ordered pass per depth. //! [`separation_buckets`] computes the same assignment at [`Depth::MAX`] in one pass over the @@ -18,25 +18,30 @@ use crate::{ morton::{Depth, MortonKey}, }; -/// Assigns every point its bucket. -/// -/// The bucket is the shallowest grid depth at which the point first occupies its cell. +/// Assigns each point its first unclaimed grid cell's depth, capped at `deepest`. /// /// The cascade scans depths coarse to fine. At each depth, every occupied cell that no -/// earlier-assigned point lies in receives its first still-unassigned point in rank order; the rest -/// continue deeper. Points never claiming a cell - co-located within one deepest-grid cell - take -/// `deepest`, the catch-all bucket, so `deepest` is the one bucket holding more than one point per -/// cell. +/// earlier-assigned point lies in receives its first still-unassigned point in rank order. The rest +/// continue deeper. Points never claiming a cell take `deepest`, the catch-all bucket. Only the +/// catch-all can hold more than one point per cell. +/// +/// The assignment is a pure function of the keys, the ranking, and `deepest`. `ranking` must be a +/// valid permutation of the rows in `keys`. +/// +/// # Properties /// -/// Delivering every point with a bucket at or below a cut depth therefore covers every occupied -/// cell of the cut's grid. [`verify_coverage`] rechecks that claim for one generation. +/// For every cut d ≤ `deepest`, points with bucket at or below d cover every occupied depth-d cell. +/// Before the catch-all cut, exactly one delivered point represents each such cell. /// -/// The assignment is a pure function of the keys, the ranking, and `deepest`, over whatever row -/// domain `R` the two agree on. +/// # Complexity +/// +/// For N points and D = `deepest`, the cascade makes D + 1 rank-ordered passes and uses O(N) +/// storage. With expected constant-time hash-set operations, its time is O(N · (D + 1)). /// /// # Panics /// -/// This panics when `keys` and `ranking` disagree on the row count. +/// Panics when `keys` and `ranking.row_of_rank` disagree on the row count, or when the ranking +/// contains a row outside `keys`. #[must_use] pub(crate) fn buckets( keys: &IdSlice, @@ -49,21 +54,21 @@ pub(crate) fn buckets( "the keys and the ranking must cover the same rows", ); - // Rows that no pass assigns keep `deepest`, the catch-all bucket. + // rows that no pass assigns keep `deepest`, the catch-all bucket let mut buckets = IdVec::::from_elem(deepest, keys.len()); let mut assigned = DenseBitSet::::new_empty(keys.len()); - // A hash set holds the cells. Its elements are `prefix(depth)` keys, and their `4^depth`-cell - // domain outgrows the row count from depth ~10 on while the populated cells stay bounded by the - // rows, so the hash set pays only for the cells the cascade touches. The row set fills a linear - // domain, which a dense bit set fits. + // The occupied cells number at most one per row, while the cell domain grows as 4ᵈ at depth d. + // A hash set allocates for occupied cells. The row set has a linear domain and uses a dense bit + // set. let mut seen = fast_hash_set(); - // One rank-ordered pass per depth suffices with a single cell set. Within any cell an - // assigned point always outranks every still-unassigned point, because every point of the - // current cell sat inside the shallower cell it claimed and lost that claim on rank. An - // assigned point therefore marks its cell before any unassigned visitor arrives. The first - // unassigned visitor of an unmarked cell holds the cell's best still-unassigned rank. + // Every depth-d cell lies inside exactly one cell at each shallower depth. Inductively, a point + // assigned in a shallower cell outranks every still-unassigned point there, including those in + // its depth-d cell. Scanning in rank order marks that cell before any unassigned point can + // claim it. In an unmarked cell, the first visitor has its best remaining rank and establishes + // the same invariant. Therefore one pass per depth suffices to preserve coverage and one + // delivered representative per cell below the catch-all. for depth in 0..=deepest.get() { let depth = Depth::new(depth).expect("every depth at or below `deepest` is a valid depth"); @@ -76,8 +81,7 @@ pub(crate) fn buckets( buckets[row] = depth; assigned.insert(row); } else { - // The cell is already claimed at this depth; the row - // stays unassigned for a deeper pass. + // the occupied cell leaves this row unassigned for a deeper pass } } } @@ -87,20 +91,29 @@ pub(crate) fn buckets( /// Assigns every point its natural bucket by neighbour separation. /// -/// The closed form of [`buckets`] at [`Depth::MAX`], over points sorted ascending by -/// `(key, rank)`: both assign the same bucket to every point. A point's bucket is one past the -/// deepest grid it shares with any better-ranked point, because that is the first depth at which -/// its cell holds no better-ranked occupant. The best-ranked point takes [`Depth::MIN`], and a -/// point sharing its key with a better-ranked point shares every grid and takes [`Depth::MAX`], -/// the catch-all. +/// Computes the closed form of [`buckets`] at [`Depth::MAX`]. `points` must ascend by `(key, rank)` +/// under the accessors, ranks must be pairwise distinct, and the accessors must return consistent +/// values throughout the call. Smaller ranks have higher precedence. The output follows `points` +/// order, using `alloc` for the result and `scratch` for temporary storage. +/// +/// # Properties /// -/// The keys ascend, so the deepest grid a point shares with any better-ranked point is the -/// deepest it shares with the key-nearest better-ranked point on either side. One -/// monotonic-stack pass finds both neighbours, and the assignment costs `O(points)` after the -/// sort that ordered them. +/// For point i, let Dᵢ be the deepest shared grid with any better-ranked point, as measured by +/// [`MortonKey::shared_depth`]. Its bucket is min(Dᵢ + 1, 32), the first depth with no +/// better-ranked occupant, capped at the catch-all. The best-ranked point takes [`Depth::MIN`]. +/// Equal keys share every grid, putting the worse-ranked point in [`Depth::MAX`]. Both this formula +/// and [`buckets`] assign every point the same bucket at the full key width. /// -/// Caller requirement: `points` ascends by `(key, rank)` under the given accessors, and the -/// ranks are pairwise distinct. +/// A Morton prefix occupies a contiguous key interval. If a better-ranked point shares a prefix +/// with i, every intervening key shares it too. The nearest better-ranked point on either side +/// therefore attains the deepest shared grid on that side. It is sufficient to compare these two +/// neighbours. +/// +/// # Complexity +/// +/// The monotonic-stack pass takes O(N) accessor calls and comparisons for N points, plus O(N) +/// result and scratch storage. Each point enters and leaves the stack at most once. Sorting the +/// input is a separate cost. #[must_use] pub(crate) fn separation_buckets_in( points: &[T], @@ -152,7 +165,8 @@ pub(crate) fn separation_buckets_in( /// Assigns every point its natural bucket by neighbour separation. /// -/// See: [`separation_buckets_in`]. +/// Uses the global allocator for both output and scratch storage. Input requirements and the bucket +/// formula are those of [`separation_buckets_in`]. #[must_use] pub(crate) fn separation_buckets( points: &[T], @@ -174,14 +188,18 @@ pub(crate) struct CoverageGap { /// Checks the cascade's coverage contract over one assignment. /// -/// For every depth up to `deepest` and every occupied cell of that depth's grid, at least one point -/// of the cell carries a bucket at or below the depth; delivering the buckets-at-or-below-cut -/// prefix then shows every occupied cell. The cascade guarantees this by construction - the check -/// is the publishable evidence, not a consumer's obligation. +/// Checks that every occupied cell at every depth up to `deepest` has at least one point with a +/// bucket at or below that depth. This checks coverage alone, not the rank choice or representative +/// uniqueness. +/// +/// # Errors +/// +/// Returns a [`CoverageGap`] at the shallowest failing depth, for the first uncovered cell +/// encountered in key-column order. /// /// # Panics /// -/// This panics when `keys` and `buckets` disagree on the row count. +/// Panics when `keys` and `buckets` disagree on the row count. #[cfg(any(test, feature = "bench"))] #[expect( clippy::panic_in_result_fn, diff --git a/libs/@local/graph/atlas/src/salt/lod/key.rs b/libs/@local/graph/atlas/src/salt/lod/key.rs index 83ddea8b36b..c053bb900d5 100644 --- a/libs/@local/graph/atlas/src/salt/lod/key.rs +++ b/libs/@local/graph/atlas/src/salt/lod/key.rs @@ -9,12 +9,16 @@ use crate::{ /// Quantizes each point onto the frame's 32-bit-per-axis grid. /// -/// [`Bounds2::quantize`] maps each point onto the grid. The grid outresolves the data, so only -/// points closer together than one coordinate ULP share a cell. +/// Uses [`Bounds2::quantize`] and preserves input order. For generation indexing, the frame must +/// contain every point. The fixed [`super::stage::WIRE_FRAME`] gives each grid cell an axis width +/// of 2⁻³¹. Quantization in that frame clamps coordinates outside `[-1, 1]` onto its boundary +/// cells. A zero-extent frame axis maps to cell zero. /// -/// The caller owns the frame and guarantees that it contains every point. The frame fit produces -/// such a frame from these coordinates. Coordinates outside the frame clamp onto the boundary -/// cells. +/// # Warning +/// +/// Quantization can merge distinct `f32` coordinates, because its uniform grid does not resolve +/// every representable value near zero and arbitrary frames also inherit the extent rounding of +/// [`Bounds2::quantize`]. #[must_use] pub(crate) fn keys(points: &[Vec2], frame: Bounds2) -> Box<[MortonKey]> { points diff --git a/libs/@local/graph/atlas/src/salt/lod/mod.rs b/libs/@local/graph/atlas/src/salt/lod/mod.rs index 7583d6b347d..0b5147a282e 100644 --- a/libs/@local/graph/atlas/src/salt/lod/mod.rs +++ b/libs/@local/graph/atlas/src/salt/lod/mod.rs @@ -1,17 +1,15 @@ //! The level-of-detail structure of one generation. //! -//! Ranks, Morton keys, buckets, and the base delivery order. +//! [`stage`] derives the spatial serving columns from one generation's coordinate rows. It fits the +//! world frame and normalizes coordinates into the wire frame. It also derives Morton keys +//! ([`key`]) and an importance ranking ([`rank`]). The first-occupant [`cascade`] assigns each +//! point a minimum grid-depth bucket. [`order`] sorts every served column into the base delivery +//! order, and [`quad`] partitions that order into tile runs. //! -//! One generation's points enter as coordinate rows and leave as the columns serving slices from: a -//! deterministic importance ranking ([`rank`]), Morton keys quantized over the generation's frame -//! ([`key`]), a minimum-zoom bucket per point from the first-occupant cascade ([`cascade`]), and -//! the base delivery order that sorts every served column ([`order`]). [`stage`] assembles the -//! whole derivation (frame fit, wire normalization, keys, ranking, cascade, sort, gather) and -//! measures the publish evidence over the result; [`quad`] cuts the finished columns into the quad -//! file's tile topology. -//! -//! Every function here is pure. Equal inputs give byte-equal columns, so a generation's spatial -//! index is reproducible from its coordinates, rank inputs, and seed alone. +//! Rebuilding requires the same coordinates, rank columns, seed, and [`stage::LodConfig`]. Ranking +//! replay also depends on identity-byte encoding and the equal-key ordering described by +//! [`rank::Ranking::new`]. [`stage::Lod::measurements`] and [`quad::QuadTree::measurements`] report +//! build statistics for calibrating the schedule. #[cfg(feature = "bench")] pub mod bench; diff --git a/libs/@local/graph/atlas/src/salt/lod/order.rs b/libs/@local/graph/atlas/src/salt/lod/order.rs index 97408eadf1d..2a5541ed6bd 100644 --- a/libs/@local/graph/atlas/src/salt/lod/order.rs +++ b/libs/@local/graph/atlas/src/salt/lod/order.rs @@ -1,4 +1,4 @@ -//! The base delivery order. +//! Bucket-major ordering for aligned serving columns. use hashql_core::id::{Id as _, IdSlice, IdVec}; @@ -8,13 +8,16 @@ use crate::{ morton::{Depth, MortonKey}, }; -/// The base delivery order of one generation. +/// A row permutation ordered by bucket, then Morton key, then importance rank. /// -/// Bucket-major, Morton within bucket, rank within key ties. +/// Every served column publishes in this order. A bucket occupies one segment, and a tile's keys +/// within that segment occupy a contiguous run. /// -/// Every served column publishes in this order, filters index it, and a tile is a set of contiguous -/// runs of it. The order is a pure function of buckets, keys, and ranking - the third sort -/// component makes ties total, so the permutation is unique and reproducible. +/// # Properties +/// +/// For every valid [`Ranking`], ranks are pairwise distinct. The `(bucket, key, rank)` sort key +/// therefore distinguishes every row. Equal buckets, keys, and ranking give exactly the same +/// permutation and its inverse. #[derive(Debug, PartialEq, Eq)] pub(crate) struct BaseOrder { /// Each base position's row. @@ -23,7 +26,7 @@ pub(crate) struct BaseOrder { pub row_of_position: Box>, /// Each row's base position. /// - /// The row-to-position permutation the filter contract maps entity ids through. + /// The inverse of [`Self::row_of_position`]. pub position_of_row: Box>, } @@ -32,7 +35,8 @@ impl BaseOrder { /// /// # Panics /// - /// This panics when `keys`, `buckets`, and `ranking` disagree on the row count. + /// Panics when `keys`, `buckets`, and `ranking.row_of_rank` disagree on the row count, or when + /// `ranking.rank_of_row` omits a row in `keys`. #[must_use] pub(crate) fn new( keys: &IdSlice, diff --git a/libs/@local/graph/atlas/src/salt/lod/quad.rs b/libs/@local/graph/atlas/src/salt/lod/quad.rs index 42f346944df..c5da4430dac 100644 --- a/libs/@local/graph/atlas/src/salt/lod/quad.rs +++ b/libs/@local/graph/atlas/src/salt/lod/quad.rs @@ -1,29 +1,30 @@ //! The quadtree build, which cuts the base delivery order into tiles. //! -//! [`QuadTree::build`] derives the quad file's regions from the finished lod columns. The tree -//! holds one node for every tile the bucket-cut schedule delivers something new into. Each node -//! carries its own-bucket run of the base order. A node also records the point count of its subtree -//! and the set of direct types under it. +//! [`QuadTree::build`] derives nodes and direct-type sets from finished [`Lod`] columns. A node +//! records the run first delivered at its cut, the population of its whole cell, and every direct +//! type in that cell. Population and type sets include points delivered at shallower cuts. //! -//! The root always exists and covers the wire frame. A deeper cell gets a node exactly when it -//! contains a point the parent tile's cut did not deliver, a point whose bucket is at least the -//! cell's own cut `z + span_log2`. The cascade shapes the tree in two ways: +//! Let m be [`LodConfig::span`], z a tile's depth, and b a point's bucket. The root always exists. +//! A deeper cell gets a node exactly when it contains a point with b ≥ z + m, beyond its parent's +//! cut. A node may have an empty own-bucket run while retaining descendants that deliver later +//! buckets. //! -//! - Chains self-terminate. A point alone in its depth-`d` cell is that cell's best-ranked -//! occupant, so the first-occupant cascade assigns it a bucket no deeper than the first depth -//! where it stands alone. Isolated points never force node chains. -//! - Runs partition the base order. A point with bucket `b` beyond the root's cut shows up in -//! exactly one node's run - its cell at depth `b - span_log2`, which exists because the point -//! itself witnesses the rule - and the root's run carries buckets `0..=span_log2` whole and -//! contiguous, because the base order is bucket-major. The tile pyramid therefore delivers every -//! point exactly once. +//! # Partition and termination //! -//! The recursion needs no depth cap. The cascade assigns no bucket beyond `max_tile_depth + -//! span_log2`, so leaves at the deepest tile zoom fall out by construction. +//! Bucket-major order makes buckets 0 through m one contiguous root run. Every point with b > m +//! belongs to exactly one cell at depth b − m, and that point requires its cell's node to exist. +//! That node's run selects bucket b. Therefore the node runs partition the base order and deliver +//! every point exactly once. //! -//! The partition-point searches of `file/morton`'s `run` query narrow each node's run (first code -//! at or above the cell's minimum key, first beyond its maximum), so the builder and the served -//! lookups can never disagree about a run's extent. +//! The cascade assigns a point no later than the first grid where it stands alone. Such a point +//! needs no descendant node once the tile cut reaches its bucket. More generally, every bucket is +//! at or below [`LodConfig::deepest`]. At the maximum tile depth, no point remains beyond the cut, +//! and recursion terminates without a separate depth cap. +//! +//! Each Morton cell is one inclusive key interval. Within a sorted bucket, the first code at or +//! above the cell's minimum and the first code beyond its maximum delimit that cell's run. Both the +//! builder and [`MortonFile::run`](crate::file::morton::read::MortonFile::run) use these boundaries +//! over the same codes. Their run extents agree. use alloc::collections::BTreeSet; use core::ops::Range; @@ -44,7 +45,7 @@ use crate::{ morton::{Depth, MortonCell, MortonKey}, }; -/// Building the quadtree failed. +/// A schedule, column, or encoding limit that prevents quadtree construction. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum QuadError { /// The configuration names a schedule no 64-bit key resolves. @@ -52,12 +53,10 @@ pub(crate) enum QuadError { /// The type column covers a different row count than the lod columns. Columns { rows: usize }, /// The lod columns hold points in a bucket beyond the configuration's deepest grid. - /// - /// The lod ran under a different configuration. Bucket { bucket: u8 }, /// A direct type names an ontology row beyond the `u32` ordinals the quad file stores. TypeOrdinal { row: NodeRowId, id: u64 }, - /// The tree needs more nodes than `u32` indexes address. + /// A node index would reach or exceed the absent-child sentinel. Nodes, } @@ -91,11 +90,20 @@ impl core::fmt::Display for QuadError { impl core::error::Error for QuadError {} -/// The quad file's regions for one generation, in writable form. +/// A quadtree's node table and per-node direct-type sets. +/// +/// [`Self::build`] places the root at index zero and records nodes in depth-first pre-order, with +/// children in Morton order. Every child index points farther into the table. +/// +/// # Memory usage /// -/// Node 0 is the root; records are in depth-first pre-order with children in Morton child order, so -/// every child index points deeper in the table. [`measurements`](Self::measurements) reads the -/// finished tree and yields the numbers that belong in the generation's metadata document. +/// Each node retains its cell's complete direct-type set. A type present along a depth-h branch can +/// repeat in all h + 1 node sets. +/// +/// # Panics +/// +/// Writing panics if the node count is at least [`Node::NO_CHILD`], if `sets` covers a different +/// node count, or if a child index lies outside the table or does not follow its parent. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct QuadTree { /// The node table in depth-first pre-order. @@ -109,21 +117,23 @@ pub(crate) struct QuadTree { impl QuadTree { /// Builds the quadtree over the finished lod columns. /// - /// `types` holds each row's direct types in **row** order, exactly as the dataset streams them - /// (ascending, deduplicated). The builder gathers them into base order through the lod's - /// permutation. + /// `types` holds each row's direct types in row order. Each node receives the sorted, + /// deduplicated union for its whole cell. `lod` must retain the permutation, fencepost, and + /// sorted-code invariants of [`Lod::build`]. /// - /// `config` must be the configuration the lod ran under. A mismatch surfaces as - /// [`QuadError::Bucket`] when the lod's cascade ran deeper than the configuration allows, and - /// nothing else here detects it. + /// `config` must be the configuration the lod ran under. A bucket beyond the supplied deepest + /// grid produces [`QuadError::Bucket`]. Other valid configuration mismatches can change the + /// tree without an error. /// /// # Errors /// - /// Returns [`QuadError::Schedule`] when the configuration exceeds the key width, - /// [`QuadError::Columns`] when the type column disagrees with the lod columns, - /// [`QuadError::Bucket`] when the lod holds points beyond the configuration's deepest grid, - /// [`QuadError::TypeOrdinal`] when a direct type escapes the file's `u32` ordinals, and - /// [`QuadError::Nodes`] when the tree escapes `u32` node indexes. + /// Returns a [`QuadError`] for an invalid schedule, incompatible columns, or a row or node + /// index beyond its encoding. + /// + /// # Panics + /// + /// Panics if the lod's row permutation or fenceposts address elements outside their + /// corresponding columns. #[tracing::instrument(skip_all)] pub(crate) fn build( lod: &Lod, @@ -143,8 +153,7 @@ impl QuadTree { } } - // Gather the type column into base order once, so the recursion - // touches each position's types without indirection. + // gathering once removes row indirection from the recursive type unions let position_types = lod .row_of_position .iter() @@ -198,14 +207,6 @@ impl WriteAs for QuadTree {} impl WriteInto for QuadTree { type Error = io::Error; - /// Writes the tree as a quad file. - /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. - /// - /// # Errors - /// - /// Returns an error when the underlying writer fails. fn write_into(&self, write: impl io::Write) -> io::Result { let mut writer = Writer { accumulator: Sha256::new(), @@ -235,9 +236,10 @@ pub(crate) struct QuadMeasurements { pub type_entries: u64, } -/// The positions of each bucket's codes inside the cell under construction, one range per bucket. +/// Per-bucket ranges of codes inside the cell under construction. type BucketRanges = [Range; SEGMENTS]; +/// An in-progress quad tree in depth-first pre-order, with one type set per node. struct Builder<'lod> { /// The code column in base order, segment-sorted. codes: &'lod IdSlice, @@ -260,7 +262,17 @@ struct Builder<'lod> { impl Builder<'_> { /// Builds the node for `cell` over the bucket `ranges` narrowed to it. /// - /// Returns the node's table index. + /// Returns the node's table index. `ranges` must delimit this cell within each sorted bucket, + /// and the cell's cut must not exceed `self.deepest`. + /// + /// # Errors + /// + /// Returns [`QuadError::Nodes`] if the next node index cannot fit below [`Node::NO_CHILD`]. + /// + /// # Panics + /// + /// Panics for ranges outside the code or type columns, a population exceeding `u32`, or a cell + /// whose cut exceeds `self.deepest`. fn node(&mut self, cell: MortonCell, ranges: &BucketRanges) -> Result { let Ok(index) = u32::try_from(self.nodes.len()) else { return Err(QuadError::Nodes); @@ -269,7 +281,7 @@ impl Builder<'_> { return Err(QuadError::Nodes); } - // Reserve the pre-order slot before the children claim theirs. + // reserve the parent's pre-order slot before appending its descendants self.nodes.push(Node::new([None; 4], 0, 0, 0)); self.sets.push(Vec::new()); self.depth = self.depth.max(cell.depth()); @@ -284,8 +296,7 @@ impl Builder<'_> { let mut children = [None; 4]; let mut set = BTreeSet::new(); if self.exhausted(cut, ranges) { - // This tile's cut delivers every point in the cell. The cell is a leaf, and all of its - // points feed the type set directly. + // the leaf has no child sets to union: gather every point in its cell for range in ranges { self.gather(&mut set, range.clone()); } @@ -297,8 +308,8 @@ impl Builder<'_> { for (quadrant, child_cell) in cells.into_iter().enumerate() { let child_ranges = self.narrow(ranges, child_cell); if self.exhausted(cut, &child_ranges) { - // This quadrant needs no node, so its points contribute their types here, at - // the deepest node containing them. + // this quadrant's points contribute directly at the deepest node containing + // them for range in &child_ranges { self.gather(&mut set, range.clone()); } @@ -323,8 +334,12 @@ impl Builder<'_> { /// Returns the own-bucket run. /// - /// Bucket `z + span_log2` for a tile at zoom `z`, buckets `0..=span_log2` whole for the root - - /// a single contiguous range because the base order is bucket-major. + /// Selects bucket `z + span_log2` at depth `z`, or all buckets `0..=span_log2` for the root. + /// Bucket-major order makes the root's whole segments contiguous. + /// + /// # Panics + /// + /// Panics when `depth + self.span_log2` exceeds [`Depth::MAX`]. fn run(&self, depth: Depth, ranges: &BucketRanges) -> Range { if depth == Depth::MIN { let cut = usize::from(self.span_log2); @@ -344,7 +359,11 @@ impl Builder<'_> { /// Returns whether `ranges` holds no point in a bucket beyond `cut`. /// - /// Nothing below this tile's zoom delivers anything new. + /// `cut` must be at or below `self.deepest`. + /// + /// # Panics + /// + /// Panics when `cut` exceeds `self.deepest`. fn exhausted(&self, cut: u8, ranges: &BucketRanges) -> bool { ranges[usize::from(cut) + 1..=usize::from(self.deepest.get())] .iter() @@ -353,13 +372,17 @@ impl Builder<'_> { /// Narrows every bucket's range to the codes inside `cell`. /// - /// By the partition-point searches of `file/morton`'s `run` query. A partition point over the - /// range's sub-slice is an offset into the range, which the range's own start rebases into a - /// base position. + /// Each range must delimit sorted codes. The result uses base positions rather than offsets + /// within a range. + /// + /// # Panics + /// + /// Panics when a range lies outside `self.codes`. fn narrow(&self, ranges: &BucketRanges, cell: MortonCell) -> BucketRanges { core::array::from_fn(|bucket| { let range = &ranges[bucket]; let slice = &self.codes[range.clone()]; + // partition points are relative to `slice`: add the enclosing range's base position let start = range .start .plus(slice.partition_point(|&code| code < cell.min_key())); @@ -370,7 +393,11 @@ impl Builder<'_> { }) } - /// Feeds the types of every position in `range` into `set`. + /// Adds the direct types of every position in `range` to `set`. + /// + /// # Panics + /// + /// Panics when `range` addresses a position outside `self.position_types`. fn gather(&self, set: &mut BTreeSet, range: Range) { for position in range { set.extend(self.position_types[position].iter().copied()); diff --git a/libs/@local/graph/atlas/src/salt/lod/rank.rs b/libs/@local/graph/atlas/src/salt/lod/rank.rs index b0d57b97a38..133237abe37 100644 --- a/libs/@local/graph/atlas/src/salt/lod/rank.rs +++ b/libs/@local/graph/atlas/src/salt/lod/rank.rs @@ -1,4 +1,4 @@ -//! The deterministic importance ranking. +//! Importance ordering with seeded identity tie-breaking. use hashql_core::id::{Id, IdSlice, IdVec}; use rayon::iter::ParallelIterator as _; @@ -9,12 +9,11 @@ use crate::{ integrity::{Sha256, Update as _}, }; -/// The per-row inputs of the rank pass, one entry per point row. +/// Equal-length importance, priority, and identity columns for ranking point rows. /// -/// Rows rank by configured importance, then stable semantic priority, then a seeded hash of the -/// entity identity, so the order is total and reproducible from the columns and the seed alone. The -/// columns are row-indexed, equal-length by construction, and the row count fits the `u32` row -/// encoding. `I` is the dataset's node id type. The ranking consumes its canonical bytes alone. +/// The row count fits `u32`. [`Ranking::new`] orders the scores and uses a seeded hash of each +/// identity's bytes to break score ties. Construction checks lengths alone, accepting every score +/// bit pattern and repeated identities. #[derive(Debug, Copy, Clone)] pub(crate) struct RankInputs<'columns, I> { importance: &'columns IdSlice, @@ -23,7 +22,7 @@ pub(crate) struct RankInputs<'columns, I> { } impl<'columns, I> RankInputs<'columns, I> { - /// Wraps the rank columns. + /// Checks that the columns cover the same `u32`-sized row domain. /// /// Returns [`None`] when the columns disagree on length or the row count does not fit the `u32` /// row encoding. @@ -59,12 +58,11 @@ impl<'columns, I> RankInputs<'columns, I> { } } -/// The rank order of one row universe: a permutation and its inverse. +/// A row permutation and its inverse, with rank zero first in importance order. /// -/// Rank 0 is the most important row. The cascade consumes the ascending direction to claim cells. -/// The published columns record each position's rank through the inverse. The row domain `R` is -/// whatever universe the ranking orders - the generation's rows at fit time, or a view's own row -/// vocabulary when a scope ranks its visible subset. A ranking applies only to the rows it ranked. +/// `row_of_rank` must contain each row exactly once, and `rank_of_row` must be its inverse over the +/// same domain. `R` identifies that domain: generation rows or the local rows of a visible subset. +/// A ranking applies only to the rows it ranked. #[derive(Debug, PartialEq, Eq)] pub(crate) struct Ranking { /// Row by rank: `row_of_rank[rank]` is the row holding that rank. @@ -79,8 +77,12 @@ where { /// Completes a ranking from its filled rank order. /// - /// `row_of_rank` must be a permutation of the row universe it ranks. The inverse view follows - /// from it. + /// `row_of_rank` must be a permutation of the row universe it ranks. Duplicate rows overwrite + /// their inverse entries and leave other entries at rank zero. + /// + /// # Panics + /// + /// Panics when a row index is at or beyond `row_of_rank.len()`. #[must_use] pub(crate) fn from_row_of_rank(row_of_rank: IdVec) -> Self { let mut rank_of_row = IdVec::from_elem(ImportanceRank::MIN, row_of_rank.len()); @@ -98,13 +100,18 @@ where impl Ranking { /// Ranks the rows by descending importance. /// - /// Then descending priority, then the seeded identity hash ascending. + /// Ties compare by descending priority, then ascending seeded identity hash. Scores use + /// [`f32::total_cmp`], including its ordering of signed zeros and NaNs. + /// + /// Equal columns and seed reproduce the ranking with the current sorting implementation. Hashes + /// have only 64 bits: equal scores and hashes leave a tie whose relative order can change with + /// the sorting implementation. Cross-target replay also requires identical identity bytes on + /// each target. [`IntoBytes`] alone does not establish a canonical byte order. + /// + /// # Complexity /// - /// Scores compare under IEEE 754 `totalOrder`, so the ranking is total and deterministic for - /// every bit pattern; both score columns arrive finite - importance by the - /// [`ImportanceSignal`](crate::salt::importance::ImportanceSignal) contract, priority as a - /// constant column until it grows a source - and nothing here re-checks them. Equal seeds give - /// equal rankings. The generation's metadata records the seed. + /// For N rows, sorting costs O(N log N) comparisons and O(N) storage. Hashing additionally + /// reads every identity byte once. #[must_use] pub(crate) fn new(inputs: RankInputs<'_, I>, seed: u64) -> Self where @@ -118,9 +125,7 @@ impl Ranking { let mut row_of_rank: IdVec<_, _> = inputs.identities.ids().collect(); row_of_rank.par_sort_unstable_by(|&left, &right| { - // Descending importance, then descending priority: the - // reversed comparisons spell the descending lexicographic - // key over the scores. + // reverse only the score comparisons: smaller hashes retain precedence inputs.importance[right] .total_cmp(&inputs.importance[left]) .then_with(|| inputs.priority[right].total_cmp(&inputs.priority[left])) @@ -133,9 +138,9 @@ impl Ranking { /// Hashes one entity identity under the ranking seed. /// -/// The first eight digest bytes, little endian, of the SHA-256 over the seed followed by the -/// identity bytes. Identities are unique per row, SHA-256 is collision-resistant at this width for -/// corpus-scale row counts, and a new seed reshuffles every tie deterministically. +/// Interprets the first eight digest bytes as a little-endian `u64`. The digest input is the +/// little-endian seed followed by the identity's bytes. Changing the seed changes that input, +/// without guaranteeing a different hash or tie order. #[expect( clippy::little_endian_bytes, reason = "the hash is pinned to the same canonical little-endian bytes on every platform" @@ -146,5 +151,6 @@ fn tiebreak(seed: u64, identity: &I) -> u64 hasher.update(identity.as_bytes()); let digest = hasher.finalize().to_bytes(); + // SHA-256 returns 32 bytes, of which this fixed slice selects exactly eight. u64::from_le_bytes(digest[..8].try_into().expect("eight bytes are eight bytes")) } diff --git a/libs/@local/graph/atlas/src/salt/lod/stage.rs b/libs/@local/graph/atlas/src/salt/lod/stage.rs index 81f6139ff81..915992dfd16 100644 --- a/libs/@local/graph/atlas/src/salt/lod/stage.rs +++ b/libs/@local/graph/atlas/src/salt/lod/stage.rs @@ -1,8 +1,5 @@ //! The lod stage, which derives the served columns from canonical coordinates. //! -//! The result is a pure function of the coordinates, the rank inputs, the seed, and the -//! configuration, so equal generations produce byte-equal columns. -//! //! [`Lod::build`] runs the whole level-of-detail derivation for one generation: //! //! 1. Fit the world frame. @@ -38,8 +35,8 @@ use crate::{ /// The fixed frame every wire coordinate lives in. /// -/// The world frame normalizes onto it at publish, and the online placement path re-derives the -/// identical map from the recorded world frame and this constant. +/// [`Bounds2::normalize_into`] maps the fitted world frame onto this `[-1, 1]` square. The world +/// frame and this constant specify the same map for later point placement. pub(crate) const WIRE_FRAME: Bounds2 = Bounds2::new(Vec2::new(-1.0, -1.0), Vec2::new(1.0, 1.0)) .expect("the wire frame corners are finite and ordered"); @@ -54,10 +51,9 @@ const DEFAULT_SPAN: Log2 = Log2::new(6).expect("6 lies below the shift width"); pub(crate) struct LodConfig { /// Cells per tile axis of the delivery cut, as its base-2 log. /// - /// A tile at zoom `z` delivers buckets at or below `z + span`, sampling a `2^span` by `2^span` - /// grid per tile. Regular buckets deliver at most `4^span` points per incremental tile, and the - /// deepest catch-all may exceed that cap by its co-located residue, measured as - /// [`LodMeasurements::co_location_excess`]. + /// At zoom z, the cumulative cut includes buckets at or below z + m, where m is this span. Each tile contains a 2ᵐ by 2ᵐ cut grid. Before the catch-all, one representative per occupied cut cell bounds both cumulative and incremental delivery by 4ᵐ points per tile. The catch-all includes every remaining point and can exceed that cap. + /// + /// By default m = 6, or 64 cells per axis. pub span: Log2 = DEFAULT_SPAN, /// The deepest tile zoom the schedule serves. /// @@ -78,9 +74,8 @@ impl LodConfig { /// /// `max_tile_depth + span`, the catch-all bucket of the cut schedule. /// - /// Returns [`None`] when the sum exceeds the 32 subdivisions a 64-bit Morton key resolves - the - /// key-width inequality `z_max + m ≤ 32` - in which case the configuration matches no - /// buildable schedule. + /// Returns [`None`] when the sum exceeds the 32 subdivisions a 64-bit Morton key resolves. For + /// maximum tile zoom zₘₐₓ and span m, a buildable schedule requires zₘₐₓ + m ≤ 32. #[must_use] pub(crate) const fn deepest(self) -> Option { let Some(sum) = self.span.get().checked_add(self.max_tile_depth) else { @@ -91,7 +86,7 @@ impl LodConfig { } } -/// Building the lod structure failed. +/// A schedule or input-column condition that prevents LOD construction. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum LodError { /// The configuration names a schedule no 64-bit key resolves. @@ -101,7 +96,7 @@ pub(crate) enum LodError { /// Columns disagreeing among themselves cannot reach here: [`RankInputs`] admits only /// equal-length columns. Columns { coordinates: usize }, - /// The coordinates hold no rows, so no world frame exists. + /// The coordinates hold no rows to fit a world frame from. Frame, } @@ -148,31 +143,30 @@ pub(crate) struct LodMeasurements { pub catch_all_population: u64, /// Catch-all points beyond one per distinct deepest-grid cell. /// - /// The population no cut depth can thin. + /// C − G, where C is the catch-all population and G counts the deepest-grid cells represented + /// within that bucket. Lower-bucket occupants of those cells contribute to neither count. pub co_location_excess: u64, - /// The largest own-bucket delta any tile of the schedule delivers. + /// The largest single-bucket population within a tile at that bucket's first zoom. /// - /// Verified against the geometric cap `4^span`; co-location can exceed the cap only in - /// the catch-all bucket. + /// Each bucket b uses tile depth max(b − m, 0), where m is the configured span. Buckets at or + /// below m are measured separately, although the root delivers their sum. This statistic can + /// therefore undercount the root's delivered delta. pub max_tile_delta: u64, } -/// The level-of-detail structure of one generation, every column in base delivery order. +/// Aligned spatial serving columns and row permutations for one generation. /// -/// The serving artifacts are the wire coordinate column, the Morton code column with its bucket -/// fenceposts, the rank column, and the row permutations. [`measurements`](Self::measurements) -/// reads the finished columns and yields the numbers that belong in the generation's metadata -/// document. +/// Coordinates, Morton codes, and importance ranks follow the [`BaseOrder`] permutation. The +/// fenceposts delimit the same bucket segments in each column. The inverse permutations translate +/// rows, ranks, and base positions without searching. #[derive(Debug, PartialEq)] pub(crate) struct Lod { /// The world frame the normalization mapped onto the wire frame. /// - /// Together with the fixed `[-1, 1]` wire frame this is the frame transform: the manifest - /// records it, clients and the placement path re-derive the identical map from it. + /// Together with [`WIRE_FRAME`], this specifies the per-axis normalization for subsequent + /// point placement. pub world: Bounds2, /// Wire coordinates in base order: the canonical coordinates normalized into the wire frame. - /// - /// This column is the wire. pub coordinates: Box>, /// Morton codes in base order, segmented by [`Self::fenceposts`]. pub codes: Box>, @@ -180,9 +174,9 @@ pub(crate) struct Lod { pub fenceposts: Fenceposts, /// Each base position's importance rank. pub rank_of_position: Box>, - /// Each rank's base position: the traversal order of filter registration. + /// Each rank's base position, for traversing the columns in importance order. pub position_of_rank: Box>, - /// Each row's base position: the permutation the filter contract maps entity bitmaps through. + /// Each row's base position, the inverse of [`Self::row_of_position`]. pub position_of_row: Box>, /// Each base position's row. /// @@ -196,16 +190,21 @@ impl Lod { /// `coordinates` is the canonical column in row order, proven finite by its type. `inputs` /// holds the per-row rank columns and `seed` the generation's reproducibility seed. /// - /// The build fits the world frame from the coordinates and normalizes each axis onto `[-1, 1]` - /// in `f64` with one final rounding, so the wire column is within `2^-23` of exact everywhere - /// and reproducible across targets. Keys quantize the normalized column, not the input, so wire - /// coordinates and tile cells can never disagree. + /// The build fits a tight world frame and maps each axis onto `[-1, 1]` with + /// [`Bounds2::normalize_into`]. For an axis with minimum a and positive extent e, the exact map + /// is 2 · (x − a) / e − 1. A zero-extent axis maps to zero. The computation uses `f64` + /// intermediates and a final narrowing to `f32`, giving an absolute coordinate error at most + /// 2⁻²³ within the fitted frame. The bound is absolute, and near zero it amounts to many ULPs + /// of the result. + /// + /// Morton keys quantize the resulting wire coordinates, keeping the spatial index tied to the + /// published column. [`Ranking::new`] specifies the byte-encoding and equal-key conditions for + /// ranking replay. /// /// # Errors /// - /// Returns [`LodError::Schedule`] when the configuration exceeds the key width, - /// [`LodError::Columns`] when the rank columns disagree with the coordinates, and - /// [`LodError::Frame`] when the coordinates hold no rows. + /// Returns a [`LodError`] for an invalid schedule, mismatched coordinate and rank counts, or an + /// empty coordinate column. pub(crate) fn build( coordinates: &FinitePointField, inputs: RankInputs<'_, I>, @@ -222,7 +221,7 @@ impl Lod { }); } - // Finite input leaves emptiness as the one way no frame exists. + // finite input leaves emptiness as the only reason no frame exists let world = Bounds2::from_slice_par(coordinates.as_slice().as_raw()).ok_or(LodError::Frame)?; let normalized = world.normalize_into(WIRE_FRAME, coordinates.as_slice().as_raw()); @@ -236,10 +235,9 @@ impl Lod { let buckets = cascade::buckets(keyed, &ranking, deepest); let order = BaseOrder::new(keyed, &buckets, &ranking); - // row_of_position is the gather order. Walking it assembles any row-ordered column into - // base delivery order. Each gather is an index swizzle whose element work is one copy, so - // parallelism pays per column, not per element. position_of_rank composes the permutations - // by taking a rank's row and then that row's base position. + // row_of_position gathers each row-ordered column into base order. Parallelizing by column + // keeps each task a sequential gather of copied values. position_of_rank composes + // rank-to-row with row-to-position to preserve importance traversal. let mut coordinates = IdVec::::new(); let mut codes = IdVec::::new(); @@ -297,30 +295,32 @@ impl Lod { /// Measures the finished columns for the generation metadata. /// - /// The manifest records the bucket histogram (whose tail calibrates `max_tile_depth`), the - /// catch-all population and its co-location excess, and the observed per-tile own-bucket - /// maximum against the geometric cap. + /// `config` must be the configuration used by [`Self::build`]. A different valid schedule + /// changes the catch-all and tile-depth choices without detecting the mismatch. + /// [`LodMeasurements`] defines each statistic, including the separate treatment of the root's + /// buckets. + /// + /// # Complexity + /// + /// The scans take O(N + D) time for N points and deepest grid D, using constant additional + /// storage. /// /// # Panics /// - /// This panics when `config` is not the configuration the structure ran under, which shows up - /// as an unbuildable schedule. + /// Panics when `config.deepest()` is [`None`], or when the fenceposts address codes outside the + /// column. #[must_use] pub(crate) fn measurements(&self, config: LodConfig) -> LodMeasurements { let deepest = config .deepest() .expect("the structure was built under this configuration"); - // Codes sort within every segment, so cell populations are consecutive equal-prefix groups. - // One linear scan per measurement suffices. + // sorted segment codes put each cell's population in one consecutive equal-prefix group let catch_all = self.segment_codes(deepest); let catch_all_population = catch_all.len() as u64; let co_location_excess = catch_all_population - distinct_prefixes(catch_all, deepest); - // Per bucket, the largest number of its points sharing one - // tile of the bucket's own zoom; the tile grid sits span - // above the bucket's grid, and buckets at or below span - // belong to the zoom-0 root tile. + // each bucket uses its first tile zoom; the root's buckets are scanned separately let mut max_tile_delta = 0; for bucket in 0..=deepest.get() { let tile = Depth::new(bucket.saturating_sub(config.span.get())) @@ -341,17 +341,24 @@ impl Lod { /// Borrows one bucket's slice of the code column. /// - /// The returned slice re-bases at the segment, so its indices are bucket offsets rather than - /// base positions. The helpers scanning it consume values alone. + /// Indices in the returned slice are bucket offsets rather than base positions. + /// + /// # Panics + /// + /// Panics when the fenceposts address codes outside the column. fn segment_codes(&self, bucket: Depth) -> &[MortonKey] { &self.codes[self.fenceposts.segment(bucket)] } } -/// The morton artifact of a built lod: the index, fencepost, and code regions. +/// Borrowed bucket segments and Morton codes for writing a spatial index. +/// +/// Writing emits a Morton file with one index key per [`PAGE_STRIDE`] codes. /// -/// Borrows the finished columns; writing streams them as one morton file under the production -/// page-filling index stride. +/// # Panics +/// +/// Writing panics if the code count differs from the fencepost count or if codes decrease within +/// any bucket segment. pub(crate) struct MortonColumn<'lod> { /// The bucket segmentation of the code column. pub fenceposts: &'lod Fenceposts, @@ -364,14 +371,6 @@ impl WriteAs for MortonColumn<'_> {} impl WriteInto for MortonColumn<'_> { type Error = io::Error; - /// Writes the columns as a morton file. - /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. - /// - /// # Errors - /// - /// Returns an error when the underlying writer fails. fn write_into(&self, write: impl io::Write) -> io::Result { let mut writer = Writer { accumulator: Sha256::new(), diff --git a/libs/@local/graph/atlas/src/salt/lod/tests.rs b/libs/@local/graph/atlas/src/salt/lod/tests.rs index dac8f79b5d8..05a41d66956 100644 --- a/libs/@local/graph/atlas/src/salt/lod/tests.rs +++ b/libs/@local/graph/atlas/src/salt/lod/tests.rs @@ -19,11 +19,18 @@ use crate::{ postgres::id::ArchivedEntityId, }; -/// Pins raw fixture coordinates as the proven-finite corpus field. +/// Borrows fixture coordinates as a finite point field. +/// +/// # Panics +/// +/// Panics if a coordinate is not finite. fn finite(points: &[Vec2]) -> &FinitePointField { FinitePointField::new(IdSlice::from_raw(points)).expect("the fixture coordinates are finite") } +/// Derives distinct entity identities from the index's UUID representation. +/// +/// The web UUID encodes `index` and the entity UUID encodes 31 · `index` modulo 2¹²⁸. fn identity(index: u128) -> ArchivedEntityId { ArchivedEntityId { web_id: Uuid::from_u128(index).into(), @@ -31,11 +38,12 @@ fn identity(index: u128) -> ArchivedEntityId { } } +/// Generates `count` distinct identities from consecutive indexes. fn identities(count: u128) -> Vec { (0..count).map(identity).collect() } -/// Typed rank columns over raw fixture slices. +/// Checks raw fixture columns with [`RankInputs::new`]. fn rank_inputs<'columns>( importance: &'columns [f32], priority: &'columns [f32], @@ -48,7 +56,13 @@ fn rank_inputs<'columns>( ) } -/// A ranking straight from a hand-written rank order. +/// Constructs both ranking directions from a hand-written row permutation. +/// +/// `row_of_rank` must contain each row exactly once. +/// +/// # Panics +/// +/// Panics when a row index is at or beyond `row_of_rank.len()`. fn ranking_of(row_of_rank: &[u32]) -> Ranking { let row_of_rank: Vec = row_of_rank .iter() @@ -68,10 +82,20 @@ fn ranking_of(row_of_rank: &[u32]) -> Ranking { } } +/// Checks a literal subdivision depth. +/// +/// # Panics +/// +/// Panics above [`Depth::MAX`]. fn depth(value: u8) -> Depth { Depth::new(value).expect("test depths lie within the documented domain") } +/// Checks a literal span exponent. +/// +/// # Panics +/// +/// Panics at or above the 64-bit shift width. fn log2(value: u8) -> Log2 { Log2::new(value).expect("test spans lie below the shift width") } @@ -109,7 +133,7 @@ fn rank_orders_by_importance_then_priority_then_tiebreak() { #[test] fn seed_reshuffles_ties() { - // All scores tie, so the order is the seeded hash's alone. + // equal score columns leave only the seeded hashes to order these rows let importance = [1.0_f32; 8]; let priority = [1.0_f32; 8]; let ids = identities(8); @@ -136,8 +160,7 @@ fn rank_inputs_reject_disagreeing_columns() { #[test] fn non_finite_scores_rank_deterministically() { - // The dataset contract keeps scores finite. `totalOrder` keeps the pass total and reproducible - // even when a dataset breaks that contract. + // total_cmp orders NaNs and infinities without requiring finite scores let importance = [f32::NAN, 1.0, f32::NAN, f32::INFINITY]; let priority = [0.0_f32; 4]; let ids = identities(4); @@ -167,7 +190,7 @@ fn keys_quantize_the_frame_corners_center_and_degenerate_axis() { assert_eq!(keys[0].coordinates(), [1 << 31, 0]); } -/// Keys whose depth-1 and depth-2 cells are hand-picked. +/// Constructs keys with prescribed depth-1 and depth-2 cells. /// /// Axis values place their top two bits at (depth-1 quadrant, depth-2 sub-cell). fn hand_keys() -> [MortonKey; 4] { @@ -251,9 +274,7 @@ fn base_order_sorts_buckets_then_keys_then_ranks() { [0, 1, 2, 3].map(BasePosition::from_u32), ); - // Reversing the two catch-all rows' keys is invisible to the sort - // only if rank breaks the tie: give them one key and check rank - // order decides. + // equal catch-all keys leave rank as the deciding component let tied = [keys[0], keys[1], keys[2], keys[2]]; let tied = IdSlice::::from_raw(&tied); let ranking = ranking_of(&[0, 1, 3, 2]); @@ -283,8 +304,8 @@ fn separation_assigns_the_hand_computed_natural_buckets() { let buckets = cascade::separation_buckets(&points, |point| point.0, |point| point.1); - // a claims the whole domain. c shares a's cells through depth 31, so it first claims at 32. - // d shares depth 1 with a and c and claims depth 2. b parts from everything at the first + // a claims the whole domain. For c, the deepest shared grid with a is D = 31, giving bucket D + + // 1 = 32. For d, D = 1 gives bucket 2. b separates from every other point at the first // subdivision and claims depth 1. assert_eq!(*buckets, [depth(0), depth(32), depth(2), depth(1)]); @@ -298,7 +319,6 @@ fn separation_assigns_the_hand_computed_natural_buckets() { assert_eq!(*buckets, [depth(0), depth(32), depth(1)]); } -/// The neighbour-separation closed form is the cascade at the full key width. #[property_test] fn separation_is_the_cascade_at_the_key_width( #[strategy = proptest::collection::vec(any::(), 1..48)] bits: Vec, @@ -322,9 +342,6 @@ fn separation_is_the_cascade_at_the_key_width( } } -/// Coverage holds for every input. -/// -/// Each occupied cell at each depth of the schedule keeps a representative in the delivered prefix. #[property_test] fn cascade_coverage_is_total( #[strategy = proptest::collection::vec(any::(), 1..48)] bits: Vec, @@ -341,7 +358,6 @@ fn cascade_coverage_is_total( prop_assert_eq!(cascade::verify_coverage(keyed, &buckets, deepest), Ok(())); } -/// Below the catch-all, a bucket holds at most one point per cell of its own grid. #[property_test] fn buckets_claim_cells_once( #[strategy = proptest::collection::vec(any::(), 1..48)] bits: Vec, @@ -376,9 +392,6 @@ fn buckets_claim_cells_once( } } -/// The base order is the unique (bucket, key, rank) sort. -/// -/// Its permutations invert each other. #[property_test] fn base_order_is_the_unique_total_sort( #[strategy = proptest::collection::vec(any::(), 1..48)] bits: Vec, @@ -411,9 +424,11 @@ fn base_order_is_the_unique_total_sort( } } -/// A deterministic ranking for property tests. +/// Ranks equal-score rows by seeded identity hashes. +/// +/// # Panics /// -/// Rows ranked by the seeded tiebreak alone, through the real rank pass. +/// Panics when `rows` exceeds `u32::MAX`. fn seeded_ranking(rows: usize, seed: u64) -> Ranking { let importance = vec![0.0_f32; rows]; let priority = vec![0.0_f32; rows]; @@ -425,7 +440,7 @@ fn seeded_ranking(rows: usize, seed: u64) -> Ranking { #[test] fn lod_config_carries_the_key_width_bound() { - // The default schedule reaches the f32 resolution depth. + // the default grid reaches depth 24, with wire-axis cell width 2⁻²³ let config = LodConfig::default(); assert_eq!(config.span.get(), 6); assert_eq!(config.max_tile_depth, 18); @@ -456,12 +471,10 @@ fn lod_config_carries_the_key_width_bound() { ); } -/// The hand stage fixture. +/// Builds a layout with one regular depth-2 claim and one co-resident catch-all point. /// -/// The comments below compute the wire positions and cascade buckets of its four points. -/// -/// World frame [0, 1] x [0, 1]; `span` = 1, `max_tile_depth` = 1, so the deepest grid is 2 and the -/// catch-all is bucket 2. +/// The world frame is [0, 1] × [0, 1]. With `span` = 1 and `max_tile_depth` = 1, the deepest grid +/// and catch-all bucket are both 2. The table gives the exact fixture coordinates and cells: /// /// ```text /// row world wire depth-1 quadrant depth-2 cell @@ -471,10 +484,10 @@ fn lod_config_carries_the_key_width_bound() { /// 3 (0.375, 0.25) (-0.25, -0.5) (0, 0) (1, 1) /// ``` /// -/// Importance ranks the rows in index order. In the cascade, row 0 claims the domain, row 1 its -/// depth-1 quadrant, row 2 the (1, 1) depth-2 cell, and row 3 - co-resident with row 2 at the -/// deepest grid - takes the catch-all. Buckets [0, 1, 2, 2]; the base order is the row order (row -/// 2's key sorts below row 3's). +/// Importance ranks the rows in index order. Row 0 claims the domain and row 1 its depth-1 +/// quadrant. Row 2 claims the (1, 1) depth-2 cell, while co-resident row 3 takes the catch-all. The +/// buckets are [0, 1, 2, 2]. Row 2's key sorts below row 3's, keeping the base order equal to the +/// row order. fn hand_stage() -> (Lod, LodConfig) { let coordinates = [ Vec2::new(0.0, 0.0), @@ -510,9 +523,7 @@ fn build_produces_the_hand_computed_columns() { Bounds2::new(Vec2::new(0.0, 0.0), Vec2::new(1.0, 1.0)).expect("a real frame"), ); - // Wire coordinates in base order, exact: the normalization is a - // single f64 rounding per component and every fixture value is a - // dyadic rational. + // every fixture value and its image 2x − 1 are exactly representable dyadic rationals assert_eq!( *lod.coordinates.as_raw(), [ @@ -704,9 +715,6 @@ fn columns_round_trip_through_the_morton_file() { ); } -/// Every finite point set builds. -/// -/// The result upholds the serving contract's structural laws. #[property_test] fn built_columns_uphold_the_contract_laws( #[strategy = proptest::collection::vec( @@ -783,14 +791,10 @@ fn built_columns_uphold_the_contract_laws( Ok(()), ); - // At most one delivered point per depth-d cell at every cut - // d below the deepest grid, jointly across buckets - the - // uniqueness claim the mass channel leans on. The - // cascade's represented rule guarantees it. A claim never - // lands in a cell holding an earlier-assigned point, so two - // points sharing a depth-d cell cannot both carry buckets at - // or below d unless the later one is a catch-all leftover - // (bucket = deepest, never delivered below the deepest cut). + // A depth-d cell lies inside every shallower cell containing its points. The cascade never + // assigns a point in a cell already represented by an earlier-assigned point. Two depth-d + // co-residents cannot both claim at or below d. Therefore every cut below the catch-all has at + // most one delivered representative per cell, jointly across buckets. for cut in 0..deepest.get() { let cut = depth(cut); let mut occupied = std::collections::HashSet::new(); @@ -829,7 +833,7 @@ fn built_columns_uphold_the_contract_laws( prop_assert!(evidence.max_tile_delta >= 1); } -/// Direct types for the hand-stage rows: distinct enough that every union is distinguishable. +/// Assigns hand-stage direct types that distinguish the expected cell unions. fn hand_types() -> IdVec> { vec![ smallvec![OntologyRowId::new(5)], @@ -843,12 +847,10 @@ fn hand_types() -> IdVec> { #[test] fn quad_build_produces_the_hand_computed_tree() { - // The hand-stage cut at span = 1: the root's cut is bucket 1, - // so its run carries buckets 0..=1 (positions 0..2) and only - // quadrant (0, 0) - holding the two bucket-2 points - gets a - // child. That child's cut is the deepest grid: a leaf whose run - // is the catch-all pair (positions 2..4) and whose subtree also - // contains the origin point delivered by the root. + // the root's span-1 cut carries buckets 0..=1 (positions 0..2). Only quadrant (0, 0), holding + // the two bucket-2 points, gets a child. That child's deepest-grid cut delivers the catch-all + // pair (positions 2..4). Its whole-cell population also includes the origin point delivered by + // the root. let (lod, config) = hand_stage(); let tree = QuadTree::build(&lod, &hand_types(), config).expect("the fixture builds"); @@ -875,9 +877,8 @@ fn quad_build_produces_the_hand_computed_tree() { #[test] fn quad_build_gathers_types_through_the_base_order() { - // The hand-stage points with their rows permuted. Importance travels with each point, so the - // cascade and the tree are identical. The base order is no longer the row order - // (row_of_position = [1, 0, 3, 2]), and the builder must gather the type column through it. + // the permuted rows retain each point's distinct importance. The cascade and tree remain + // identical, but row_of_position = [1, 0, 3, 2] requires gathering the type column. let coordinates = [ Vec2::new(1.0, 1.0), Vec2::new(0.0, 0.0), @@ -1013,9 +1014,6 @@ fn quad_tree_round_trips_through_the_quad_file() { assert_eq!(stored, [2, 5, 9]); } -/// Every built lod cuts into a tree upholding the serving contract's structural laws. -/// -/// Certified against linear-scan references. #[property_test] fn quad_trees_uphold_the_contract_laws( #[strategy = proptest::collection::vec( @@ -1095,8 +1093,7 @@ fn quad_trees_uphold_the_contract_laws( }) .collect(); - // The runs partition the base order, so the incremental tile pyramid delivers every point - // exactly once. + // a partition requires exactly one delivery of each base position let mut delivered = vec![0_u32; lod.codes.len()]; for node in &tree.nodes { for position in node.run() { @@ -1173,7 +1170,7 @@ fn quad_trees_uphold_the_contract_laws( prop_assert!(evidence.depth.get() <= max_tile_depth); } -/// A deterministic xorshift64* stream for adversarial cases without new dependencies. +/// Advances a xorshift64 state and returns its multiplied output. fn harness_rng(state: &mut u64) -> u64 { *state ^= *state << 13; *state ^= *state >> 7; @@ -1181,7 +1178,11 @@ fn harness_rng(state: &mut u64) -> u64 { state.wrapping_mul(0x2545_F491_4F6C_DD1D) } -/// Draws a value below `bound` from the stream. +/// Reduces a stream word modulo `bound`, with possible modulo bias. +/// +/// # Panics +/// +/// Panics when `bound` is zero. #[expect( clippy::integer_division_remainder_used, clippy::cast_possible_truncation, @@ -1191,7 +1192,7 @@ fn harness_draw(state: &mut u64, bound: usize) -> usize { (harness_rng(state) as usize) % bound } -/// The deepest depth at which both keys' prefixes agree, computed through `prefix` alone. +/// Finds the deepest grid where both keys' prefixes agree. fn oracle_shared_depth(left: MortonKey, right: MortonKey) -> u8 { (0..=32_u8) .rev() @@ -1202,8 +1203,10 @@ fn oracle_shared_depth(left: MortonKey, right: MortonKey) -> u8 { .expect("depth zero prefixes are always equal") } -/// Computes each point's bucket quadratically, as one past the deepest grid the point -/// shares with ANY better-ranked point, clamped to `deepest`. The best point takes zero. +/// Computes first-separation buckets by comparing every pair of points. +/// +/// For each point, the result is min(D + 1, `deepest`), where D is its deepest shared grid with any +/// better-ranked point. The best-ranked point takes zero. fn oracle_natural_buckets(points: &[(MortonKey, ImportanceRank)], deepest: Depth) -> Vec { points .iter() @@ -1223,24 +1226,25 @@ fn oracle_natural_buckets(points: &[(MortonKey, ImportanceRank)], deepest: Depth .collect() } -/// Builds adversarial key pools that force heavy duplication, deep shared prefixes, -/// splits at even and at odd bit positions, and full-width randoms. +/// Builds duplicate-heavy key pools with prescribed shared-prefix boundaries. +/// +/// Pools include equal keys, low-bit and high-bit splits on both axes, and full-width stream words. fn harness_key_pools(state: &mut u64) -> Vec> { let base = harness_rng(state); vec![ // Every point drawn from here is co-resident at the full width. vec![MortonKey::from_bits(base)], - // The pair differs in the lowest bit, so the shared depth is 31. + // differing only in bit 0 leaves 31 complete leading bit pairs in common vec![ MortonKey::from_bits(base | 1), MortonKey::from_bits(base & !1), ], - // The pair differs at bit 62, x's top bit, so the shared depth is 0. + // differing in bit 62, x's top bit, leaves no complete leading bit pair in common vec![ MortonKey::from_bits(base | (1 << 62)), MortonKey::from_bits(base & !(1 << 62)), ], - // The pair differs at bit 63, y's top bit, so the shared depth is 0. + // differing in bit 63, y's top bit, leaves no complete leading bit pair in common vec![ MortonKey::from_bits(base | (1 << 63)), MortonKey::from_bits(base & !(1 << 63)), @@ -1256,8 +1260,6 @@ fn harness_key_pools(state: &mut u64) -> Vec> { ] } -/// `separation_buckets` equals both the quadratic closed form and `cascade::buckets` at the -/// key width, over adversarial duplicate-heavy inputs. #[test] fn separation_buckets_matches_oracle_and_cascade_adversarially() { let mut state = 0x0BAD_5EED_0BAD_5EED_u64; @@ -1336,8 +1338,6 @@ fn separation_buckets_matches_oracle_and_cascade_adversarially() { assert!(cases >= 2_000, "the sweep exercised the pools"); } -/// `cascade::buckets` itself realizes the min(D + 1, deepest) closed form at EVERY deepest, -/// not only the key width: the analytic reading the replacement rests on. #[test] fn cascade_buckets_matches_the_closed_form_at_every_deepest() { let mut state = 0xFEED_FACE_FEED_FACE_u64; diff --git a/libs/@local/graph/atlas/src/salt/mod.rs b/libs/@local/graph/atlas/src/salt/mod.rs index 95ef799ac5f..0dbe56b6806 100644 --- a/libs/@local/graph/atlas/src/salt/mod.rs +++ b/libs/@local/graph/atlas/src/salt/mod.rs @@ -1,10 +1,9 @@ -//! The SALT pipeline, which fits and serves atlas generations. +//! Offline fitting and artifact construction for atlas generations. //! -//! SALT turns one frozen [`Dataset`](crate::dataset::Dataset) into a published atlas generation: -//! -//! - a 2D map over the graph's entities -//! - artifacts on disk under `crate::file`'s formats -//! - the spatial indexes serving reads from them +//! [`fit`] builds a 2D map and its spatial indexes from a [`Dataset`](crate::dataset::Dataset), +//! publishing artifacts in [`crate::file`]'s formats for `crate::serve`. The input must provide +//! the consistent view required by the dataset contract. [`runner`] connects datasets and embedding +//! providers to fitting and report generation. pub(crate) mod adjacency; pub(crate) mod embedding; diff --git a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/groups.rs b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/groups.rs index 274613a5521..0e4e8cddf26 100644 --- a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/groups.rs +++ b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/groups.rs @@ -1,9 +1,10 @@ //! Validation-group derivation: leakage axes, near-duplication, and budgeted subdivision. //! -//! Rows that could leak shared content across a train/validation split must share a group; the -//! union runs over value-keyed leakage axes and near-duplicate pairs, and over-budget components -//! subdivide by relaxing their weakest axes in information order. The target and weight arithmetic -//! lives here beside the grouping because both read the same vote counts. +//! Rows that could leak shared content across a train/validation split share a group where the +//! group budget allows. The union runs over value-keyed leakage axes and near-duplicate pairs, and +//! over-budget components subdivide by relaxing their weakest axes in information order, down to +//! the identity edges that never relax. The target and weight arithmetic lives here beside the +//! grouping because both read the same vote counts. use std::collections::{HashMap, HashSet, hash_map::Entry}; @@ -66,7 +67,9 @@ struct RowAxes { /// Beside the identity axis and the near-duplicate pairs it always honours. #[derive(Copy, Clone)] struct Ranks { + /// Whether relation-family edges unite. family: bool, + /// Whether base-URL edges unite. base: bool, /// The inclusive cosine-distance cut for near-duplicate pairs. cut: f64, @@ -191,8 +194,8 @@ fn farthest_first_cut( .all(|part| (part.len() as f64) <= budget) }; - // Component size is monotone in the cut, so the fitting prefix of - // the candidate list is contiguous and binary-searchable. + // Component size is monotone in the cut. The fitting prefix of the + // candidate list is therefore contiguous and binary-searchable. let (mut fitting, mut exceeded) = (None, distances.len()); let mut low = 0; while low < exceeded { @@ -209,7 +212,7 @@ fn farthest_first_cut( evidence.empty_cut_components = Some(evidence.empty_cut_components.unwrap_or(0) + 1); } - // The empty cut keeps identity edges alone; the caller records + // The empty cut keeps identity edges alone. The caller records // any part still over budget. partition( component, @@ -315,11 +318,12 @@ struct NearDuplicateBoundary { /// the leading gap from zero. /// /// The boundary is the winning void's geometric midpoint, maximally far from the evidence on both -/// sides. Ties keep the lowest void so a near-duplicate stays near. +/// sides. Ties keep the lowest void, and a near-duplicate therefore stays near. /// -/// Exact coincidences (distances ≤ 0 after rounding) join under any non-negative boundary and carry -/// no void evidence. An empty region has no low tail below the ceiling, which means no duplicate -/// structure and no near-duplicate edges, so the boundary derives as zero. +/// Exact coincidences (distances ≤ 0 after rounding) join under any non-negative boundary, the zero +/// boundary included, and carry no void evidence. An empty region has no positive distance below +/// the ceiling and no void to derive from, and the boundary then derives as zero. The +/// near-duplicate edges are then exactly the coincident pairs. fn near_duplicate_boundary(mut distances: Vec) -> NearDuplicateBoundary { if distances.is_empty() { return NearDuplicateBoundary { @@ -373,8 +377,16 @@ fn near_duplicate_boundary(mut distances: Vec) -> NearDuplicateBoundary { /// Assigns every trained row its validation-group digest. /// -/// The returned digests align with `trained`; the evidence gains the group count, the derived +/// The returned digests align with `trained`. The evidence gains the group count, the derived /// near-duplicate boundary with its grounds, and the near-duplicate pair count. +/// +/// # Panics +/// +/// This panics when the embedding table holds fewer rows than `trained`, when the row count or +/// the axis-value count exceeds `u32`, or, under overflow checks, when the product +/// `rows · (rows − 1)` behind the pair-count reservation overflows `usize`, which a 32-bit `usize` +/// reaches at 65,537 rows. The assembly built the table over exactly these cards, and a table +/// mismatch is a program defect rather than a data condition. #[expect( clippy::cast_precision_loss, reason = "row and group counts are far below f64's 2^53 exact-integer range" @@ -387,7 +399,7 @@ pub(super) fn validation_groups( ) -> Vec { let rows = trained.len(); - // Value-keyed axes join cards sharing an axis value; an inverse + // Value-keyed axes join cards sharing an axis value. An inverse // pair meets at the named identity's key whether or not that // identity is itself on the corpus. let mut keys: HashMap = HashMap::new(); @@ -416,7 +428,9 @@ pub(super) fn validation_groups( }); } - // T(rows - 1) unordered pairs; the product is even, so the midpoint halves it exactly. + // rows · (rows - 1) / 2 unordered pairs. The product is even, and the midpoint therefore + // halves it exactly. A 32-bit `usize` overflows the product from 65,537 rows, and a wrapped + // product would reserve the wrong capacity. let mut distances = Vec::with_capacity(usize::midpoint(0, rows * rows.saturating_sub(1))); for left in 0..rows { let embedding = view @@ -482,8 +496,9 @@ pub(super) fn validation_groups( } evidence.fold_groups = groups.len(); - // Trained rows keep the corpus's ascending identity order, so a - // group hashes its members in one fixed order under any traversal. + // Trained rows keep the corpus's ascending identity order, and a + // group therefore hashes its members in one fixed order under any + // traversal. // The hashed bytes are each identity's canonical re-serialization, // which the wire order does not pin byte-for-byte. let mut assigned: Vec> = vec![None; rows]; diff --git a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/mod.rs b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/mod.rs index 0a273eb98e5..8f86eb2105b 100644 --- a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/mod.rs @@ -4,12 +4,12 @@ //! renders each admitted card through the canonical card template and embeds the rendered text. //! Each row then carries a smoothed soft target and a vote weight, and it joins the indivisible //! validation group its leakage axes and near-duplicate neighbours imply. The output is -//! deterministic in the corpus bytes, the embedding provider, and the [`AssemblyConfig`], so one +//! deterministic in the corpus bytes, the embedding provider, and the [`AssemblyConfig`], and one //! corpus always yields the same training set. //! //! Rendering goes through the same template, budgets, and lint as generation-time production cards, -//! so classifier-time and generation-time card text cannot drift from each other. The adapters live -//! in [`render`], and the grouping and subdivision machinery lives in [`groups`]. +//! and classifier-time and generation-time card text cannot drift from each other. The adapters +//! live in [`render`], and the grouping and subdivision machinery lives in [`groups`]. //! //! # Card policy //! @@ -21,7 +21,7 @@ //! - holdout cards stay out of training and keep their human verdict as the answer. The assembly //! still renders and embeds them, and [`AssembledCorpus::holdouts`] carries them beside their //! verdicts as the fitted model's evaluation set; -//! - cards whose votes assert no geometry class (all unclear or abstain) drop with zero weight, so +//! - cards whose votes assert no geometry class (all unclear or abstain) drop with zero weight, and //! uncertainty discounts a card without distorting its target; //! - `prescreen_stratum` is a stratification fact and has no effect here. //! @@ -38,7 +38,7 @@ //! //! the Jeffreys prior. A card's target stays a proper distribution at any vote count, and few-vote //! cards shrink toward uniform instead of asserting certainty their evidence does not carry. The -//! row weight is `m`, so the fit's cross-entropy counts each geometry vote once. +//! row weight is `m`, and the fit's cross-entropy counts each geometry vote once. //! //! # Validation groups //! @@ -52,21 +52,21 @@ //! The boundary is the geometric midpoint of the widest multiplicative void among the sorted //! pairwise distances within `(0, median · NEAR_DUPLICATE_CEILING_FRACTION]`, with the trailing //! void ending at the ceiling included and the leading gap from zero excluded. A duplicate cluster -//! sits decades below the corpus bulk, and the void between them is the boundary's evidence. A +//! lies decades below the corpus bulk, and the void between them is the boundary's evidence. A //! corpus without such structure takes the trailing void and joins the bulk's whole low tail: every //! pair below the boundary, whose count the corpus decides and the ceiling alone bounds. -//! Subdivision cuts far near-duplicate edges first, so the failure direction stays toward more +//! Subdivision cuts far near-duplicate edges first, and the failure direction stays toward more //! conservative validation. The evidence records the derived boundary with its void and its //! ceiling. The group label is the SHA-256 over the component's member identity URLs in trained-row -//! order, which is the corpus's ascending identity order, so it is stable under any traversal +//! order, which is the corpus's ascending identity order, and it is stable under any traversal //! order. //! //! A component larger than [`AssemblyConfig::maximum_group_fraction`] of the trained rows is too -//! large for grouped validation, so subdivision relaxes its axes in information order, and the axis -//! whose one edge says least about leakage goes first. Family edges drop inside the component, then -//! base-URL edges, then near-duplicate edges cut farthest-first (the kept cut is the largest -//! distance under which every part fits the budget). Identity and inverse edges never relax, so a -//! part they alone hold over budget stays one group, and the evidence records it. Relaxation is +//! large for grouped validation. Subdivision then relaxes its axes in information order, and the +//! axis whose one edge says least about leakage goes first. Family edges drop inside the component, +//! then base-URL edges, then near-duplicate edges cut farthest-first (the kept cut is the largest +//! distance under which every part fits the budget). Identity and inverse edges never relax: a part +//! they alone hold over budget stays one group, and the evidence records it. Relaxation is //! per-component, and groups already within budget keep every axis. //! //! Publisher is not a union axis. On the live corpus it collapses the 1,684 cards into 5 components @@ -105,19 +105,18 @@ const DIRICHLET_ALPHA: f64 = 0.5; /// The boundary derivation looks for a duplicate/bulk void strictly below the corpus's typical /// inter-card distance. The median is that typical distance by construction, and the quarter stands /// the region off from the bulk's lower tail. The derivation picks the widest void within the -/// region, so the fraction only bounds the search: any value materially below one and above the +/// region, and the fraction only bounds the search: any value materially below one and above the /// duplicate scale finds the same void on a corpus with duplicate structure. const NEAR_DUPLICATE_CEILING_FRACTION: f64 = 0.25; /// The prose language every corpus card must declare. /// /// This is a constant rather than a configuration field. The template renders exactly one language, -/// and its sentence segmentation, token budgets, and connective phrases all follow that language, -/// so a knob offering languages the template cannot render would be a lever wired to nothing. +/// and its sentence segmentation, token budgets, and connective phrases all follow that language. +/// A knob offering languages the template cannot render would be a lever wired to nothing. /// /// The per-card `language` field exists on the wire so a corpus declares what it holds and a -/// mismatch fails here instead of rendering garbage. When a second template arrives, the tag -/// becomes that template's property. +/// mismatch fails here instead of rendering garbage. const CARD_LANGUAGE: &str = "en"; /// Assembly settings. @@ -188,8 +187,8 @@ impl core::error::Error for AssemblyError { /// What one assembly admitted, dropped, and derived. /// -/// Destined for the generation metadata beside the fit evidence, so a published classifier names -/// the corpus policy outcomes its training ran under. +/// Destined for the generation metadata beside the fit evidence, and a published classifier +/// therefore names the corpus policy outcomes its training ran under. #[derive(Debug, Copy, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct AssemblyEvidence { /// Cards the corpus supplied. @@ -204,8 +203,9 @@ pub(crate) struct AssemblyEvidence { pub trained: usize, /// Distinct card texts embedded. pub unique_texts: usize, - /// Rows whose rendering exceeded the `hard_token_budget` or dropped more than half their - /// examples. + /// Rows whose rendering exceeded the budget or dropped over half their examples. + /// + /// The budget is `hard_token_budget`. pub severely_truncated: usize, /// Indivisible validation groups over the trained rows, after subdivision. pub fold_groups: usize, @@ -217,19 +217,22 @@ pub(crate) struct AssemblyEvidence { /// /// Zeros when the derivation found no void. pub near_duplicate_void: [f64; 2], - /// The derivation's search ceiling, which is the median pairwise distance scaled by the - /// fraction. + /// The derivation's search ceiling. + /// + /// It is the median pairwise distance scaled by the fraction. pub near_duplicate_ceiling: f64, /// Over-budget components subdivision split. pub subdivided_groups: usize, /// Groups accepted over budget because identity edges alone hold them together. pub oversized_accepted: usize, - /// Components whose subdivision found no fitting near-duplicate cut and severed every - /// near-duplicate edge under the empty cut. + /// Components whose subdivision found no fitting near-duplicate cut. + /// + /// Their subdivision severed every near-duplicate edge under the empty cut. /// /// Each counted component scattered rows the boundary judged interchangeable across smaller - /// groups, so a train/validation split may separate near-duplicates. `None` when the document - /// predates the counter: the event was possible and uncounted, which is not a zero. + /// groups, and a train/validation split may therefore separate near-duplicates. `None` when + /// the document predates the counter: the event was possible and uncounted, which is not a + /// zero. pub empty_cut_components: Option, /// The weakest axis rank subdivision engaged. pub deepest_relaxation: Relaxation, @@ -277,8 +280,8 @@ pub(crate) struct HoldoutCard { /// One assembled training set and its provenance. /// /// Row `i` of the embedding table, the training rows, and the identities all describe the same -/// card; holdout cards occupy the table rows after the trained rows, addressed through -/// [`holdouts`](Self::holdouts). The table is the artifact shape the fit stages and maps; the +/// card. Holdout cards occupy the table rows after the trained rows, addressed through +/// [`holdouts`](Self::holdouts). The table is the artifact shape the fit stages and maps, and the /// mapped embedding matrix's leading trained rows and [`rows`](Self::rows) together satisfy the /// classifier's training-set contract. #[derive(Debug)] @@ -398,10 +401,10 @@ where .filter(|finished| finished.severely_truncated()) .count(); - // Holdout rows render and embed after every trained row, so the - // trained rows keep their positions and the group derivation's - // trained-row scan bound holds. Each push hands back the holdout - // card's assigned row in the table. + // Holdout rows render and embed after every trained row. The + // trained rows therefore keep their positions, and the group + // derivation's trained-row scan bound holds. Each push hands back + // the holdout card's assigned row in the table. let mut holdout_rows = Vec::with_capacity(held_out.len()); for &(index, corpus_card, _) in &held_out { holdout_rows.push(rendered.push(render_card(index, corpus_card)?)); diff --git a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/render.rs b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/render.rs index 0563bce37dc..06d401a5ada 100644 --- a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/render.rs +++ b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/render.rs @@ -1,8 +1,8 @@ //! Corpus-card rendering through the canonical template. //! -//! The classifier trains on exactly the card text production serves, so every corpus card renders -//! through the same template, budgets, and lint as generation-time cards; helpers here adapt the -//! corpus wire shapes to the template's inputs. +//! The classifier trains on exactly the card text production serves. Every corpus card therefore +//! renders through the same template, budgets, and lint as generation-time cards, and the helpers +//! here adapt the corpus wire shapes to the template's inputs. use super::{ super::{Card, CardIdentity, Content, Phrase}, @@ -43,6 +43,11 @@ fn examples<'text>( } /// Builds one card's rendered endpoint constraints. +/// +/// # Errors +/// +/// Returns [`AssemblyError::Cardinality`] when a constraint's minimum target count exceeds its +/// maximum. fn endpoint_constraints<'text, E>( index: usize, content: &'text Content, @@ -77,6 +82,12 @@ fn endpoint_constraints<'text, E>( } /// Renders one corpus card through the canonical template. +/// +/// # Errors +/// +/// Returns [`AssemblyError::Language`] when the card declares a language other than the +/// template's, [`AssemblyError::Cardinality`] when an endpoint constraint's minimum exceeds its +/// maximum, and [`AssemblyError::Render`] when the template refuses the card. pub(super) fn render_card( index: usize, corpus_card: &Card, @@ -139,7 +150,7 @@ pub(super) fn render_card( }; // The wikidata entity token is the card's resolved source - // identifier; hash identities are URLs, which the structural lint + // identifier. Hash identities are URLs, which the structural lint // already forbids. let forbidden = match &corpus_card.identity { CardIdentity::Wikidata { url, .. } => url.rsplit('/').next(), diff --git a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/tests.rs b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/tests.rs index 326ddfb367f..a826169a206 100644 --- a/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/annotation/assembly/tests.rs @@ -22,16 +22,26 @@ use crate::{ }, }; +/// The record hash every fixture card and vote carries. const DIGEST: &str = "6cf1a86693da441a9c86ed4dcf2bcdad6cf1a86693da441a9c86ed4dcf2bcdad"; +/// Identity of the rich `part of` card, the inverse of [`HAS_PART`]. const PART_OF: &str = "http://www.wikidata.org/entity/P361"; +/// Identity of the `has part` card, the inverse of [`PART_OF`]. const HAS_PART: &str = "http://www.wikidata.org/entity/P527"; +/// Identity of one of the near-tie pair whose embeddings nearly coincide. const ALPHA: &str = "http://www.wikidata.org/entity/P600"; +/// Identity of the other near-tie card. const BETA: &str = "http://www.wikidata.org/entity/P601"; +/// Identity of a card standing alone in its own group. const GAMMA: &str = "http://www.wikidata.org/entity/P602"; +/// Identity of a card whose votes are all unclear. It carries zero weight and is dropped. const ALL_UNCLEAR: &str = "http://www.wikidata.org/entity/P700"; +/// Identity of the shot-excluded card. const SHOT: &str = "http://www.wikidata.org/entity/P800"; +/// Identity of the held-out card with a human proximal verdict. const HOLDOUT: &str = "http://www.wikidata.org/entity/P900"; +/// Identity of the hash-sourced `Employed By` card with a single simple endpoint pair. const EMPLOYED_BY: &str = "https://hash.ai/@h/types/entity-type/employed-by/v/1"; /// A uniquely named file in the system temporary directory, removed on drop. @@ -40,6 +50,7 @@ struct TempFile { } impl TempFile { + /// Writes `bytes` to a fresh uniquely numbered file in the system temp dir. fn create(bytes: &[u8]) -> Self { static COUNTER: AtomicU64 = AtomicU64::new(0); @@ -69,9 +80,10 @@ impl ProgrammedEmbedder { /// Alpha and beta lie `0.005` radians apart (`1 - cos(0.005) ≈ 1.25e-5`), a true near-duplicate /// pair decades below every other distance. Beta and delta lie `0.006` radians apart (`≈ /// 1.8e-5`), a farther member of the same duplicate cluster. The next distance up in any corpus - /// here is `≥ 1.0e-2`, so the boundary derivation finds the void between the cluster and the - /// bulk. Twins share one angle exactly, so every twin pair sits at distance zero: the - /// coincident clique the empty-cut witness scatters. + /// here is `≥ 1.0e-2` (gamma against beta, `1 - cos(0.145)`), and the boundary derivation + /// therefore finds the void between the cluster and the bulk. Twins share one angle exactly, + /// and every twin pair lies at distance zero: the coincident clique the empty-cut witness + /// scatters. fn angle(title: &str) -> f32 { match title { "part of" => 0.0, @@ -318,6 +330,7 @@ fn group_digest(members: &[&str]) -> crate::integrity::Sha256Digest { hasher.finalize() } +/// Rendering the rich fixture card reproduces the Python exporter's card text section for section. #[test] fn template_renders_the_python_card_text() { let corpus = fixture_corpus(); @@ -499,6 +512,10 @@ async fn assembly_smooths_groups_and_counts_the_fixture_corpus() { ); } +/// Forms a valid [`TrainingSet`] from a staged, remapped embedding matrix. +/// +/// The staged embedding matrix written to a file and remapped, sliced to the trained rows, forms a +/// valid `TrainingSet` with the assembled rows. #[tokio::test] async fn staged_table_and_rows_satisfy_the_training_contract() { let corpus = fixture_corpus(); @@ -525,7 +542,7 @@ async fn staged_table_and_rows_satisfy_the_training_contract() { // The trained rows lead the table; the holdout rows after them are // evaluation material, not training supply. The mapped file hands back a - // raw slice, so entering the card-row domain is this seam's own doing. + // raw slice, and entering the card-row domain is this boundary's own doing. TrainingSet::new( IdSlice::from_raw(&embeddings[..assembled.rows().len()]), assembled.rows(), @@ -551,6 +568,10 @@ async fn corpus_with_no_admissible_card_is_an_empty_assembly() { assert_matches!(error, AssemblyError::Empty); } +/// A German card fails to assemble with `AssemblyError::Language` naming card and language. +/// +/// A card recorded in German fails to assemble with `AssemblyError::Language` naming the card and +/// language. #[tokio::test] async fn language_the_template_does_not_render_is_rejected() { let mut card = wikidata_card(ALPHA, "alpha", "f-600", &[vote("overlay")]); @@ -674,8 +695,8 @@ async fn subdivision_relaxes_base_when_family_is_not_the_glue() { #[tokio::test] async fn subdivision_cuts_near_duplicates_farthest_first() { - // A near-duplicate triangle - alpha-beta at ~1.25e-5, beta-delta - // at ~1.8e-5, alpha-delta at ~6.05e-5 - against three far cards + // A near-duplicate triangle (alpha-beta at ~1.25e-5, beta-delta + // at ~1.8e-5, alpha-delta at ~6.05e-5) against three far cards // supplying the corpus bulk. The derived boundary joins the whole // triangle; the budget rejects it, and the cut drops the farther // links and keeps the nearest pair. @@ -697,7 +718,7 @@ async fn subdivision_cuts_near_duplicates_farthest_first() { assert_eq!(evidence.fold_groups, 5); assert_eq!(evidence.subdivided_groups, 1); assert_eq!(evidence.oversized_accepted, 0); - // A fitting cut exists, so the empty-cut counter stays untouched. + // A fitting cut exists, and the empty-cut counter stays untouched. assert_eq!(evidence.empty_cut_components, Some(0)); assert_eq!( evidence.deepest_relaxation, @@ -767,6 +788,10 @@ async fn identity_web_is_accepted_over_budget() { } } +/// Scatters five byte-identical cards into five groups under the empty cut. +/// +/// Five byte-identical cards form a clique no cut can split, and the assembly scatters them into +/// five groups and counts one empty-cut component. #[tokio::test] async fn empty_cut_scatters_the_coincident_clique_and_counts_itself() { let cards: Vec = [ @@ -801,6 +826,7 @@ async fn empty_cut_scatters_the_coincident_clique_and_counts_itself() { assert_eq!(twin_groups.len(), 5); } +/// Assembling the same oversized component twice yields identical groups and evidence. #[tokio::test] async fn subdivision_is_deterministic() { let cards: Vec = [ diff --git a/libs/@local/graph/atlas/src/salt/policy/annotation/mod.rs b/libs/@local/graph/atlas/src/salt/policy/annotation/mod.rs index b246580665c..42893d4d3ac 100644 --- a/libs/@local/graph/atlas/src/salt/policy/annotation/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/annotation/mod.rs @@ -11,8 +11,8 @@ //! The document is the wire boundary between the annotation tooling and the fit. Rendered card //! text, embeddings, class counts, and smoothed targets all derive in Rust from these fields. //! -//! [`AnnotationCorpus::from_slice`] runs the whole wire contract at construction, so consumers read -//! cards without re-checking. The contract covers: +//! [`AnnotationCorpus::from_slice`] runs the whole wire contract at construction, and consumers +//! read cards without re-checking. The contract covers: //! //! - the declared schema //! - cards strictly ascending by identity in byte order @@ -25,14 +25,15 @@ //! //! A card's identity is its canonical URL. A type from the hash store carries its full versioned //! URL, because versions are immutable and distinct and each version's card is its own annotation -//! subject. A wikidata record carries no version at source, so a retrieval timestamp and the digest -//! of the retrieved source record pin its entity-URL identity ([`CardIdentity`]). +//! subject. A wikidata record carries no version at source, and a retrieval timestamp and the +//! digest of the retrieved source record therefore pin its entity-URL identity +//! ([`CardIdentity`]). //! //! # Vote semantics //! //! Votes are verbatim five-way records, covering the three [`GeometryClass`]es plus `unclear` (the //! judge found the card ambiguous) and `abstain` (the judge withheld an answer). -//! [`Card::vote_counts`] derives the class counts by counting the vote list, so a document cannot +//! [`Card::vote_counts`] derives the class counts by counting the vote list, and a document cannot //! carry counts that disagree with their own provenance. Unclear and abstain votes assert no //! geometry class, and neither the per-class counts nor the target weight include them, which //! discounts an uncertain card without distorting its target distribution. @@ -87,7 +88,7 @@ pub enum InvalidAnnotationCorpus { Schema { found: Box }, /// A card is not strictly after its predecessor in byte order of identity. /// - /// Which also covers duplicated identities. + /// This also covers duplicated identities. UnorderedCards { index: usize }, /// A card's identity does not parse under its source's identity form. IdentityForm { index: usize, source: Source }, @@ -289,7 +290,7 @@ impl From for card::Direction { /// /// Every boolean field is tri-state: `true` asserts the property, `false` asserts its absence, and /// `null` records that the source does not record the fact. The exporter always writes every key, -/// so an absent key is a wire violation rather than a third spelling of `null`. +/// and an absent key is therefore a wire violation rather than a third spelling of `null`. #[derive(Debug, Clone, PartialEq, serde::Deserialize)] #[serde(deny_unknown_fields)] pub(crate) struct Constraints { @@ -337,7 +338,7 @@ pub(crate) struct Content { pub title: String, /// The relation's description prose, when any survives. /// - /// Identifier sanitization at the source can drop every sentence of a description; a card + /// Identifier sanitization at the source can drop every sentence of a description, and a card /// without one records `null`. #[serde(deserialize_with = "nullable")] pub description: Option, @@ -362,11 +363,15 @@ pub(crate) struct Content { pub slug: String, } -/// The leakage axes that group a card. +/// A card's leakage axes, the recorded values behind its validation group. /// -/// Evaluation folds union cards sharing any axis value, so related cards never straddle a -/// train/validation split. Axis values are grouping strings; identity semantics live in -/// [`CardIdentity`] alone. +/// Training-set assembly unites cards into validation groups through their identity and inverse +/// identities, their relation family and base URL, and near-duplicate embeddings. Where a group +/// exceeds its budget, subdivision relaxes the family union, then the base-URL union, then cuts +/// near-duplicate edges farthest-first, down to an empty cut that separates even zero-distance +/// pairs. Identity and inverse-identity edges never relax. The publisher is a recorded fact for +/// stratified evaluation and not a union axis. Axis values are grouping strings, and identity +/// semantics live in [`CardIdentity`] alone. #[derive(Debug, Clone, PartialEq, serde::Deserialize)] #[serde(deny_unknown_fields)] pub(crate) struct Axes { @@ -418,7 +423,7 @@ impl HoldoutClass { #[derive(Debug, Clone, PartialEq, serde::Deserialize)] #[serde(deny_unknown_fields)] pub(crate) struct Flags { - /// Annotation prompts disclosed the card's verdict, so it carries no votes. + /// Annotation prompts disclosed the card's verdict, and it therefore carries no votes. pub shot_excluded: bool, /// The human verdict class held out for evaluation, when one exists. #[serde(deserialize_with = "nullable")] @@ -598,9 +603,8 @@ struct Document { /// A validated annotation-corpus document. /// /// Construction checks the whole wire contract, and the module documentation lists the clauses. The -/// manifest pins the document by the SHA-256 of exactly the supplied bytes, whatever the format; -/// JSON-vs-columnar for the corpus and verdict documents is an open format decision on the supply -/// boundary, not a property of this type. +/// manifest pins the document by the SHA-256 of exactly the supplied bytes, and the pin does not +/// depend on the document's format. #[derive(Debug, Clone, PartialEq)] pub(crate) struct AnnotationCorpus { cards: Vec, @@ -652,14 +656,18 @@ impl AnnotationCorpus { /// One card's admission under the wire contract. /// -/// The admission owns the card's row index, so each contract clause reports its position without -/// carrying it through every check. +/// The admission owns the card's row index, and each contract clause therefore reports its +/// position without carrying it through every check. struct CorpusAdmission { index: usize, } impl CorpusAdmission { /// Checks one wire card's contract clauses and types its identity. + /// + /// # Errors + /// + /// Returns the first violated clause as an [`InvalidAnnotationCorpus`]. fn admit(&self, card: WireCard) -> Result { let index = self.index; let identity = self.identity(&card)?; @@ -703,6 +711,10 @@ impl CorpusAdmission { } /// Types a wire card's identity under its source's form and pin rules. + /// + /// # Errors + /// + /// Returns the first violated identity or pin clause as an [`InvalidAnnotationCorpus`]. fn identity(&self, card: &WireCard) -> Result { let index = self.index; match card.source { @@ -763,6 +775,11 @@ impl CorpusAdmission { } /// Checks one card's content strings and endpoint bounds. + /// + /// # Errors + /// + /// Returns the first empty field, identifier-carrying field, or inverted endpoint bound as an + /// [`InvalidAnnotationCorpus`]. fn content(&self, content: &Content) -> Result<(), InvalidAnnotationCorpus> { let index = self.index; let mut prose: Vec<(&'static str, &str)> = vec![ @@ -827,6 +844,10 @@ impl CorpusAdmission { } /// Checks one card's axis strings. + /// + /// # Errors + /// + /// Returns the first empty axis field as an [`InvalidAnnotationCorpus`]. fn axes(&self, axes: &Axes) -> Result<(), InvalidAnnotationCorpus> { let index = self.index; let mut fields: Vec<(&'static str, &str)> = vec![ @@ -844,7 +865,14 @@ impl CorpusAdmission { Ok(()) } - /// Checks every vote's provenance strings and temperature. + /// Checks every vote's provenance strings. + /// + /// The temperature needs no check here: its [`DFinite`] type refuses a non-finite reading at + /// deserialization. + /// + /// # Errors + /// + /// Returns the first empty vote field as an [`InvalidAnnotationCorpus`]. fn votes(&self, votes: &[Vote]) -> Result<(), InvalidAnnotationCorpus> { let index = self.index; for (vote_index, vote) in votes.iter().enumerate() { diff --git a/libs/@local/graph/atlas/src/salt/policy/annotation/tests.rs b/libs/@local/graph/atlas/src/salt/policy/annotation/tests.rs index 82dd08d414f..66b63f3a196 100644 --- a/libs/@local/graph/atlas/src/salt/policy/annotation/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/annotation/tests.rs @@ -8,9 +8,13 @@ use super::{ }; use crate::salt::policy::GeometryClass; +/// The SHA-256 digest the fixture cards and their source record carry. const DIGEST: &str = "2a9934acae8bf210b6a3428e553b1bcc0e220a4de113940782cd573da1ea4f4b"; +/// A second digest, used to make one vote's card hash disagree with the others. const OTHER_DIGEST: &str = "33ddbf6ffdd995dc23be2e7d4e9a05ec57b50a0ab03b0f8f44edbb26f36cf059"; +/// The versioned HASH type URL of the fixture's hash-sourced card. const EMPLOYED_BY: &str = "https://hash.ai/@h/types/entity-type/employed-by/v/1"; +/// The Wikidata entity URL of the fixture's wikidata-sourced card. const PART_OF: &str = "http://www.wikidata.org/entity/P361"; /// Composes one vote with conforming provenance. @@ -79,7 +83,7 @@ fn wikidata_card() -> Value { }) } -/// Composes a sparse hash card, with the live corpus's presence realism. +/// Composes a sparse hash card whose present and absent fields follow the live corpus. /// /// No aliases, examples, endpoint constraints, or inverse. fn hash_card() -> Value { @@ -192,6 +196,7 @@ fn vote_counts_fold_excludes_unclear_and_abstain() { assert_eq!(counts.weight(), 3); } +/// A document with a different schema tag fails with `Schema` carrying the tag found. #[test] fn foreign_schema_is_rejected() { let json = @@ -203,6 +208,7 @@ fn foreign_schema_is_rejected() { ); } +/// Cards out of canonical identity order fail with `UnorderedCards` naming the offending index. #[test] fn unordered_cards_are_rejected() { let json = document(&[hash_card(), wikidata_card()]); @@ -223,6 +229,10 @@ fn duplicated_identity_is_rejected() { ); } +/// A versionless hash base URL fails with `IdentityForm` for `Source::Hash`. +/// +/// A hash identity given as a base URL without a version fails with `IdentityForm` for +/// `Source::Hash`. #[test] fn base_url_hash_identity_is_rejected() { let mut card = hash_card(); @@ -238,6 +248,10 @@ fn base_url_hash_identity_is_rejected() { ); } +/// A wikidata identity outside `/entity/` fails with `IdentityForm` for `Source::Wikidata`. +/// +/// A wikidata identity outside the `/entity/` namespace fails with `IdentityForm` for +/// `Source::Wikidata`. #[test] fn wikidata_identity_off_the_entity_namespace_is_rejected() { let mut card = wikidata_card(); @@ -253,6 +267,10 @@ fn wikidata_identity_off_the_entity_namespace_is_rejected() { ); } +/// A wikidata card missing a pin field fails with `MissingPin` naming the field. +/// +/// A wikidata card missing `retrieved_at` or `source_record_hash` fails with `MissingPin` naming +/// the field. #[test] fn wikidata_card_without_pins_is_rejected() { let mut card = wikidata_card(); @@ -278,6 +296,7 @@ fn wikidata_card_without_pins_is_rejected() { ); } +/// A hash card carrying `retrieved_at` fails with `ForbiddenPin` naming the field. #[test] fn hash_card_with_pins_is_rejected() { let mut card = hash_card(); @@ -293,6 +312,7 @@ fn hash_card_with_pins_is_rejected() { ); } +/// A URL inside the title fails with `IdentifierInContent` naming `title`. #[test] fn url_scheme_in_prose_is_rejected() { let mut card = hash_card(); @@ -308,6 +328,7 @@ fn url_scheme_in_prose_is_rejected() { ); } +/// A UUID inside the description fails with `IdentifierInContent` naming `description`. #[test] fn uuid_shaped_token_in_prose_is_rejected() { let mut card = hash_card(); @@ -323,6 +344,7 @@ fn uuid_shaped_token_in_prose_is_rejected() { ); } +/// A null description admits and parses to `None`. #[test] fn null_description_admits() { let mut card = wikidata_card(); @@ -334,6 +356,7 @@ fn null_description_admits() { assert!(corpus.cards()[0].content.description.is_none()); } +/// An empty title fails with `EmptyField` naming `title`. #[test] fn empty_required_field_is_rejected() { let mut card = hash_card(); @@ -349,6 +372,7 @@ fn empty_required_field_is_rejected() { ); } +/// An empty string inside an axis list fails with `EmptyField` naming the axis. #[test] fn empty_axis_entry_is_rejected() { let mut card = hash_card(); @@ -364,6 +388,7 @@ fn empty_axis_entry_is_rejected() { ); } +/// A shot-excluded card that still carries votes fails with `ShotExcludedVotes`. #[test] fn shot_excluded_card_with_votes_is_rejected() { let mut card = hash_card(); @@ -376,6 +401,7 @@ fn shot_excluded_card_with_votes_is_rejected() { ); } +/// A shot-excluded card without votes admits with the flag set and zero vote weight. #[test] fn shot_excluded_card_without_votes_admits() { let mut card = hash_card(); @@ -388,6 +414,7 @@ fn shot_excluded_card_without_votes_admits() { assert_eq!(corpus.cards()[0].vote_counts().weight(), 0); } +/// A card whose votes are all unclear admits, counting the unclear votes and weighing zero. #[test] fn all_unclear_card_admits_with_zero_weight() { let mut card = hash_card(); @@ -401,6 +428,7 @@ fn all_unclear_card_admits_with_zero_weight() { assert_eq!(counts.weight(), 0); } +/// A card whose only vote abstains fails with `NoEvidence`. #[test] fn abstain_only_card_is_rejected() { let mut card = hash_card(); @@ -413,6 +441,7 @@ fn abstain_only_card_is_rejected() { ); } +/// A card with no votes and no excusing flag fails with `NoEvidence`. #[test] fn voteless_card_without_flags_is_rejected() { let mut card = hash_card(); @@ -425,6 +454,7 @@ fn voteless_card_without_flags_is_rejected() { ); } +/// Votes whose card hashes disagree fail with `DisagreeingCardHash`. #[test] fn disagreeing_vote_card_hashes_are_rejected() { let mut card = hash_card(); @@ -437,6 +467,7 @@ fn disagreeing_vote_card_hashes_are_rejected() { ); } +/// A card held out as `proximal` admits without votes and reports the proximal geometry class. #[test] fn holdout_card_admits_without_geometry_votes() { let mut card = hash_card(); @@ -459,6 +490,7 @@ fn holdout_card_admits_without_geometry_votes() { ); } +/// A card held out as `unclear` admits without votes and reports no geometry class. #[test] fn unclear_holdout_admits() { let mut card = hash_card(); @@ -477,6 +509,10 @@ fn unclear_holdout_admits() { ); } +/// An inverted endpoint constraint fails with `EndpointBounds` naming card and constraint. +/// +/// An endpoint constraint whose minimum exceeds its maximum fails with `EndpointBounds` naming the +/// card and constraint. #[test] fn inverted_endpoint_bounds_are_rejected() { let mut card = wikidata_card(); @@ -493,6 +529,7 @@ fn inverted_endpoint_bounds_are_rejected() { ); } +/// Removing a tristate constraint key fails at the JSON layer with `Json`. #[test] fn missing_tristate_key_is_rejected() { let mut card = hash_card(); @@ -508,6 +545,7 @@ fn missing_tristate_key_is_rejected() { ); } +/// An unknown card field fails at the JSON layer with `Json`. #[test] fn unknown_field_is_rejected() { let mut card = hash_card(); @@ -520,6 +558,7 @@ fn unknown_field_is_rejected() { ); } +/// An empty `model_pinned` on a vote fails with `EmptyVoteField` naming the card, vote and field. #[test] fn empty_vote_provenance_field_is_rejected() { let mut card = hash_card(); diff --git a/libs/@local/graph/atlas/src/salt/policy/artifact/mod.rs b/libs/@local/graph/atlas/src/salt/policy/artifact/mod.rs index 0d5b048d8df..969a6372f28 100644 --- a/libs/@local/graph/atlas/src/salt/policy/artifact/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/artifact/mod.rs @@ -1,10 +1,10 @@ //! Writes the resolved policy table to one policy file and reads it back through a mapping. //! //! The certified table publishes as one [`crate::file::policy`] file in the strictly ascending -//! relation order its type carries. [`PolicyTableArchive`] -//! reopens the file over a whole-file mapping and validates the table invariants once. An open -//! table then serves its rows as a borrowed [`RelationPolicy`] slice. The domain type's `repr(C)` -//! layout is the file's pinned wire row, so reads decode nothing. +//! relation order its type carries. [`PolicyTableArchive`] reopens the file over a whole-file +//! mapping and validates the table invariants once. An open table then serves its rows as a +//! borrowed [`RelationPolicy`] slice. The domain type's `repr(C)` layout is the file's pinned wire +//! row, and reads therefore decode nothing. #![cfg_attr( not(test), expect( @@ -110,8 +110,8 @@ impl WriteAs for CertifiedPolicies {} /// A published policy table opened over its mapped file. /// /// Construction checks the table invariants once (relations strictly ascending, every value in its -/// domain), so an open table only serves valid policies and consumers re-validate nothing. The rows -/// stay in the page cache under memory pressure and off the heap. +/// domain). An open table therefore only serves valid policies, and consumers re-validate nothing. +/// The rows stay in the page cache under memory pressure and off the heap. #[derive(Debug)] pub(crate) struct PolicyTableArchive { file: PolicyFile, @@ -136,8 +136,8 @@ impl PolicyTableArchive { }); } - // The domain types carry every value bound in their bit validity, so the typed - // try-cast is the whole domain check. + // The domain types carry every value bound in their bit validity, and the typed + // try-cast is therefore the whole domain check. if let Some(index) = rows .iter() .position(|row| RelationPolicy::try_read_from_bytes(row.as_bytes()).is_err()) diff --git a/libs/@local/graph/atlas/src/salt/policy/artifact/tests.rs b/libs/@local/graph/atlas/src/salt/policy/artifact/tests.rs index ff2681af571..82fa1c8742e 100644 --- a/libs/@local/graph/atlas/src/salt/policy/artifact/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/artifact/tests.rs @@ -23,6 +23,7 @@ fn scratch(name: &str) -> PathBuf { dir.join(name) } +/// Builds a policy with distinct selected and attraction distributions. fn policy(relation: u64, coincident: UnitFraction) -> RelationPolicy { RelationPolicy { relation: OntologyRowId::new(relation), @@ -40,6 +41,7 @@ fn policy(relation: u64, coincident: UnitFraction) -> RelationPolicy { } } +/// Builds ascending policies spanning zero, intermediate, and full Coincident mass. fn fixture() -> Vec { vec![ policy(2, unit_fraction!(0.0)), @@ -48,6 +50,7 @@ fn fixture() -> Vec { ] } +/// The certified fixture table written to bytes. fn fixture_bytes() -> Vec { let policies = CertifiedPolicies::new(fixture()).expect("the fixture is strictly ascending"); let mut bytes = Vec::new(); @@ -65,6 +68,7 @@ fn reopen(name: &str, bytes: &[u8]) -> Result FileHeader { FileHeader::new(CANONICAL_DIMENSIONS as u64, 5, 1.5, [0.25, -0.5, 0.125]) } +/// The fixture classifier written to bytes. fn fixture_bytes() -> Vec { let mut bytes = Vec::new(); fixture() @@ -107,6 +108,7 @@ fn reopen(name: &str, bytes: &[u8]) -> Result Classifier::from_artifact(&file) } +/// Writing the fixture classifier and reopening it yields an equal classifier. #[test] fn round_trip_is_bit_exact() { let original = fixture(); @@ -115,6 +117,7 @@ fn round_trip_is_bit_exact() { assert_eq!(reopened, original); } +/// The mapped classifier predicts the same output as the fitted one on a patterned embedding. #[test] fn mapped_and_fitted_predictions_agree() { const PATTERN: [f32; 8] = [-1.0, -0.25, 0.5, -0.75, 0.0, 0.75, -0.5, 0.25]; @@ -138,6 +141,7 @@ fn mapped_and_fitted_predictions_agree() { assert_eq!(actual, expected); } +/// A file whose coefficient rows are four wide fails to open with `Dimension { dimension: 4 }`. #[test] fn rejects_foreign_dimension() { let path = scratch("dimension.clsf"); @@ -162,6 +166,10 @@ fn rejects_foreign_dimension() { ); } +/// A negative temperature fails with `Temperature`, an infinite intercept names its class. +/// +/// A negative temperature fails with `Temperature` carrying the value and an infinite intercept +/// with `NonFiniteIntercept` naming its class. #[test] fn rejects_tampered_scalars() { // Temperature at header offset 32. @@ -183,6 +191,10 @@ fn rejects_tampered_scalars() { ); } +/// A NaN coefficient, a NaN mean and a zeroed inverse scale fail with their variants. +/// +/// A NaN coefficient fails with `NonFiniteCoefficient` naming class and component, a NaN mean with +/// `NonFiniteMean`, and a zeroed inverse scale with `InverseScale`. #[test] fn rejects_tampered_vectors() { let nan = f64::NAN.to_le_bytes(); @@ -229,6 +241,10 @@ fn rejects_tampered_vectors() { ); } +/// Lowered, negative and absent distances fail with their three variants. +/// +/// A distance lowered below its predecessors fails with `UnorderedDistances`, a negative one with +/// `Distance`, and a zeroed distance count with `EmptyDistances`. #[test] fn rejects_tampered_distances() { let distances_offset = usize::try_from( diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/applicability.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/applicability.rs index 9347c17e201..88dfd2af6d1 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/applicability.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/applicability.rs @@ -9,7 +9,7 @@ //! //! which regularizes sparse high-dimensional samples while retaining dimension-level scale where //! the sample supports it. The fit sorts and retains the training rows' own standardized distances, -//! so a prediction's applicability is its empirical upper-tail rank. An embedding far from every +//! and a prediction's applicability is its empirical upper-tail rank. An embedding far from every //! training row scores near zero, flagging the prediction as unsupported. Applicability is evidence //! about the embedding. It is not a fourth geometry class. diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/calibration.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/calibration.rs index d5454b7fcdd..1123c8e57cf 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/calibration.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/calibration.rs @@ -1,9 +1,10 @@ //! Scalar temperature calibration over out-of-fold logits. //! -//! [`fit_temperature`] minimizes weighted-mean soft-label cross-entropy over `T` in `[0.05, 20]` by -//! golden-section search in `ln T`, with `T = 1` always among the final candidates; ties prefer the -//! temperature closest to one. [`metrics`] reports cross-entropy and Brier score at `T = 1` and at -//! the deployment temperature, so raw and calibrated quality both stay visible. +//! [`fit_temperature`] searches for a temperature T ∈ [0.05, 20] minimizing weighted-mean +//! soft-label cross-entropy. Golden-section search in ln T uses a fixed iteration budget. The final +//! candidates always include T = 1, and equal losses prefer smaller |ln T|, then smaller ln T. +//! [`metrics`] reports cross-entropy and Brier score at T = 1 and at the deployment temperature to +//! compare raw and calibrated quality. use hashql_core::id::IdSlice; @@ -19,13 +20,17 @@ type Logits = IdSlice; /// The card-row-aligned training-row column calibration reads. type Rows = IdSlice; +/// The lower end of the temperature search interval. const TEMPERATURE_MINIMUM: f64 = 0.05; +/// The upper end of the temperature search interval. const TEMPERATURE_MAXIMUM: f64 = 20.0; -// The bracket spans ln(400) ~ 6 nats and contracts by 0.618 per -// iteration, reaching f64 resolution near iteration 74; 96 bounds it -// with margin at negligible cost. +/// The golden-section iteration budget over `ln T`. +// in exact arithmetic, the bracket width after n iterations is ln(400) · (φ − 1)ⁿ, where φ is the +// golden ratio. At n = 96 this is below f64::EPSILON. The budget permits that contraction without +// requiring floating-point loss comparisons to locate the minimizer to the same precision. const TEMPERATURE_ITERATIONS: usize = 96; -// 1 / φ = φ - 1; the subtraction is exact (φ ∈ [1, 2]). +/// The golden-section contraction ratio `1/φ`. +// 1/φ = φ − 1. Subtracting 1 from the stored φ is exact: both operands are within a factor of two. const GOLDEN_RATIO_CONJUGATE: f64 = core::f64::consts::GOLDEN_RATIO - 1.0; /// Probability floor inside the cross-entropy logarithm. const PROBABILITY_FLOOR: f64 = 1.0e-12; @@ -40,6 +45,12 @@ pub(super) struct ValidationMetrics { } /// Fits the deployment temperature on out-of-fold logits. +/// +/// # Panics +/// +/// In debug builds, panics if [`Posterior::softmax`] produces a component outside `[0, 1]`, +/// including NaN. Finite logits and positive finite search temperatures produce components in that +/// interval. pub(super) fn fit_temperature(rows: &Rows, logits: &Logits) -> f64 { let mut lower = TEMPERATURE_MINIMUM.ln(); let mut upper = TEMPERATURE_MAXIMUM.ln(); @@ -51,8 +62,8 @@ pub(super) fn fit_temperature(rows: &Rows, logits: &Logits) -> f64 { let mut right_value = cross_entropy(rows, logits, right.exp()); for _ in 0..TEMPERATURE_ITERATIONS { - // Equal values contract toward `ln T = 0`: the tie preference - // for a temperature near one applies during the search too. + // equal losses prefer the point with smaller |ln T|, applying the identity-temperature + // preference during the search. if (left_value, left.abs()) <= (right_value, right.abs()) { upper = right; right = left; @@ -68,8 +79,9 @@ pub(super) fn fit_temperature(rows: &Rows, logits: &Logits) -> f64 { } } - // `0.0` is `ln 1`: the identity temperature always competes, so - // calibration can never worsen the cross-entropy it minimizes. + // the minimum of a finite candidate set is at most each member. Including ln T = 0 makes the + // raw computed loss one candidate. Therefore the selected computed cross-entropy can never + // exceed the raw computed cross-entropy when all candidate losses are finite. [lower, left, 0.0, right, upper] .into_iter() .map(|candidate| (cross_entropy(rows, logits, candidate.exp()), candidate)) @@ -88,15 +100,22 @@ pub(super) fn fit_temperature(rows: &Rows, logits: &Logits) -> f64 { /// /// # Errors /// -/// Returns [`FitError::NonFinite`] when a weighted mean leaves the non-negative finite domain. -/// Both means are non-negative by derivation (targets and weights are validated at -/// `TrainingSet::new`, posteriors lie in the unit interval, and the cross-entropy logarithm is -/// floored), so the refusal names a weights defect. +/// Returns [`FitError::NonFinite`] when any resulting weighted mean is negative or non-finite, +/// including a mean over no paired rows. +/// +/// # Panics +/// +/// In debug builds, panics if [`Posterior::softmax`] produces a component outside `[0, 1]`, +/// including NaN. Finite logits and a positive finite `temperature` produce components in that +/// interval. pub(super) fn metrics( rows: &Rows, logits: &Logits, temperature: f64, ) -> Result { + // for target q and posterior p in [0, 1], −q ln(max(p, 10⁻¹²)) and (p − q)² are finite and + // non-negative. Positive finite weights can still overflow the weighted loss or total weight. + // Therefore each resulting mean must be finite and non-negative for admission. let admit = |value: f64| DNonNegative::new(value).ok_or(FitError::NonFinite); Ok(ValidationMetrics { @@ -107,12 +126,17 @@ pub(super) fn metrics( }) } -/// Weighted-mean soft-label cross-entropy of the uncalibrated posteriors. +/// Computes weighted-mean soft-label cross-entropy at the identity temperature. +/// +/// # Panics +/// +/// In debug builds, panics if [`Posterior::softmax`] produces a component outside `[0, 1]`, +/// including NaN. Finite logits produce components in that interval at T = 1. pub(super) fn raw_cross_entropy(rows: &Rows, logits: &Logits) -> f64 { cross_entropy(rows, logits, 1.0) } -/// Weighted-mean soft-label cross-entropy at one temperature. +/// Computes weighted-mean soft-label cross-entropy at one temperature. fn cross_entropy(rows: &Rows, logits: &Logits, temperature: f64) -> f64 { let mut loss = 0.0; let mut total_weight = 0.0; @@ -133,7 +157,7 @@ fn cross_entropy(rows: &Rows, logits: &Logits, temperature: f64) -> f64 { loss / total_weight } -/// Weighted-mean Brier score at one temperature. +/// Computes the weighted-mean Brier score at one temperature. fn brier(rows: &Rows, logits: &Logits, temperature: f64) -> f64 { let mut loss = 0.0; let mut total_weight = 0.0; diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/mod.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/mod.rs index 9c41ee1437e..0d3d1525b3e 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/mod.rs @@ -11,16 +11,17 @@ //! through the deterministic bounded trust-region exact-Newton [`solver`], which operates in //! contrast coordinates and certifies every solution against its gradient threshold. The data Gram //! matrix behind the solver's row-space factorization assembles once per fit and every fold solve -//! reads its subset through a member view. Whole relation groups go to seeded, size-balanced folds -//! before fitting, so near-duplicate corpus entries never straddle a train/validation split. -//! [`regularization`] selects the penalty strength λ over those folds. Every candidate's fold -//! models fit in parallel and the minimum out-of-fold cross-entropy wins, with an exact tie +//! reads its subset through a member view. Whole validation groups go to seeded, size-balanced +//! folds before fitting, and no fold divides a group. The corpus assembly decides which rows share +//! a group (related or near-duplicate cards, up to its budgeted relaxation) before the rows reach +//! the fit. [`regularization`] selects the penalty strength λ over those folds. Every candidate's +//! fold models fit in parallel and the minimum out-of-fold cross-entropy wins, with an exact tie //! preferring the stronger penalty. The deployment model then fits at the winning strength over the //! complete corpus. The winner's concatenated out-of-fold logits calibrate one scalar deployment //! temperature ([`calibration`]), and [`applicability`] fits the applicability distribution over //! the complete corpus. Each fit's arithmetic is sequential and every candidate reuses the same -//! fold assignment, so the result is deterministic. The out-of-fold metrics judge the selected -//! configuration on the same folds that chose it. +//! fold assignment. The result is therefore deterministic. The out-of-fold metrics judge the +//! selected configuration on the same folds that chose it. //! //! Every fit certifies at the configured gradient threshold. A solve that ends at any typed //! terminal is an error, and the fit returns no best-effort model. The typed terminals are an @@ -159,10 +160,12 @@ impl Error for FitError {} /// One soft label, vote weight, and indivisible validation group. /// -/// The group digest names the finest unit a validation split never divides. Corpus assembly unions -/// every leakage axis (relation family, inverse pair, base URL, publisher, near-duplicate card -/// family) into this one label before rows reach the fit, so near-identical corpus entries can -/// never straddle a train/validation boundary and inflate the out-of-fold metrics. +/// The group digest names the finest unit a validation split never divides. Corpus assembly +/// derives it before rows reach the fit, uniting cards through their identity and inverse +/// identities, their relation family and base URL, and near-duplicate embeddings, and relaxing the +/// family, base-URL and near-duplicate unions where a group would exceed its budget. The fit keeps +/// the groups it receives whole and reconstructs no leakage relation itself. A pair the assembly +/// separated under its budget can therefore straddle a train/validation boundary. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct TrainingRow { /// Soft target over the geometry classes, in class order. @@ -176,7 +179,7 @@ pub(crate) struct TrainingRow { /// Validated borrowed classifier training data. /// /// Both columns index by the corpus's card rows: the row at a card row labels the embedding at the -/// same card row. The types carry the alignment claim across domains; the lengths still validate at +/// same card row. The types carry the alignment claim across domains. The lengths still validate at /// construction. #[derive(Debug, Copy, Clone)] pub(crate) struct TrainingSet<'training> { @@ -206,10 +209,9 @@ impl<'training> TrainingSet<'training> { }); } - // The scan touches every component once (a few MB of SIMD - // compares at annotation-corpus scale, well under a - // millisecond) and runs once per fit; the borrowed slice type - // carries no finiteness guarantee of its own. + // The scan touches every component once (a few MB of SIMD compares at annotation-corpus + // scale, well under a millisecond) and runs once per fit. The borrowed slice type + // has no finiteness guarantee of its own. for (row_index, embedding) in embeddings.iter_enumerated() { if embedding.is_finite() { continue; @@ -285,7 +287,7 @@ impl<'training> TrainingSet<'training> { /// Solver and grouped-validation settings. /// /// The solver defaults are the deployment configuration, with the regularization strength selected -/// per fit ([`regularization`]); the out-of-fold metrics in [`FitEvidence`] judge the selected +/// per fit ([`regularization`]). The out-of-fold metrics in [`FitEvidence`] judge the selected /// configuration. #[derive(Debug, Copy, Clone, PartialEq, Default)] pub(crate) struct FitConfig { @@ -351,7 +353,7 @@ pub(crate) struct Fit { /// /// # Errors /// -/// Returns a [`FitError`] for invalid configuration, too few relation groups, a training portion +/// Returns a [`FitError`] for too few relation groups, a training portion /// violating the preparation contract, a solve ending at a typed terminal, or a non-finite /// out-of-fold evaluation. /// @@ -368,10 +370,10 @@ pub(crate) fn fit( let folds = grouped_folds(training.rows(), config.folds, config.seed)?; progress.classifier_started(config.folds); - // One Gram assembly serves every fold solve and the deployment fit; the assembly charge - // rides the deployment fit's counters, and fold solves read the shared matrix uncharged. - // The matrix speaks packed positional row indices by its own contract, so the card-row - // domain ends at its assembly boundary. + // One Gram assembly serves every fold solve and the deployment fit. The assembly charge is + // counted against the deployment fit's counters, and fold solves read the shared matrix + // uncharged. The matrix indexes packed positional rows by its own contract, and the + // card-row domain ends at its assembly boundary. let mut assembly_counters = WorkCounters::default(); let gram = Gram::assemble(training.embeddings.as_raw(), &mut assembly_counters); @@ -439,7 +441,7 @@ impl FoldedTraining<'_> { ) -> Result<(Parameters, u64), FitError> { // The held-out fold's complement materializes densely because the solver traverses whole // corpora and fold membership is not its contract. The gather re-bases the complement into - // the solve's own positional row space, so the card-row domain ends here. The + // the solve's own positional row space, and the card-row domain ends here. The // window records each solve row's original corpus index, which is the form the Gram // view documents. let subset = held_out.map(|held_out| { @@ -532,8 +534,18 @@ fn split_parameters( /// Assigns whole groups to seeded, size-balanced folds. /// /// Each group joins the currently smallest fold, largest group first. Equal sizes break by a seeded -/// hash of the group digest and then by the digest itself, so the assignment is deterministic and -/// independent of row order. +/// hash of the group digest and then by the digest itself. The assignment is therefore +/// deterministic and independent of row order. +/// +/// # Errors +/// +/// Returns [`FitError::InsufficientGroups`] when the rows hold fewer distinct groups than +/// `fold_count`. +/// +/// # Panics +/// +/// This panics for a zero `fold_count` with nonempty rows, where no fold can receive the first +/// group. Zero folds with empty rows return an empty assignment. fn grouped_folds( rows: &IdSlice, fold_count: usize, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/objective.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/objective.rs index c772a28e005..af33c638531 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/objective.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/objective.rs @@ -2,7 +2,7 @@ //! //! Parameters are one flat vector `[w_C | w_P | w_O | b]`: the three coefficient rows in class //! order followed by the three intercepts. The bounded solver ([`solver`](super::solver)) fits -//! models in contrast coordinates; [`expand_point`] returns its solutions to this layout, and +//! models in contrast coordinates. [`expand_point`] returns its solutions to this layout, and //! [`logits`] evaluates one embedding under it. `f32` embeddings enter the double-precision logits //! through [`AlignedVecN::dot_wide`]. @@ -22,7 +22,7 @@ pub(super) const PARAMETER_COUNT: usize = COEFFICIENT_COUNT + GeometryClass::COU /// The flat parameter vector. pub(super) type Parameters = BoxedDVecN; -/// Class logits of one embedding under flat parameters. +/// Computes the class logits of one embedding under flat parameters. pub(super) fn logits( parameters: &Parameters, embedding: &AlignedVecN, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/regularization.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/regularization.rs index 4c17a82055c..17e2d2b9fd1 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/regularization.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/regularization.rs @@ -1,15 +1,15 @@ //! Regularization-strength selection by grouped cross-validation. //! //! [`select`](FoldedTraining::select) fits one model per candidate strength and fold over the -//! shared seeded fold -//! assignment. It scores every candidate by the weighted-mean out-of-fold cross-entropy of its -//! uncalibrated posteriors and picks the minimizer. An exact tie prefers the stronger penalty. The -//! fit evidence records the full curve alongside the winner, so a reader sees the plateau the -//! selection stood on. +//! shared seeded fold assignment. It scores every candidate by the weighted-mean out-of-fold +//! cross-entropy of its uncalibrated posteriors and picks the minimizer. An exact tie prefers the +//! stronger penalty. The fit evidence records the full curve alongside the winner to expose flat +//! regions and competing minima. //! -//! The candidate grid never changes. The preparation scaling normalizes coefficient coordinates -//! before the solver sees any candidate, so the parity strength `1.0` is the natural center and the -//! grid brackets it by three decades either side in a logarithmic 1-3 progression. +//! The candidate grid never changes. It brackets the default strength `1.0` by three decades either +//! side in a logarithmic 1-3 progression. Each candidate uses the solver's [initial +//! scaling](super::solver::prepare) to normalize coordinate curvature. This scaling conditions the +//! solve without changing the candidate's penalty in the physical objective. use core::{ iter, @@ -69,8 +69,10 @@ pub(super) struct Selection { pub out_of_fold_logits: IdVec, } -/// Returns the position of the minimum cross-entropy, with an exact tie going to the stronger -/// penalty. +/// Returns the index of the least cross-entropy, preferring the later entry on an exact tie. +/// +/// An ascending strength curve therefore prefers the stronger penalty. Returns zero for an empty +/// curve. pub(super) fn winner(curve: &[RegularizationReading]) -> usize { let mut winner = 0; for (candidate, reading) in curve.iter().enumerate() { @@ -85,9 +87,8 @@ pub(super) fn winner(curve: &[RegularizationReading]) -> usize { impl FoldedTraining<'_> { /// Selects the deployment regularization strength over the shared fold assignment. /// - /// Every `(candidate, fold)` model fits in parallel; a fold reports completed to `progress` - /// when its last candidate finishes, so the fold counter keeps its meaning under the widened - /// wave. + /// Every `(candidate, fold)` model fits in parallel. A fold reports completed to `progress` + /// only when its last candidate succeeds. Failed models return before reporting completion. /// /// # Errors /// @@ -136,9 +137,12 @@ impl FoldedTraining<'_> { return Err(FitError::NonFinite); } - // Non-negative by the mean's own derivation (targets and weights are validated at - // `TrainingSet::new`, posteriors lie in the unit interval), so the construction refuses - // exactly the non-finite escapes: a NaN or infinite mean is a weights defect. + // nonnegative targets and probabilities in [0, 1] give nonnegative cross-entropy. + // TrainingSet::new validates the targets and positive finite weights. The finite logits + // above yield valid posteriors at T = 1, and the 10⁻¹² probability floor makes every + // row loss finite. The weighted loss or total weight can still overflow during + // reduction. Therefore the mean cannot be negative, and this conversion rejects exactly + // its non-finite results. let cross_entropy = DNonNegative::new(calibration::raw_cross_entropy( self.training.rows(), &logits, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/basis.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/basis.rs index 6eb3246e666..c91ce302ddf 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/basis.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/basis.rs @@ -1,13 +1,13 @@ //! The `HelmertV1` contrast basis between class logits and contrast coordinates. //! -//! Adding a common scalar to every class logit moves no probability, so one degree of freedom among -//! the class logits is redundant. The solver therefore parameterizes the classifier in a +//! Adding a common scalar to every class logit moves no probability, and one degree of freedom +//! among the class logits is redundant. The solver therefore parameterizes the classifier in a //! two-dimensional contrast space reached through the fixed matrix `B ∈ ℝ^(3×2)` whose rows, in //! [`GeometryClass`] discriminant order, are `(1/√2, 1/√6)`, `(−1/√2, 1/√6)`, and `(0, −2/√6)`. Its -//! columns are orthonormal (`BᵀB = I₂`) and orthogonal to the constant vector (`Bᵀ1 = 0`), so -//! contrast coordinates span exactly the shift-free directions of logit space: [`expand`] maps a -//! contrast vector to class logits and [`reduce`] projects a class-space vector back, and the -//! roundtrip is the identity up to rounding. +//! columns are orthonormal (`BᵀB = I₂`) and orthogonal to the constant vector (`Bᵀ1 = 0`). +//! Contrast coordinates therefore span exactly the shift-free directions of logit space: [`expand`] +//! maps a contrast vector to class logits and [`reduce`] projects a class-space vector back, and +//! the roundtrip is the identity up to rounding. //! //! The source fixes all six matrix entries as IEEE-754 bit patterns, and `to_bits` tests pin them. //! No platform math call recomputes them. Each magnitude is the correctly rounded value of its real @@ -25,7 +25,7 @@ const FRAC_1_SQRT_6: f64 = 0.408_248_290_463_863; /// `2/√6`, the exact double of [`FRAC_1_SQRT_6`] (bit pattern `0x3FEA20BD700C2C3E`). /// -/// Doubling is exact in binary floating point, so this is also the correctly rounded value of the +/// Doubling is exact in binary floating point, and this is also the correctly rounded value of the /// real `2/√6`. const FRAC_2_SQRT_6: f64 = 0.816_496_580_927_726; diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/boundary.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/boundary.rs index 1eb3efa0411..55e6c1e76ec 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/boundary.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/boundary.rs @@ -6,10 +6,11 @@ //! the cancellation-free quadratic root `q = −½·(b + copysign(√(b² − 4ac), b))`, and it then checks //! its own result, requiring the normalized step to lie within the gross-defect guard of unit norm //! both as built and as returned after the radius rescaling. The matching Hessian product extends -//! along the same `τ`, so a boundary step never charges a fresh Hessian-vector product. +//! along the same `τ`, and a boundary step therefore never charges a fresh Hessian-vector +//! product. //! //! Every arithmetic escape - a non-finite normalization, coefficient, discriminant, root, or -//! product - returns [`None`]; the caller maps [`None`] onto its typed no-finite-boundary-step +//! product - returns [`None`]. The caller maps [`None`] onto its typed no-finite-boundary-step //! failure. use super::{SOLVER_DIMENSIONS, flat}; @@ -23,14 +24,14 @@ use crate::math::{AlignedDVecN, BoxedDVecN, DPositive, DVecN}; /// finite-positive-τ rule instead, because either mathematical root lies on the unit boundary. /// Honest striped-fold rounding stays orders of magnitude below the guard, and the margin is /// asymmetric by intent. A false abort costs a production fit, while the final gradient certificate -/// still gates any drift the guard admits. The exact value is an implementation choice that no +/// still bounds any drift the guard admits. The exact value is an implementation choice that no /// configuration exposes, no file persists, and no cross-target identity depends on. pub(super) const GROSS_DEFECT_GUARD: f64 = 4096.0 * f64::EPSILON; /// A validated step onto the numerical trust-region boundary. /// /// The step's radius-normalized norm lies within the gross-defect guard of one, and the product -/// rides the same crossing, so later logic trusts the tag carrying this payload instead of +/// extends along the same crossing. Later logic therefore trusts a value of this type instead of /// re-deriving boundary contact from a fresh norm. #[derive(Debug, Clone, PartialEq)] pub(super) struct BoundaryStep { @@ -43,7 +44,7 @@ pub(super) struct BoundaryStep { /// Advances an interior iterate to the numerical trust-region boundary. /// /// Normalizes `u = p/Δ` and `v = d/Δ`, solves `‖u + τv‖² = 1` as `aτ² + bτ + c = 0` with -/// `a = v·v`, `b = 2·u·v`, `c = u·u − 1` (the interior iterate keeps `c < 0`, so exactly one +/// `a = v·v`, `b = 2·u·v`, `c = u·u − 1` (the interior iterate keeps `c < 0`, and exactly one /// root is positive), and picks the finite positive `τ` from the paired roots `q/a` and `c/q`. /// The built `u + τv` and the returned `(Δ·(u + τv))/Δ` must both lie within /// [`GROSS_DEFECT_GUARD`] of unit norm. @@ -79,8 +80,8 @@ pub(super) fn boundary_step( return None; } - // With c < 0 the roots q/a and c/q carry opposite signs. The selection order never varies, so a - // degenerate pair still resolves deterministically. + // With c < 0 the roots q/a and c/q carry opposite signs. The selection order never varies, and + // a degenerate pair therefore still resolves deterministically. let crossing = [root / quadratic, constant / root] .into_iter() .find_map(DPositive::new)?; diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/config.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/config.rs index 69c8a914c0a..126af6c841e 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/config.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/config.rs @@ -1,13 +1,12 @@ //! Validated solver-loop configuration. //! -//! [`SolverConfig`] carries every knob of the trust-region exact-Newton loop: the radius domain, +//! `SolverOptions` carries every knob of the trust-region exact-Newton loop: the radius domain, //! shrink and expansion factors, acceptance thresholds, convergence tolerances, ulp counts, and the -//! inclusive outer-iteration budget. Per-field domains travel in the field types - the validated -//! scalars of [`math`](crate::math) and the non-zero integers of [`core::num`] - so a configuration -//! value that exists is in domain. [`validate`](SolverConfig::validate) checks only what no field -//! type can carry alone: the radius ordering and the acceptance-threshold ordering, in declared -//! order, reporting the first violation. The preparation-side knobs ride along as -//! [`PreparationSettings`], so one validated configuration covers the whole fit. +//! inclusive outer-iteration budget. Per-field domains are carried by the field types, the +//! validated scalars of [`math`](crate::math) and the non-zero integers of [`core::num`]. +//! `SolverConfig::new` checks the radius and acceptance-threshold orderings that no field type +//! can carry alone. A [`SolverConfig`] value is therefore in domain. [`PreparationSettings`] +//! supplies the preparation-side knobs within the same configuration. //! //! The outer-iteration budget is an inclusive maximum: equality is allowed and starting one more //! iteration fails the solve. It is the loop's only work limit. Per-request work is bounded by the @@ -26,13 +25,18 @@ use crate::math::{DNonNegative, DPositive, GreaterThanOne, OpenUnitFraction}; pub(crate) enum SolverConfigError { /// The radius domain violates `minimum ≤ initial ≤ maximum`. RadiusDomain { + /// The configured minimum radius. minimum: DPositive, + /// The configured initial radius. initial: DPositive, + /// The configured maximum radius. maximum: DPositive, }, /// The acceptance thresholds violate `accept < expand`. AcceptanceThresholds { + /// The configured acceptance threshold. accept: OpenUnitFraction, + /// The configured expansion threshold. expand: OpenUnitFraction, }, } @@ -96,7 +100,7 @@ pub(crate) struct SolverConfig { } impl SolverConfig { - /// Admits the configuration or names the first violated cross-field constraint. + /// Admits raw solver options after checking their cross-field orderings. /// /// # Errors /// @@ -124,12 +128,13 @@ impl SolverConfig { Ok(()) } - /// The gradient-certificate threshold `max(absolute, relative·‖gζ,0‖₂)`, derived once from the - /// initial scaled gradient norm. + /// Derives the gradient-certificate threshold from the initial scaled gradient norm. + /// + /// The threshold is `max(absolute, relative·‖gζ,0‖₂)`. /// /// A zero threshold is valid. With the absolute floor at zero and an exactly-zero initial norm, /// only an exactly-zero gradient certifies. The derivation is total: the relative tolerance - /// lies below one, so the scaled term never exceeds the norm. The maximum of two in-domain + /// lies below one, and the scaled term never exceeds the norm. The maximum of two in-domain /// values therefore stays in domain. pub(super) const fn gradient_threshold(&self, initial_norm: DNonNegative) -> DNonNegative { self.absolute_scaled_gradient_tolerance diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/evaluate.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/evaluate.rs index c3ce40b7008..927d3b2360a 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/evaluate.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/evaluate.rs @@ -5,8 +5,8 @@ //! one stable log-sum-exp over the reference differences and zero in class order, and finally //! probabilities from the same shifted exponentials. The objective, the gradient residual `p − q`, //! the Hessian curvature `diag(p) − ppᵀ`, and the per-row contrast curvature blocks of the exact -//! Newton assembly all read these shared bytes, so no evaluation can disagree with another about a -//! row's logits. +//! Newton assembly all read these shared bytes. No evaluation can therefore disagree with another +//! about a row's logits. //! //! Every evaluation normalizes its result by the total weight `S`: //! @@ -44,15 +44,17 @@ pub(super) struct JointEvaluation { pub(super) struct CurvatureEvaluation { /// The unweighted contrast curvature `Cᵢ` per row, packed as `(c11, c21, c22)`. pub blocks: Vec<[f64; 3]>, - /// The normalized Hessian columns `H[0|e_k]` of the intercept unit directions: coefficient - /// coupling in the coefficient rows, intercept curvature in the intercepts. + /// The normalized Hessian columns `H[0|e_k]` of the intercept unit directions. + /// + /// Coefficient coupling sits in the coefficient rows, intercept curvature in the intercepts. pub intercept_columns: [ContrastVector; CONTRAST_ROWS], } /// One shared per-row logits evaluation. struct RowPrelude { - /// Reference differences `δ_c` of the leading classes. The reference's own is zero by - /// construction. + /// Reference differences `δ_c` of the leading classes. + /// + /// The reference's own is zero by construction. delta: [f64; LEADING_CLASSES], /// `logsumexp(δ, 0)`, stable under the class-order shifted fold. log_normalizer: f64, @@ -67,8 +69,8 @@ impl RowPrelude { let reference = logits[GeometryClass::COUNT - 1]; let delta: [f64; LEADING_CLASSES] = core::array::from_fn(|class| logits[class] - reference); - // Stable shifted fold over the reference differences and zero in class order; the shift - // keeps every exponential in [0, 1] and a NaN input propagates through the exponentials + // Stable shifted fold over the reference differences and zero in class order. The shift + // keeps every exponential in [0, 1], and a NaN input propagates through the exponentials // into every output. let shift = delta.into_iter().reduce(f64::max).unwrap_or(0.0).max(0.0); let exponentials: [f64; GeometryClass::COUNT] = core::array::from_fn(|class| { @@ -91,7 +93,9 @@ impl RowPrelude { } } - /// The reference-difference data loss `logsumexp(δ, 0) − Σ_c u_cδ_c`, folded in class order. + /// Computes the reference-difference data loss, folded in class order. + /// + /// The loss is `logsumexp(δ, 0) − Σ_c u_cδ_c`. fn loss(&self, leading: [f64; LEADING_CLASSES]) -> f64 { let mut value = self.log_normalizer; for (target, delta) in leading.into_iter().zip(self.delta) { @@ -140,8 +144,8 @@ pub(super) struct CurvatureReading { /// Every row's curvature reading at one parameter point. /// /// The readings pair each scale with its weight at the source, and the total is the -/// preparation-validated `S`, so a weight share computed from the census divides by a positive -/// and finite total. +/// preparation-validated `S`. A weight share computed from the census therefore divides by a +/// positive and finite total. #[derive(Debug)] pub(super) struct CurvatureCensus { /// The per-row readings, in ascending original row order. @@ -326,7 +330,7 @@ impl Prepared<'_> { /// normalized Hessian columns of the intercept unit directions, /// `H[0|e_k] = (1/S)·Σᵢ wᵢ·(Cᵢe_k)·[x̄ᵢᵀ | 1]` - the coefficient coupling block and the /// intercept curvature block of the Newton system in one pass. Intercepts carry no - /// regularization, so the columns are complete as accumulated. + /// regularization, and the columns are complete as accumulated. /// /// Rows accumulate in ascending original index and classes fold in discriminant order, as /// every other evaluation here. Returns [`None`] for a non-finite request, which visits no @@ -401,15 +405,15 @@ impl Prepared<'_> { /// Takes the census of every row's data-Hessian curvature scale `max_c p_c(1−p_c)`. /// - /// One traversal pairs each row's scale with that row's training weight, so the pairing - /// cannot drift, and the census carries the preparation-validated total weight for its weight + /// One traversal pairs each row's scale with that row's training weight, and the pairing + /// cannot drift. The census carries the preparation-validated total weight for its weight /// shares. The scale reads the same shared logits path as the objective, gradient, and - /// Hessian-vector product, so the census cannot disagree with them about a row's + /// Hessian-vector product, and the census cannot disagree with them about a row's /// probabilities. A diagnostic observer for the solver's report probe: it visits every row /// but charges no work counters, because it participates in no solve. pub(super) fn curvature_census(&self, parameters: &ContrastVector) -> CurvatureCensus { // The zip is total: preparation refuses a corpus whose embedding and row counts differ, - // so the two slices share one validated row domain. + // and the two slices share one validated row domain. let readings = self .embeddings .iter() @@ -438,9 +442,9 @@ impl Prepared<'_> { /// Adds the regularizer to the accumulated data loss and normalizes by the total weight. fn finish_objective(&self, data_loss: f64, parameters: &ContrastVector) -> f64 { // ‖A‖² through the house striped kernel, one row at a time in contrast order. The - // derivation rides raw to one exit because the parameters are unbounded solver state: + // derivation stays raw to one exit because the parameters are unbounded solver state: // initialization deliberately admits a non-finite origin objective, and resolution and - // final certification refuse it by name where the design says so. + // final certification refuse it by name. let mut coefficient_norm = Derivation::::ZERO; for row in ¶meters.coefficients { coefficient_norm += row.norm_squared(); diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/flat.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/flat.rs index 2f82357efbb..0ef6fe5de05 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/flat.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/flat.rs @@ -3,13 +3,13 @@ //! The update recurrences of the solver loop - steps, residuals, directions, and gradient //! differences - are componentwise advances `base + factor·along`, one fused multiply-add per //! coordinate through the house kernel [`DVecN::mul_add`]. Results are plain values that may be -//! non-finite; [`AlignedDVecN::is_finite`] is the shared escape check each caller maps onto its own +//! non-finite. [`AlignedDVecN::is_finite`] is the shared escape check each caller maps onto its own //! typed outcome. use super::SOLVER_DIMENSIONS; use crate::math::{AlignedDVecN, BoxedDVecN, DFinite, DVecN}; -/// The componentwise advance `base + factor·along`, one fused multiply-add per coordinate. +/// Computes the componentwise advance `base + factor·along`, one fused multiply-add per coordinate. pub(super) fn advance( base: &AlignedDVecN, factor: DFinite, @@ -20,7 +20,7 @@ pub(super) fn advance( advanced } -/// The componentwise negation. +/// Computes the componentwise negation. pub(super) fn negated(vector: &AlignedDVecN) -> BoxedDVecN { let mut negation = BoxedDVecN::new(DVecN::from_ref(vector.as_array())); negation.negate(); diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/gram.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/gram.rs index 2ddfd7024de..12ec5db4593 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/gram.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/gram.rs @@ -1,12 +1,12 @@ //! The data Gram matrix `K` and the fold views the inner solves read it through. //! //! The exact Newton engine's capacitance blocks read pairwise embedding products `Kᵢⱼ = x̄ᵢᵀx̄ⱼ`. -//! Embeddings never change across outer iterations, regularization candidates, or folds, so the fit -//! assembles [`Gram`] once over the full training corpus and every fold solve reads its subset -//! through a [`GramView`] carrying the fold's member indices. Entries accumulate in `f64` through -//! the exact-product kernel [`AlignedVecN::dot_accumulated`], one independent dot per entry, so the -//! assembled bytes are deterministic and a view's entries equal a direct assembly over the subset -//! bit for bit. +//! Embeddings never change across outer iterations, regularization candidates, or folds. The fit +//! therefore assembles [`Gram`] once over the full training corpus and every fold solve reads its +//! subset through a [`GramView`] carrying the fold's member indices. Entries accumulate in `f64` +//! through the exact-product kernel [`AlignedVecN::dot_accumulated`], one independent dot per +//! entry. The assembled bytes are therefore deterministic, and a view's entries equal a direct +//! assembly over the subset bit for bit. //! //! Storage is the packed lower triangle - `n(n+1)/2` components for `n` rows - and lookups are //! symmetric: `entry(i, j)` and `entry(j, i)` read the same component. @@ -26,7 +26,7 @@ pub(crate) struct Gram { impl Gram { /// Assembles the Gram matrix over the corpus embeddings in one charged pass. /// - /// Each entry is one independent double-accumulated dot, so the assembly is deterministic at + /// Each entry is one independent double-accumulated dot, and the assembly is deterministic at /// any traversal order. Rows fill in ascending index. The work is `n(n+1)/2` wide dots, once /// per fit. pub(crate) fn assemble( @@ -50,7 +50,7 @@ impl Gram { } } - /// Corpus rows covered by the matrix. + /// Returns the corpus rows covered by the matrix. pub(crate) const fn order(&self) -> usize { self.order } @@ -92,7 +92,7 @@ const fn packed_length(order: usize) -> usize { /// /// A full-corpus solve reads the matrix directly. A fold solve carries the ascending original /// indices of its member rows and reads the full matrix through them. Either way, `entry(i, j)` -/// speaks the solve's own row indices. +/// takes the solve's own row indices. #[derive(Debug, Copy, Clone)] pub(crate) struct GramView<'fit> { /// The fit-level matrix. @@ -104,7 +104,7 @@ pub(crate) struct GramView<'fit> { } impl<'fit> GramView<'fit> { - /// The identity view of a full-corpus solve. + /// Creates the identity view of a full-corpus solve. pub(crate) const fn full(gram: &'fit Gram) -> Self { Self { gram, @@ -112,7 +112,7 @@ impl<'fit> GramView<'fit> { } } - /// A fold view reading the member rows of the full matrix. + /// Creates a fold view reading the member rows of the full matrix. /// /// # Panics /// @@ -129,7 +129,7 @@ impl<'fit> GramView<'fit> { } } - /// Rows covered by the view. + /// Returns the rows covered by the view. pub(crate) fn order(&self) -> usize { self.members .map_or_else(|| self.gram.order(), <[usize]>::len) diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/mod.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/mod.rs index 58a3d8d5f52..4594ed71159 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/mod.rs @@ -2,16 +2,16 @@ //! //! The classifier's external objective combines weighted soft-target cross-entropy with //! coefficient-only L2 regularization over the class logits. A common shift of the logits moves no -//! probability, so the solver operates in the contrast space of dimension one below the class -//! count, reached through the fixed [`basis`], where coefficient regularization plus positive -//! aggregate class mass makes the minimizer finite and unique. +//! probability, and the solver therefore operates in the contrast space of dimension one below +//! the class count, reached through the fixed [`basis`], where coefficient regularization plus +//! positive aggregate class mass makes the minimizer finite and unique. //! //! Everything here is deterministic bounded work. Arithmetic visits rows in ascending original //! index and classes in discriminant order, no step uses randomness, and explicit counters charge //! every row traversal. Every reduction whose value steers a branch goes through the checked //! vector reductions ([`AlignedDVecN::checked_dot`], [`AlignedDVecN::checked_norm_squared`], and -//! [`AlignedDVecN::checked_stable_l2`]), so an overflow or NaN never steers a control decision, -//! and each call site maps [`None`] onto its own typed failure. +//! [`AlignedDVecN::checked_stable_l2`]). An overflow or NaN therefore never steers a control +//! decision, and each call site maps [`None`] onto its own typed failure. //! //! # Coordinates and vector layout //! @@ -89,7 +89,7 @@ pub(super) struct ContrastVector { } impl ContrastVector { - /// The zero vector. + /// Returns the zero vector. fn zero() -> Self { Self { coefficients: core::array::from_fn(|_index| BoxedDVecN::zero()), @@ -126,13 +126,13 @@ impl ContrastVector { flat } - /// Whether every coordinate is finite. + /// Returns whether every coordinate is finite. fn is_finite(&self) -> bool { self.coefficients.iter().all(|row| row.is_finite()) && self.intercepts.iter().all(|value| value.is_finite()) } - /// Contrast logits `t = T·x̄`: one wide dot plus the intercept per contrast row. + /// Computes the contrast logits `t = T·x̄`: one wide dot plus the intercept per contrast row. fn logits(&self, embedding: &AlignedVecN) -> [f64; CONTRAST_ROWS] { core::array::from_fn(|row| { embedding.dot_wide(&self.coefficients[row]) + self.intercepts[row] diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/newton.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/newton.rs index 97b5a2cdff7..2087a41c8ec 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/newton.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/newton.rs @@ -2,8 +2,8 @@ //! //! One inner solve computes the Newton step of the local quadratic model at the accepted point, //! exactly and at conditioning-independent cost. The data Hessian has rank at most `2n` for `n` -//! corpus rows, far below the parameter dimension, so the coefficient block solves through the -//! Woodbury identity in row space and the two intercepts through a Schur complement: +//! corpus rows, far below the parameter dimension. The coefficient block therefore solves through +//! the Woodbury identity in row space and the two intercepts through a Schur complement: //! //! ```text //! A₁₁ = (λ/S)·(I + ŨŨᵀ), ũ_{i,k} = √(wᵢ/λ)·(Lᵢ[:,k] ⊗ x̄ᵢ), LᵢLᵢᵀ = Cᵢ @@ -74,33 +74,33 @@ pub(super) struct NewtonOutcome { } impl NewtonOutcome { - /// The returned step `p` in scaled coordinates. + /// Returns the step `p` in scaled coordinates. pub(super) const fn step(&self) -> &AlignedDVecN { &self.step } - /// The matching product `Hζ·p` returned with the step. + /// Returns the matching product `Hζ·p` returned with the step. pub(super) const fn hessian_step(&self) -> &AlignedDVecN { &self.hessian_step } - /// Whether the outcome carries a validated boundary crossing. + /// Returns whether the outcome carries a validated boundary crossing. pub(super) const fn is_boundary(&self) -> bool { !matches!(self.tag, NewtonTag::NewtonInterior) } - /// The terminating tag of the outcome. + /// Returns the terminating tag of the outcome. pub(super) const fn tag(&self) -> NewtonTag { self.tag } - /// The relative Newton residual, where the solve priced the Newton product. + /// Returns the relative Newton residual, where the solve priced the Newton product. pub(super) const fn residual(&self) -> Option { self.residual } } -/// The typed non-finite failure of one Newton stage. +/// Names the typed non-finite failure of one Newton stage. const fn non_finite(stage: NewtonStage) -> SolverFailure { SolverFailure::NonFiniteNewton { stage } } @@ -111,7 +111,7 @@ type RowFactor = [f64; 3]; /// The coefficient rows of one structured solver vector. type CoefficientRows = [BoxedDVecN; CONTRAST_ROWS]; -/// Zeroed coefficient rows. +/// Returns zeroed coefficient rows. fn zero_rows() -> CoefficientRows { core::array::from_fn(|_index| BoxedDVecN::zero()) } @@ -119,8 +119,8 @@ fn zero_rows() -> CoefficientRows { /// Factors a `2×2` PSD block in packed lower-triangle order, zeroing a rank-dropped column. /// /// A non-positive leading entry zeroes the first column - the fate of a saturated row whose -/// probabilities sit at a vertex - and a non-positive trailing pivot zeroes the second, so every -/// PSD block factors and `L·Lᵀ` reproduces the block exactly on its numerical rank. +/// probabilities lie at a vertex - and a non-positive trailing pivot zeroes the second. Every PSD +/// block therefore factors, and `L·Lᵀ` reproduces the block exactly on its numerical rank. pub(super) fn factor_block(c11: f64, c21: f64, c22: f64) -> RowFactor { if c11 <= 0.0 { let l22 = if c22 > 0.0 { c22.sqrt() } else { 0.0 }; @@ -134,7 +134,7 @@ pub(super) fn factor_block(c11: f64, c21: f64, c22: f64) -> RowFactor { [l11, l21, l22] } -/// The dot of packed factor columns `L̃ᵢ[:,k]·L̃ⱼ[:,l]`. +/// Computes the dot of packed factor columns `L̃ᵢ[:,k]·L̃ⱼ[:,l]`. #[expect( clippy::min_ident_chars, reason = "k and l are the factor-column indices of the written algebra" @@ -228,7 +228,7 @@ pub(super) fn newton_step( factors.push(factor_block(scaled[0], scaled[1], scaled[2])); } - // Capacitance C = I + ŨᵀŨ, lower triangle only; block (i, j) reads Kᵢⱼ once per entry. + // Capacitance C = I + ŨᵀŨ, lower triangle only. Block (i, j) reads Kᵢⱼ once per entry. let order = 2 * rows; let mut capacitance = DSquareMatrix::zeroed(order); for i in 0..rows { @@ -258,7 +258,7 @@ pub(super) fn newton_step( control.counters.record_factorization(); let factor = capacitance.cholesky().map_err(|error| match error { - // A non-finite pivot is the fate of a non-finite assembled entry; a finite non-positive + // A non-finite pivot is the fate of a non-finite assembled entry. A finite non-positive // pivot rejects the factorization itself. DCholeskyError::NonFinitePivot { .. } => non_finite(NewtonStage::Capacitance), DCholeskyError::NonPositivePivot { .. } => non_finite(NewtonStage::Factor), @@ -434,7 +434,7 @@ pub(super) fn newton_step( return steepest_crossing(gradient, &hessian_gradient, control.radius); } - // The dogleg leg from the interior Cauchy point toward the Newton point; the Newton + // The dogleg leg from the interior Cauchy point toward the Newton point. The Newton // product prices here and carries the recorded residual. let hessian_newton = priced_product(problem, point, &step, control, NewtonStage::NewtonPoint)?; let residual = newton_residual(&hessian_newton, gradient); @@ -463,10 +463,12 @@ pub(super) fn newton_step( }) } -/// The structured coefficient dot `Σ_slot left[slot]·right[slot]`, folded in contrast order. +/// Computes the structured coefficient dot, folded in contrast order. +/// +/// The dot is `Σ_slot left[slot]·right[slot]`. /// -/// The fold is data-dependent with no finiteness theorem, so the sum rides as an unclaimed -/// derivation to each consumer's own finish. +/// The fold is data-dependent with no finiteness theorem, and the sum returns as an unclaimed +/// derivation for each consumer's own finish. fn pair_dot(left: &CoefficientRows, right: &CoefficientRows) -> Derivation { let mut sum = Derivation::ZERO; @@ -495,7 +497,8 @@ fn solve_intercepts( let rhs = rhs.map(Derivation::into_raw); let l11 = s11.sqrt(); - // With finite entries the second pivot is finite or -∞, so the ordering test is total. + // With finite entries the second pivot is finite or -∞, and the ordering test is therefore + // total. let l21 = s21 / l11; let pivot = l21.mul_add(-l21, s22); if pivot <= 0.0 { @@ -511,7 +514,12 @@ fn solve_intercepts( Some([s0, s1]) } -/// One oracle Hessian-vector product, counted and checked finite. +/// Prices one oracle Hessian-vector product, counted and checked finite. +/// +/// # Errors +/// +/// Returns [`SolverFailure::NonFiniteNewton`] at `stage` when the oracle refuses the request (a +/// non-finite point or direction) or when the product it returns is not finite. fn priced_product( problem: &ScaledProblem<'_>, point: &ContrastVector, @@ -526,7 +534,7 @@ fn priced_product( .ok_or(SolverFailure::NonFiniteNewton { stage }) } -/// The relative Newton residual `‖Hζ·p_N + gζ‖/‖gζ‖` of a priced Newton product. +/// Computes the relative Newton residual `‖Hζ·p_N + gζ‖/‖gζ‖` of a priced Newton product. fn newton_residual( hessian_newton: &AlignedDVecN, gradient: &AlignedDVecN, @@ -537,7 +545,7 @@ fn newton_residual( defect_norm.checked_div(gradient_norm) } -/// The steepest-descent crossing onto the trust boundary, from the origin along `−g`. +/// Constructs the steepest-descent crossing onto the trust boundary, from the origin along `−g`. fn steepest_crossing( gradient: &AlignedDVecN, hessian_gradient: &AlignedDVecN, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/prepare.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/prepare.rs index b26d117d5a9..52ceca2ccbc 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/prepare.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/prepare.rs @@ -17,8 +17,8 @@ //! //! Every fit starts at physical `T₀ = 0`. At zero the normalized initial Hessian diagonal for //! coefficient coordinate `j` is `h_jj = (1/(3S))·Σ_i w_i x̄_ij² + λ/S`, identical for both contrast -//! rows. The intercept moment is `S` itself, so the intercept curvature is exactly `⅓`, and -//! preparation reads it from [`INTERCEPT_CURVATURE`]. +//! rows. The intercept moment is `S` itself, and the intercept curvature is exactly `⅓`. +//! Preparation reads it from [`INTERCEPT_CURVATURE`]. //! //! Each coordinate then takes the scale `D_j = √(max(h_jj, floor))` with `floor = //! curvature_relative_floor · max_k h_kk`. The floor follows the corpus's measured curvature scale @@ -66,8 +66,9 @@ pub(crate) enum PreparationError { InvalidTotalWeight { value: f64 }, /// A class carries no positive aggregate mass. MissingClassMass { class: GeometryClass }, - /// An initial-scaling curvature is not finite, or a constructed scale is not finite and - /// positive. + /// An initial-scaling curvature or a constructed scale left its domain. + /// + /// The curvature is not finite, or the scale is not finite and positive. InvalidScaling { coordinate: usize, value: f64 }, } @@ -281,7 +282,7 @@ fn validate_row( /// The intercept coordinate's normalized initial curvature. /// -/// The intercept moment is the total weight itself, so its normalized curvature is `S/(3S) = ⅓` +/// The intercept moment is the total weight itself, and its normalized curvature is `S/(3S) = ⅓` /// for every corpus. Written as the constant, it survives a `3·S` overflow that would flush the /// computed quotient to zero, and it anchors the curvature maximum - and with it the derived /// floor - strictly above zero. @@ -301,7 +302,7 @@ struct InitialScaling { /// /// Only the coefficient coordinates carry the `λ/S` regularization term. The intercept curvature /// is [`INTERCEPT_CURVATURE`]. The floor is the largest curvature scaled by the configured -/// relative floor, so it follows the corpus's curvature scale. +/// relative floor, and it follows the corpus's curvature scale. fn initial_scaling( moments: &AlignedDVecN, total_weight: DPositive, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/problem.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/problem.rs index 93401d4e406..11893163d8a 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/problem.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/problem.rs @@ -4,7 +4,7 @@ //! and the corpus's [`GramView`], and exposes the physical evaluations transformed into the //! solver's scaled coordinates. With the preparation diagonal `D`, points map as `θ(ζ) = D⁻¹ζ`, //! gradients as `gζ = D⁻¹gθ`, and Hessian-vector products as `Hζ[v] = D⁻¹·Hθ[D⁻¹v]`. Physical -//! evaluation receives `θ(ζ)` only at this boundary, so every quantity the loop compares or +//! evaluation receives `θ(ζ)` only at this boundary, and every quantity the loop compares or //! accumulates lives in one coordinate system. //! //! The underlying evaluations charge all the work themselves, and the coordinate transformations @@ -28,7 +28,7 @@ pub(crate) struct ScaledProblem<'corpus> { } impl ScaledProblem<'_> { - /// The physical contrast point `θ(ζ) = D⁻¹ζ`. + /// Returns the physical contrast point `θ(ζ) = D⁻¹ζ`. pub(crate) fn point(&self, zeta: &AlignedDVecN) -> ContrastVector { ContrastVector::from_flat(&self.prepared.scaling.divide(zeta)) } diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/receipt.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/receipt.rs index 590f51d2369..5f6c66e8403 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/receipt.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/receipt.rs @@ -29,8 +29,9 @@ use crate::{ pub(crate) enum ReceiptDetail { /// No stored receipts: the routine posture. None, - /// One receipt per started outer iteration, start-state digests included: a debugging - /// consumer's request. + /// One receipt per started outer iteration, start-state digests included. + /// + /// A debugging consumer's request. Digests, } @@ -51,8 +52,8 @@ pub(crate) struct OuterReceipt { /// Trust radius entering the iteration. pub radius: DPositive, /// Accepted objective entering the iteration. - // Raw on purpose: the first outer iteration after an admitted non-finite origin objective - // records the admission honestly. + // Raw on purpose: the first outer iteration can retain an admitted non-finite origin + // objective, which a finite domain type could not record. pub objective: f64, /// Accepted scaled-gradient norm entering the iteration. pub gradient_norm: DNonNegative, @@ -108,8 +109,9 @@ pub(crate) enum CurvatureDiagnostic { pub(crate) struct OuterOutcome { /// The inner Newton outcome tag. pub tag: Option, - /// The relative Newton residual `‖Hζ·p_N + gζ‖/‖gζ‖` of the priced Newton point: the per-outer - /// certificate of the factorization against the oracle. + /// The relative Newton residual of the priced Newton point. + /// + /// `‖Hζ·p_N + gζ‖/‖gζ‖` is the per-outer certificate of the factorization against the oracle. pub newton_residual: Option, /// Norm of the returned step `‖p‖`. pub step_norm: Option, @@ -133,7 +135,7 @@ pub(crate) struct OuterOutcome { /// The exposed coordinate/version identity of one run's receipts and digests. /// -/// Archived digest bytes are not self-describing, so this value names the digest version and the +/// Archived digest bytes are not self-describing, and this value names the digest version and the /// coordinate system that generated them. The domain tag and declared dimension are byte-identical /// to the prefix of every digest preimage. This identity exposes the coordinate system on its own, /// and the coordinate system does not enter the preimage. @@ -156,8 +158,10 @@ impl ReceiptCoordinates { }; } -/// The digest of a flat solver vector: the domain tag's UTF-8 bytes, the declared dimension, then -/// every component in vector order, both as native in-memory bytes. +/// Computes the digest of a flat solver vector. +/// +/// The preimage is the domain tag's UTF-8 bytes, the declared dimension, then every component in +/// vector order, both as native in-memory bytes. /// /// Digest identity is environment-scoped. A replay in the same environment reproduces the bytes /// bit-for-bit, and byte identity does not extend across builds or architectures. diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/mod.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/mod.rs index 5a37dffa3f3..7c9103babf4 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/mod.rs @@ -1,6 +1,6 @@ -//! Diagnosis instruments for the solver over frozen classifier corpora. +//! Diagnostic probes for the solver over frozen classifier corpora. //! -//! Nothing here is pipeline machinery: a report observes a solve, it never participates in one. +//! A report observes a solve and never participates in one. mod probe; diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/probe.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/probe.rs index a1c522821b1..d1bf93aef89 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/probe.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/probe.rs @@ -5,7 +5,7 @@ //! a supplied directory, re-runs the bounded solver over one fold subset, and dumps every receipt. //! The terminal is the observation, and each receipt reports the outer's Newton residual, the //! per-outer certificate of the factorization against the oracle. The caller owns the -//! fold-assignment seed and the regularization strength, so the probe accepts any assignment and +//! fold-assignment seed and the regularization strength, and the probe accepts any assignment and //! any candidate strength, the production CV candidates included. //! //! The per-row curvature-scale census prints at the origin and at the final accepted point, and a @@ -48,6 +48,7 @@ pub(crate) enum ProbeCorpus<'caller> { generation: GenerationId, }, /// Supplied artifact files under their staged names, under the compiled deployment defaults. + /// /// The corpus of a fit that never published probes through this form. Supplied { /// The directory holding the three annotation artifacts. @@ -117,7 +118,7 @@ pub(crate) async fn probe_fold( let folds = grouped_folds(rows, config.folds, settings.seed).expect("the corpus has enough groups"); // The gather re-bases the fold complement into the solo solve's own - // positional row space, mirroring the production fold gather; the + // positional row space, mirroring the production fold gather. The // card-row domain ends here. let members: Vec = folds .iter_enumerated() @@ -141,7 +142,7 @@ pub(crate) async fn probe_fold( &mut counters, ) .expect("the fold corpus prepares"); - // A solo solve assembles its own Gram over the fold subset; the entries equal the + // A solo solve assembles its own Gram over the fold subset. The entries equal the // production fold view's bit for bit, one independent dot per pair either way. let gram = Gram::assemble(fold_embeddings.rows(), &mut counters); // The replay of a census outer re-enters the solve with these exact charges. @@ -245,13 +246,14 @@ pub(crate) async fn probe_fold( ); } -/// Replays the production outer trajectory to the start of `target`, certifying every replayed -/// outer against its production receipt. +/// Replays the production outer trajectory to the start of `target`. +/// +/// The replay certifies every replayed outer against its production receipt. /// /// Returns the accepted state entering `target` and its trust radius. The replay re-runs the -/// production functions over the same problem from the same charged counters, so equality of radii, -/// objectives, counters, and start-state digests at every outer proves the replayed trajectory is -/// the production trajectory. +/// production functions over the same problem from the same charged counters. Equality of radii, +/// objectives, counters, and start-state digests at every outer then identifies the replayed +/// trajectory with the production trajectory, up to the digest's collision resistance. /// /// # Panics /// @@ -363,6 +365,10 @@ fn replay_to_outer( } /// Prints the cumulative decade census of one curvature-scale reading. +/// +/// # Panics +/// +/// This panics on a census without readings: the median indexes the sorted scales. #[expect( clippy::print_stdout, reason = "the probe's receipt dump is its whole output" diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/tests.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/tests.rs index 7b1c6184275..613df9d427c 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/report/tests.rs @@ -1,4 +1,4 @@ -//! Certificates of the report instruments: the probe's curvature census. +//! Certificates of the report probes: the probe's curvature census. use super::super::{ SOLVER_DIMENSIONS, @@ -45,8 +45,14 @@ fn fixture_corpus() -> (MatrixN, Vec) { (embeddings, training) } -/// At the physical origin every row's probabilities are uniform, so the curvature scale is exactly -/// `(1/3)·(2/3)` for every row. +/// Reads every origin row's curvature scale as the `f64` reference `p·(1 − p)`, bit for bit. +/// +/// At the physical origin every row's probabilities are uniform, and the curvature scale +/// `max_c p_c(1 − p_c)` is `(1/3)·(2/3) = 2/9` in real arithmetic. The census computes it as the +/// `f64` expression `p·(1 − p)` with `p = 1/3`, the rounded expression the test builds as its +/// reference, and the two agree bit for bit. That reference differs from the `f64` nearest to +/// `2/9` by one unit in the last place, and the assertion pins the computed expression rather +/// than the exactly rounded fraction. #[test] fn curvature_scales_at_the_origin_are_uniform() { let (embeddings, training) = fixture_corpus(); @@ -82,8 +88,10 @@ fn curvature_scales_at_the_origin_are_uniform() { } } -/// The census pairs each row's reading with that row's own weight in original order and carries -/// the preparation-validated total, so a weight share divides by the exact row-order sum. +/// Pairs each census reading with its row's own weight and the preparation-validated total. +/// +/// The census pairs each row's reading with that row's own weight in original order and carries the +/// preparation-validated total: a weight share divides by the exact row-order sum. #[test] fn census_readings_carry_row_weights_and_the_validated_total() { let (embeddings, _) = fixture_corpus(); diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/resolution.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/resolution.rs index adf9f5b0598..93ee871d1e7 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/resolution.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/resolution.rs @@ -1,19 +1,19 @@ //! Objective resolution: the smallest decrease distinguishable from rounding noise. //! //! Comparing objective values cannot resolve a predicted reduction no larger than the objective's -//! own representational spacing, so the solver treats such a reduction as a stall rather than -//! progress. [`objective_resolution`] returns that threshold, `R(F̄) = ulps · ulp(|F̄|)`, where `ulp` -//! is the spacing of the f64 grid at the objective's magnitude. Below the maximum finite value that -//! spacing is the next-up spacing. At the maximum it becomes the predecessor spacing, because the -//! next-up spacing would be infinite. Zero and subnormal magnitudes take the minimum positive -//! subnormal, the grid's smallest step. A finite negative objective remains valid evidence and -//! contributes only its magnitude to the spacing. +//! own representational spacing. The solver therefore treats such a reduction as a stall rather +//! than progress. [`objective_resolution`] returns that threshold, `R(F̄) = ulps · ulp(|F̄|)`, where +//! `ulp` is the spacing of the f64 grid at the objective's magnitude. Below the maximum finite +//! value that spacing is the next-up spacing. At the maximum it becomes the predecessor spacing, +//! because the next-up spacing would be infinite. Zero and subnormal magnitudes take the minimum +//! positive subnormal, the grid's smallest step. A finite negative objective remains valid evidence +//! and contributes only its magnitude to the spacing. use core::num::NonZero; use crate::math::{DFinite, DNonNegative, DPositive}; -/// The f64 grid spacing at a finite non-negative magnitude. +/// Returns the f64 grid spacing at a finite non-negative magnitude. fn ulp(magnitude: DNonNegative) -> f64 { if magnitude.is_zero() || magnitude.is_subnormal() { // The smallest positive subnormal is the spacing of the grid around zero. @@ -21,7 +21,7 @@ fn ulp(magnitude: DNonNegative) -> f64 { } if magnitude == f64::MAX { - // The next-up spacing is infinite at the top of the grid; the predecessor spacing is + // The next-up spacing is infinite at the top of the grid. The predecessor spacing is // the honest step size there. return f64::MAX - f64::MAX.next_down(); } @@ -29,7 +29,7 @@ fn ulp(magnitude: DNonNegative) -> f64 { magnitude.get().next_up() - magnitude.get() } -/// The resolution `R(F̄) = ulps · ulp(|F̄|)` of a finite objective value. +/// Computes the resolution `R(F̄) = ulps · ulp(|F̄|)` of a finite objective value. /// /// This checks the multiplication and accepts only a finite, positive result. Returns [`None`] for /// a non-finite objective or an invalid resolution, and the caller maps [`None`] onto its typed diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/scale.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/scale.rs index ee24eefb9b4..a10f68e7755 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/scale.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/scale.rs @@ -19,8 +19,9 @@ pub(crate) struct Scaling { } impl Scaling { - /// Expands per-augmented-coordinate scales into the flat solver layout, one copy per contrast - /// row. + /// Expands per-augmented-coordinate scales into the flat solver layout. + /// + /// The layout copies each scale once per contrast row. pub(super) fn from_augmented(scales: &AlignedDVecN) -> Self { let mut diagonal = BoxedDVecN::zero(); for row in 0..CONTRAST_ROWS { @@ -54,7 +55,7 @@ impl Scaling { product } - /// The diagonal in flat contrast-major layout. + /// Returns the diagonal in flat contrast-major layout. #[inline] #[cfg(test)] // The solver tests compare the assembled diagonal against hand values. pub(super) const fn diagonal(&self) -> &AlignedDVecN { diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/solve.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/solve.rs index f0acf72c80d..ac50cbfafee 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/solve.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/solve.rs @@ -6,7 +6,7 @@ //! ratio. The accepted point moves only on acceptance. Rejection shrinks the trust radius toward //! its minimum and an expanded radius requires a validated boundary step. Success is [`Converged`] //! (a fresh final joint evaluation re-proving the certificate) and every other terminal is a -//! typed [`SolverFailure`] in the normative precedence order: validation, accepted-gradient +//! typed [`SolverFailure`] in this precedence order: validation, accepted-gradient //! success, outer budget, inner Newton, invalid predicted reduction, resolution construction, //! resolution stall, candidate numerical failure, ratio classification, then radius underflow. @@ -35,8 +35,7 @@ pub(crate) struct AcceptedPoint { pub zeta: BoxedDVecN, /// The normalized objective at the point. // Raw on purpose: initialization admits a non-finite origin objective - the certificate - // tests only the gradient - and resolution and final certification refuse it by name where - // the design says so. + // tests only the gradient - and resolution and final certification refuse it by name. pub objective: f64, /// The scaled gradient at the point. pub scaled_gradient: BoxedDVecN, @@ -59,7 +58,7 @@ pub(crate) struct SolverControl { } impl SolverControl { - /// Fresh control state carrying the preparation-charged counters. + /// Creates fresh control state carrying the preparation-charged counters. const fn new(radius: DPositive, counters: WorkCounters) -> Self { Self { radius, @@ -86,8 +85,9 @@ pub(crate) struct Converged { pub point: AcceptedPoint, } -/// Everything one solve reports: the terminal, the last accepted state, control, evidence, and -/// receipts. +/// Everything one solve reports. +/// +/// The terminal, the last accepted state, control, evidence, and receipts. #[derive(Debug)] pub(crate) struct SolverRun { /// The certified solution or the typed failure. @@ -106,11 +106,13 @@ pub(crate) struct SolverRun { ) )] pub certificate: Option, - /// One receipt per started outer iteration under a debugging request. A routine fit leaves - /// this empty. + /// One receipt per started outer iteration under a debugging request. + /// + /// A routine fit leaves this empty. pub receipts: Vec, - /// The coordinate/version identity of the receipts and their digests, present only under a - /// debugging request, with the receipts it describes. + /// The coordinate/version identity of the receipts and their digests. + /// + /// Present only under a debugging request, with the receipts it describes. pub coordinates: Option, } @@ -243,7 +245,7 @@ fn run( continue; } - // Acceptance commits only after the candidate gradient proves finite; a rejected + // Acceptance commits only after the candidate gradient proves finite. A rejected // request and a non-finite gradient share the terminal. let Some(trial_gradient) = problem .gradient(&trial_point, &mut control.counters) @@ -309,7 +311,7 @@ const fn gradient_threshold( /// Stores the started outer iteration's receipt and returns the outcome it records into. /// /// A debugging request stores one receipt per started outer iteration and the returned outcome -/// collects the iteration's diagnostics. The routine posture stores none and returns [`None`], so +/// collects the iteration's diagnostics. The routine posture stores none and returns [`None`], and /// the diagnostic-only arithmetic never runs. fn start_receipt<'receipts>( receipts: &'receipts mut Vec, @@ -348,11 +350,13 @@ pub(super) const fn derive_certificate( } } -/// Returns the predicted model reduction `−g·p − ½·p·Hp` from the returned step and product alone, -/// recording the inner-step summaries when the solve stores a receipt. +/// Returns the predicted model reduction from the returned step and product alone. +/// +/// The reduction is `−g·p − ½·p·Hp`. When the solve stores a receipt, it records the inner-step +/// summaries. /// /// The dots are algorithm inputs and always compute. The norms are diagnostic-only and compute -/// solely for a stored receipt. The reduction rides as an unclaimed derivation, and the caller's +/// solely for a stored receipt. The reduction returns as an unclaimed derivation, and the caller's /// finish refuses a non-finite value. fn record_inner_step( recorded: Option<&mut OuterOutcome>, @@ -433,7 +437,10 @@ pub(super) fn certify( }) } -/// The accepted-step curvature diagnostic `p·y` and `(p·y) / (p·p)` with `y = g_trial − g`. +/// Computes the accepted step's curvature diagnostic. +/// +/// The readings are `p·y` and `(p·y) / (p·p)` with `y = g_trial − g`, or the first non-finite +/// intermediate that prevented them. fn curvature_diagnostic( step: &AlignedDVecN, trial_gradient: &AlignedDVecN, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/target.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/target.rs index 06a5b711840..9426d62d0bc 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/target.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/target.rs @@ -9,7 +9,7 @@ //! by approximation. A common shift of the class logits provably moves no loss. //! //! Construction reports the raw sum and the largest normalization adjustment `max_c |u_c − t_c/s|` -//! alongside the target, so preparation can aggregate the raw sum range and `maximum_adjustment` as +//! alongside the target, and preparation aggregates the raw sum range and `maximum_adjustment` as //! evidence of how much canonicalization actually moved the data. use core::num::NonZeroU32; @@ -38,7 +38,7 @@ pub(super) struct Canonicalization { /// A soft target over the geometry classes with an exact unit sum. /// /// Stores the leading normalized components and derives the reference component `u_ref = 1 − Σ_c -/// u_c` on demand, so it can never disagree with the stored components. +/// u_c` on demand, and it can never disagree with the stored components. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct ClosedTarget { /// The stored leading components, one per class ahead of the reference. diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/terminal.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/terminal.rs index 17eefd9b80c..6b0c33c86fe 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/terminal.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/terminal.rs @@ -45,8 +45,9 @@ pub(crate) enum SolverFailure { /// The solve stage that produced the non-finite value. stage: NewtonStage, }, - /// The intercept Schur system is not positive-definite: no row carries interior probabilities, - /// so the corpus offers the intercepts no curvature. + /// The intercept Schur system is not positive-definite. + /// + /// No row carries interior probabilities, and the corpus offers the intercepts no curvature. SingularInterceptCurvature, /// The boundary search found no finite positive crossing. NoFiniteBoundaryStep, diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/tests.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/tests.rs index 5d330ce6cff..fcebfb75759 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/tests.rs @@ -56,6 +56,7 @@ fn flat(assignments: &[(usize, f64)]) -> BoxedDVecN { vector } +/// The most rows a solver fixture corpus holds. const CAPACITY: usize = 4; /// An owned solver-test corpus growing row by row. @@ -65,6 +66,7 @@ struct Corpus { } impl Corpus { + /// An empty corpus with zeroed storage. fn new() -> Self { Self { storage: BoxedVecN::zero(), @@ -72,6 +74,14 @@ impl Corpus { } } + /// Appends a row with the given leading embedding components, soft target and weight. + /// + /// The embedding starts with `leading` and is zero beyond it, and the row joins the shared + /// fixture group. + /// + /// # Panics + /// + /// Panics when the corpus already holds [`CAPACITY`] rows. fn push(&mut self, leading: &[f32], target: [f64; GeometryClass::COUNT], weight: f64) { let index = self.rows.len(); assert!(index < CAPACITY, "the fixture fits the capacity"); @@ -84,18 +94,23 @@ impl Corpus { }); } + /// The pushed rows' embeddings as an aligned slice. fn embeddings(&self) -> &[AlignedVecN] { AlignedVecN::from_slice(&self.storage.as_array()[..self.rows.len() * CANONICAL_DIMENSIONS]) .expect("boxed storage is aligned") } } +/// The SHA-256 digest of `bytes`, used as a group identity. fn digest(bytes: &[u8]) -> Sha256Digest { let mut hasher = Sha256::new(); hasher.update(bytes); hasher.finalize() } +/// Preparation settings for the fixtures. +/// +/// Regularisation `0.5`, a one-ulp target-sum tolerance and a `1e-12` relative curvature floor. fn settings() -> PreparationSettings { PreparationSettings { regularization: d_positive!(0.5), @@ -104,8 +119,9 @@ fn settings() -> PreparationSettings { } } -/// Builds the valid three-row corpus with exact targets, weights summing to five, and known leading -/// components. +/// Builds the valid three-row fixture corpus. +/// +/// Exact targets, weights summing to five, and known leading components. fn valid_corpus() -> Corpus { let mut corpus = Corpus::new(); corpus.push(&[2.0, -1.0], [1.0, 0.0, 0.0], 1.5); @@ -125,7 +141,7 @@ fn solver_parameters() -> ContrastVector { parameters } -/// Reconstructs raw parameters `W = BA`, `b = Ba` in the legacy flat layout. +/// Reconstructs raw parameters `W = BA`, `b = Ba` in the objective's flat class layout. fn raw_parameters(contrast: &ContrastVector) -> Parameters { let mut raw = Parameters::zero(); let (rows, intercepts) = raw.as_array_mut().as_chunks_mut::(); @@ -169,8 +185,9 @@ fn flat_dot(left: &ContrastVector, right: &ContrastVector) -> f64 { ) } -/// Builds a coefficient axis, an intercept axis, and a mixed vector as the deterministic test -/// directions. +/// Builds the three deterministic test directions. +/// +/// A coefficient axis, an intercept axis, and a mixed vector. fn directions() -> [ContrastVector; 3] { let mut coefficient = ContrastVector::zero(); coefficient.coefficients[0].as_array_mut()[0] = 1.0; @@ -242,8 +259,8 @@ fn basis_roundtrip_is_the_identity_up_to_rounding() { let logits = basis::expand(contrast); let recovered = basis::reduce(logits); - // Rounding in the expanded logits is proportional to the largest coordinate, so the - // roundtrip error of every component carries that scale after cancellation. + // Rounding in the expanded logits is proportional to the largest coordinate, and the + // roundtrip error of every component therefore carries that scale after cancellation. let magnitude = contrast[0].abs().max(contrast[1].abs()).max(1.0); for (out, initial) in recovered.iter().zip(contrast) { assert!( @@ -265,10 +282,10 @@ fn expanded_logits_stay_shift_free() { } } -/// [`AlignedDVecN::checked_dot`] passes an exactly representable dot through the gate. +/// [`AlignedDVecN::checked_dot`] passes an exactly representable dot through the check. #[test] fn checked_dot_passes_finite_values() { - // Both terms are exact, so 2·4 + (−3)·5 = −7 under any fold shape. + // Both terms are exact: 2·4 + (−3)·5 = −7 under any fold shape. let left = flat(&[(0, 2.0), (9, -3.0)]); let right = flat(&[(0, 4.0), (9, 5.0)]); @@ -293,8 +310,10 @@ fn checked_dot_rejects_non_finite_results() { assert_eq!(poisoned.checked_dot(&ones), None); } -/// [`AlignedDVecN::checked_stable_l2`] passes exact norms through and reports non-finite -/// components as [`None`]. +/// Passes exact norms through `checked_stable_l2` and refuses non-finite components. +/// +/// [`AlignedDVecN::checked_stable_l2`] passes exact norms through and reports non-finite components +/// as [`None`]. #[test] fn checked_stable_l2_gates_the_house_norm() { // 3-4-5 triangle: every ratio and square of the house kernel is exact. @@ -399,6 +418,8 @@ fn preparation_accumulates_statistics_and_charges_work() { assert_eq!(prepared.evidence.maximum_adjustment, 0.0); } +/// Derives the initial diagonal `h_jj` identically for both contrast rows. +/// /// The initial diagonal follows `h_jj = (1/(3S))·Σ w x̄² + (λ/S)·1{coefficient}` with the derived /// floor inside the square root, identically for both contrast rows. #[test] @@ -632,7 +653,7 @@ fn contrast_vector_roundtrips_the_flat_layout() { /// The derived reference component reports the canonicalization distance. #[test] fn closed_target_records_the_derived_adjustment() { - // The raw sum is one ulp above 1, so normalization moves the components. + // The raw sum is one ulp above 1, and normalization moves the components. let raw = [0.5, 0.25, 0.25 + f64::EPSILON]; let (closed, evidence) = ClosedTarget::new(raw, one_ulp()).expect("within tolerance"); @@ -661,8 +682,9 @@ fn objective_at_zero_is_ln_three() { ); } -/// The raw-space objective over the full corpus: a per-row reference over [`objective::logits`] -/// sharing no code with the contrast evaluation. +/// Computes the raw-space objective over the full corpus. +/// +/// A per-row reference over [`objective::logits`] sharing no code with the contrast evaluation. fn raw_objective(corpus: &Corpus, raw: &Parameters, regularization: f64) -> f64 { let mut objective = 0.0; for (row, embedding) in corpus.rows.iter().zip(corpus.embeddings()) { @@ -1013,7 +1035,7 @@ fn objective_resolution_pins_the_ulp_exceptional_cases() { ); // The top of the grid uses the predecessor spacing and stays finite: the returned domain - // proves finiteness, so the expectation is the whole assertion. + // proves finiteness, and the expectation is the whole assertion. let top = objective_resolution(f64::MAX, one_ulp()).expect("finite at the maximum"); assert_eq!(top, f64::MAX - f64::MAX.next_down()); let widest = NonZeroU32::new(u32::MAX).expect("u32::MAX is nonzero"); @@ -1046,8 +1068,8 @@ fn solver_config() -> SolverConfig { /// The in-domain fixture and the cross-field boundaries validate. /// -/// Per-field domains hold by construction. Only the cross-field orderings remain for `validate` -/// to accept. +/// Per-field domains hold by construction. Only the cross-field orderings remain for the +/// constructor to accept. #[test] fn config_accepts_the_domain_boundaries() { solver_config() @@ -1098,6 +1120,8 @@ fn config_rejects_misordered_fields() { } } +/// Reads the gradient threshold as the stated maximum and refuses only a non-finite norm. +/// /// The gradient threshold is the stated maximum, zero is valid, and only a non-finite norm maps /// away. #[test] @@ -1164,7 +1188,7 @@ fn boundary_step_picks_the_far_crossing_for_a_backward_direction() { let crossed = boundary_step(&interior, &direction, &zero, &zero, d_positive!(2.0)) .expect("the backward crossing is exact"); - // From p = e₀ along −e₀ the boundary sits at −Δ·e₀, a crossing of τ = 3. + // From p = e₀ along −e₀ the boundary lies at −Δ·e₀, a crossing of τ = 3. assert_eq!(crossed.step.as_array()[0], -2.0); assert_eq!(crossed.hessian_step.as_array()[0], 0.0); } @@ -1192,7 +1216,7 @@ fn boundary_step_advances_the_hessian_product_along_the_crossing() { assert_eq!(crossed.hessian_step.as_array()[1], 0.5); } -/// An irrational crossing under an odd radius passes both norm gates of the gross-defect guard. +/// An irrational crossing under an odd radius passes both norm checks of the gross-defect guard. #[test] fn boundary_step_survives_an_irrational_crossing() { let interior = flat(&[(0, 0.75)]); @@ -1248,6 +1272,8 @@ fn boundary_step_rejects_a_zero_direction() { ); } +/// Rejects an overflowing radius normalization before any coefficient forms. +/// /// The boundary construction rejects an overflowing radius normalization before any coefficient /// forms. #[test] @@ -1287,13 +1313,13 @@ fn boundary_step_rejects_an_overflowing_hessian_extension() { fn boundary_revalidation_rejects_a_subnormal_rescaling() { let zero = flat(&[]); // 2⁻¹⁰⁴⁰: a subnormal radius whose rescaled step components quantize on the 2⁻¹⁰⁷⁴ grid, - // mangling the returned geometry by ~2⁻²⁸ relative - far outside the gross-defect guard. + // mangling the returned geometry by ~2⁻²⁸ relative, far outside the gross-defect guard. let radius = DPositive::new(f64::from_bits(1_u64 << 34)).expect("the subnormal is positive"); let mut direction = BoxedDVecN::::zero(); direction.as_array_mut().fill(f64::from(radius)); - // The same geometry at radius one passes both gates: the normalized crossing itself is - // sound, so a rejection below can only come from the rescaling revalidation. + // The same geometry at radius one passes both checks: the normalized crossing itself is + // sound, and a rejection below can only come from the rescaling revalidation. let mut unit_direction = BoxedDVecN::::zero(); unit_direction.as_array_mut().fill(1.0); boundary_step(&zero, &unit_direction, &zero, &zero, d_positive!(1.0)) @@ -1302,9 +1328,10 @@ fn boundary_revalidation_rejects_a_subnormal_rescaling() { assert_eq!(boundary_step(&zero, &direction, &zero, &zero, radius), None,); } -/// Returns a deterministic dense component from the Weyl sequence on the golden ratio at the given -/// phase, folded to `[-1, 1]`, with every third component thinned to vary magnitudes within the -/// fill. +/// Returns a deterministic dense component from the golden-ratio Weyl sequence. +/// +/// The component at the given phase folds to `[-1, 1]`, with every third component thinned to vary +/// magnitudes within the fill. #[expect( clippy::cast_precision_loss, reason = "solver indices stay far below 2^52" @@ -1321,8 +1348,9 @@ fn dense_component(index: usize, phase: f64) -> f64 { } } -/// A dense solver vector with two full-scale leading components over a `1e-3` tail, scaled to the -/// given Euclidean norm. +/// A dense solver vector scaled to the given Euclidean norm. +/// +/// Two full-scale leading components over a `1e-3` tail. /// /// The magnitude split concentrates the norm in two components while thousands of small squares /// absorb into the fold accumulators, the structure that drives the reductions' rounding. Uniform @@ -1362,11 +1390,13 @@ fn compensated_l2(values: &[f64]) -> f64 { (sum + compensation).sqrt() } -/// Replicates the boundary construction through the same primitives, returning the rescaled step -/// with the built and returned radius-normalized residuals in ulps of one. +/// Replicates the boundary construction through the same primitives. +/// +/// Returns the rescaled step with the built and returned radius-normalized residuals in ulps of +/// one. /// -/// The built residual is internal to [`boundary_step`]; a consumer ties the replica to the -/// implementation by asserting the returned step equals the admitted payload bit-for-bit, so the +/// The built residual is internal to [`boundary_step`]. A consumer ties the replica to the +/// implementation by asserting the returned step equals the admitted payload bit-for-bit: the /// replica cannot drift from the arithmetic it reports on. fn boundary_construction_replica( interior: &BoxedDVecN, @@ -1415,7 +1445,7 @@ fn boundary_construction_replica( /// A dense solver-dimension crossing constructs, and the replica ties it byte-for-byte. /// /// The crossing has `‖interior‖ = 0.9392·Δ`, `‖direction‖ = 3.404·Δ`, `interior·direction ≈ -/// −1.5624`, `Δ = 0.7`: a dense, cancellation-prone construction whose honest residuals sit orders +/// −1.5624`, `Δ = 0.7`: a dense, cancellation-prone construction whose honest residuals lie orders /// of magnitude inside the gross-defect guard. The byte-tie keeps the replica honest: its reported /// residuals describe exactly the arithmetic that produced the admitted step. #[test] @@ -1458,7 +1488,10 @@ fn boundary_step_admits_the_dense_crossing_and_ties_its_replica() { ); } -/// Honest crossings across the historical calibration grid's corners sit far inside the guard. +/// Keeps honest crossings at the grid corners far inside the guard. +/// +/// Honest crossings at the corners of the radius and interior-fraction grid lie far inside the +/// guard. #[test] fn honest_boundary_residuals_sit_inside_the_gross_defect_guard() { let guard_ulps = GROSS_DEFECT_GUARD / f64::EPSILON; @@ -1479,8 +1512,8 @@ fn honest_boundary_residuals_sit_inside_the_gross_defect_guard() { /// The guard rejects a collapsed discriminant: the double-root value solves no crossing. /// -/// A construction whose discriminant flushed to zero yields `τ = −b/(2a)`; on a real crossing that -/// value lands the step far off unit norm, and the guard names the defect. +/// A construction whose discriminant flushed to zero yields `τ = −b/(2a)`. On a real crossing that +/// value leaves the step far off unit norm, and the guard names the defect. #[test] fn the_gross_defect_guard_rejects_a_collapsed_discriminant() { // u = 0.6·e₀, v = e₀ + e₁ at Δ = 1 gives a = 2, b = 1.2, c = −0.64, honest τ = 0.34. @@ -1571,7 +1604,7 @@ fn solve_certifies_immediately_when_the_initial_gradient_passes() { ); assert!((converged.point.objective - 3.0_f64.ln()).abs() < 1.0e-12); - // No outer iteration started, so the receipt list stays empty and the inner counters stay at + // No outer iteration started: the receipt list stays empty and the inner counters stay at // zero. assert_eq!(run.control.outer_iterations_started, 0); assert!(run.receipts.is_empty()); @@ -1748,7 +1781,10 @@ fn solve_certificate_tie_returns_at_equality() { assert!(tie.receipts.is_empty()); } -/// A certificate out of reach exhausts the outer budget with one receipt per started iteration. +/// Exhausts the outer budget on an unmeetable certificate, one receipt per iteration. +/// +/// A certificate the tolerances cannot meet exhausts the outer budget with one receipt per started +/// iteration. #[test] fn solve_fails_the_outer_iteration_budget() { let corpus = valid_corpus(); @@ -1767,11 +1803,6 @@ fn solve_fails_the_outer_iteration_budget() { assert_eq!(run.receipts.len(), 1); } -/// A rejection at the minimum trust radius underflows. -/// -/// The fixture's first full Newton step lands where curvature has risen against the model: its -/// measured ratio is `0.99048`, so an acceptance threshold of `0.995` rejects it deterministically, -/// and the rejection at the minimum radius reaches the terminal that no budget precedes any more. #[test] fn solve_underflows_the_radius_on_a_rejection_at_the_minimum() { let corpus = valid_corpus(); @@ -1786,7 +1817,8 @@ fn solve_underflows_the_radius_on_a_rejection_at_the_minimum() { ..solver_config() }; - // The rejection happens at the minimum radius, so the radius test fires. + // the strict acceptance threshold 0.995 rejects this candidate. With the initial radius + // already at the minimum, rejection returns `RadiusUnderflow`. let radius_starved = run_solver(&corpus, strict); assert_matches!(radius_starved.outcome, Err(SolverFailure::RadiusUnderflow)); assert_eq!( @@ -1837,12 +1869,12 @@ fn solve_expands_the_radius_on_an_expanded_boundary_step() { ); // The `1e-3` radius forces a boundary step whose small-step ratio expands the radius once, the - // certificate stays out of reach, and the outer budget then ends the run. + // certificate stays unmet, and the outer budget then ends the run. assert_matches!(run.outcome, Err(SolverFailure::OuterIterationBudget)); assert_eq!(run.control.counters.candidate_acceptances, 1); assert_eq!(run.control.radius, 2.0e-3); assert_eq!(run.receipts[0].radius, 1.0e-3); - // The Newton point sits far outside that radius, so the crossing is a boundary tag. + // The Newton point lies far outside that radius: the crossing is a boundary tag. assert_matches!( run.receipts[0].outcome.tag, Some(NewtonTag::CauchyBoundary | NewtonTag::DoglegBoundary), @@ -1852,7 +1884,7 @@ fn solve_expands_the_radius_on_an_expanded_boundary_step() { /// A valid degenerate corpus drives the full machine into the typed non-finite Newton terminal. /// /// Weights of `f64::MAX / 4` keep `S` and every preparation aggregate finite, and the initial -/// scaled gradient stays finite yet fails its relative certificate, so an inner solve must run. The +/// scaled gradient stays finite yet fails its relative certificate: an inner solve must run. The /// per-row factor scale `wᵢ/λ` then overflows against the subnormal regularization, the weighted /// curvature block leaves the finite domain, and the machine reaches `NonFiniteNewton { Weights }` /// with the curvature traversal already charged. @@ -1901,7 +1933,7 @@ fn solve_reaches_the_non_finite_newton_terminal_on_a_degenerate_scale() { /// Zero embeddings and one-hot targets keep every prepared datum and the scaled gradient finite /// (the class residuals cancel to rounding residue, well inside the absolute tolerance), but the /// weights push the accumulated origin data loss past the finite range. Initialization admits the -/// infinite objective - the certificate tests only the gradient - and the reserved final evaluation +/// infinite objective (the certificate tests only the gradient), and the reserved final evaluation /// then fails `FinalCertificationNonFinite` by name. #[test] fn solve_fails_final_certification_on_a_non_finite_admitted_objective() { @@ -1937,8 +1969,9 @@ fn solve_fails_final_certification_on_a_non_finite_admitted_objective() { assert_eq!(run.control.counters.joint_passes, 2); } -/// The exposed domain tag and dimension are the exact digest-preimage prefix. The coordinate -/// system rides only the exposed identity. +/// The exposed domain tag and dimension are the exact digest-preimage prefix. +/// +/// The coordinate system enters only the exposed identity. #[test] #[expect( clippy::host_endian_bytes, @@ -1976,8 +2009,9 @@ fn receipt_domain_tag_and_dimension_are_the_exact_digest_prefix() { assert_eq!(hasher.finalize(), vector_digest(&vector)); } -/// The certificate pins the initial norm and its derived threshold. The norm's domain makes the -/// derivation total. +/// The certificate pins the initial norm and its derived threshold. +/// +/// The norm's domain makes the derivation total. #[test] fn derive_certificate_pins_the_threshold_formula() { let config = solver_config(); @@ -2119,6 +2153,8 @@ fn factor_block_reproduces_psd_blocks_and_drops_rank() { assert_eq!(factor_block(0.0, 0.0, -1.0e-17), [0.0, 0.0, 0.0]); } +/// Reads Gram entries as symmetric exact-product dots, bit for bit through the fold view. +/// /// Gram entries are the exact-product dots, symmetric, and a fold view reads the full matrix bit /// for bit as a direct dot over the member embeddings. #[test] @@ -2158,7 +2194,7 @@ fn gram_views_read_the_assembled_dots_bit_for_bit() { /// The curvature pass's intercept columns are the oracle's Hessian columns. /// -/// The pass accumulates `H[0|e_k]` through the moment identity `C = q − mmᵀ`; the oracle evaluates +/// The pass accumulates `H[0|e_k]` through the moment identity `C = q − mmᵀ`. The oracle evaluates /// the same columns through its shifted-probability path. Agreement at a generic point ties the /// Newton assembly to the finite-difference-certified oracle at the block level, with rounding as /// the only separation. @@ -2200,9 +2236,10 @@ fn curvature_pass_matches_the_oracle_intercept_columns() { } } -/// Prepares the corpus and drives one inner Newton solve at the origin under the validated -/// configuration, returning the outcome with the counters before and after the solve. The control -/// radius is the configuration's initial radius. +/// Drives one inner Newton solve at the origin under the validated configuration. +/// +/// Prepares the corpus first and returns the outcome with the counters before and after the solve. +/// The control radius is the configuration's initial radius. fn newton_at_origin( corpus: &Corpus, config: SolverConfig, @@ -2246,9 +2283,9 @@ fn newton_at_origin( /// The interior Newton point inverts the oracle within its recorded residual. /// -/// The residual reads the step's Hessian product from the finite-difference-certified oracle, so -/// the residual proves the factorization against an implementation it shares no arithmetic with. At -/// the origin the scaled system is near-identity, so backward-stable factorization keeps the +/// The residual reads the step's Hessian product from the finite-difference-certified oracle, and +/// it therefore proves the factorization against an implementation it shares no arithmetic with. +/// At the origin the scaled system is near-identity, where backward-stable factorization keeps the /// relative residual within a few ulps, and the bound carries three orders of margin over that /// derivation. Each solve charges three assembly traversals, one factorization, and one priced /// product. @@ -2283,7 +2320,7 @@ fn newton_step_inverts_the_oracle_within_its_residual() { baseline.started_row_traversals + 4, ); - // A second identical solve returns identical bytes, so the engine is deterministic. + // A second identical solve returns identical bytes: the engine is deterministic. let (again, _, _) = newton_at_origin(&corpus, config); let again = again.expect("the identical solve succeeds identically"); assert_eq!(outcome.step().as_array(), again.step().as_array()); @@ -2295,8 +2332,8 @@ fn newton_step_inverts_the_oracle_within_its_residual() { /// A small trust radius exits the inner solve through the steepest-descent crossing. /// -/// At the origin the scaled gradient norm is about `1.4`, so both the Newton point and the Cauchy -/// point sit far outside a radius of `1e-4`. The crossing follows `−g` from the origin through the +/// At the origin the scaled gradient norm is about `1.4`, and both the Newton point and the Cauchy +/// point lie far outside a radius of `1e-4`. The crossing follows `−g` from the origin through the /// validated boundary construction. It prices one oracle product for the Cauchy curvature and never /// requests the Newton product. #[test] @@ -2386,7 +2423,7 @@ fn newton_step_crosses_the_dogleg_leg_between_cauchy_and_newton() { .expect("the gradient is finite"); let cauchy_norm = gradient_square.get() / curvature * gradient_norm; - // The dogleg premise of the witness: the fixture's Cauchy point sits strictly inside the + // The dogleg premise of the witness: the fixture's Cauchy point lies strictly inside the // Newton length. assert!( cauchy_norm < 0.9 * newton_norm, @@ -2451,7 +2488,7 @@ fn newton_step_survives_saturated_rows() { config, }; - // Rows 0 and 2 carry a nonzero leading coordinate, so their reference differences reach + // Rows 0 and 2 carry a nonzero leading coordinate: their reference differences reach // `±O(10³)` and the shifted exponentials underflow to exact vertices. Row 1 stays interior. let mut point = ContrastVector::zero(); point.coefficients[0].as_array_mut()[0] = 4000.0; diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/work.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/work.rs index c05fe013d57..70f04e282ee 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/work.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/solver/work.rs @@ -8,10 +8,10 @@ //! it accesses its first row, a visit per row it examines, and a completion once it has seen //! every row. //! -//! The increment rules live here as methods rather than at call sites, so a joint pass can never -//! forget that it serves one objective request and one gradient request with a single traversal, -//! and a preparation row visit can never reach the preparation counters without also charging the -//! global row-visit count. +//! Keeping the increment rules here as methods rather than at call sites means a joint pass can +//! never forget that it serves one objective request and one gradient request with a single +//! traversal, and a preparation row visit can never reach the preparation counters without also +//! charging the global row-visit count. /// Logical and physical work counters of one fit. #[derive(Debug, Copy, Clone, PartialEq, Eq, Default)] @@ -89,7 +89,7 @@ impl WorkCounters { /// Charges a joint request: one objective and one gradient request. /// - /// The first row access charges the pass and its traversal on their own, so a request rejected + /// The first row access charges the pass and its traversal on their own, and a request rejected /// before any row access (a non-finite input) still counts as requested work. pub(super) const fn request_joint(&mut self) { self.objective_requests += 1; diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/tests.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/tests.rs index fee488f1296..1db77c82d96 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/fit/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/fit/tests.rs @@ -28,6 +28,7 @@ use crate::{ salt::policy::GeometryClass, }; +/// The most rows a fixture corpus holds. const CAPACITY: usize = 8; /// An owned training corpus growing row by row. @@ -37,6 +38,7 @@ struct Corpus { } impl Corpus { + /// An empty corpus with zeroed storage. fn new() -> Self { Self { storage: BoxedVecN::zero(), @@ -44,6 +46,13 @@ impl Corpus { } } + /// Appends a row with the given leading embedding components, soft target, weight and group. + /// + /// The embedding starts with `leading` and is zero beyond it. + /// + /// # Panics + /// + /// Panics when the corpus already holds [`CAPACITY`] rows. fn push( &mut self, leading: &[f32], @@ -62,6 +71,7 @@ impl Corpus { }); } + /// The pushed rows' embeddings as an aligned card-row slice. fn embeddings(&self) -> &IdSlice> { let raw = AlignedVecN::from_slice( &self.storage.as_array()[..self.rows.len() * CANONICAL_DIMENSIONS], @@ -70,11 +80,13 @@ impl Corpus { IdSlice::from_raw(raw) } + /// The validated training set over the pushed rows. fn training(&self) -> TrainingSet<'_> { TrainingSet::new(self.embeddings(), &self.rows).expect("the fixture corpus validates") } } +/// The SHA-256 digest of `bytes`, used as a group identity. fn digest(bytes: &[u8]) -> Sha256Digest { let mut hasher = Sha256::new(); hasher.update(bytes); @@ -108,6 +120,11 @@ fn mixed_corpus() -> Corpus { corpus } +/// Rejects malformed `TrainingSet::new` inputs with the variant naming the offending row and value. +/// +/// `TrainingSet::new` rejects an empty corpus, mismatched row counts, a non-finite embedding +/// component, a target outside `[0, 1]`, a target not summing to one and a zero weight, each with +/// the variant naming the offending row and value. #[test] fn training_set_rejects_contract_violations() { let empty = TrainingSet::new(IdSlice::empty(), IdSlice::empty()) @@ -242,6 +259,7 @@ fn one_hot_corpus() -> Corpus { corpus } +/// The Euclidean norm of the coefficient block of `parameters`, intercepts excluded. fn coefficient_norm(parameters: &Parameters) -> f64 { parameters.as_array()[..PARAMETER_COUNT - GeometryClass::COUNT] .iter() @@ -250,6 +268,10 @@ fn coefficient_norm(parameters: &Parameters) -> f64 { .sqrt() } +/// Shrinks the coefficient norm under regularisation `10` against `0.1`. +/// +/// Fitting the one-hot corpus at regularisation `10` yields a smaller coefficient norm than at +/// `0.1`. #[test] fn stronger_regularization_shrinks_the_fitted_coefficients() { let corpus = one_hot_corpus(); @@ -264,8 +286,7 @@ fn stronger_regularization_shrinks_the_fitted_coefficients() { }; let gram = Gram::assemble(corpus.embeddings().as_raw(), &mut WorkCounters::default()); - // Every row assigned to fold 0: a raw fixture driving the fold fit directly, - // since a single-fold run is the point of this fixture. + // every row belongs to fold 0. Passing None fits the complete corpus at each strength. let folded = FoldedTraining { training, folds: IdSlice::from_raw(&[0, 0, 0]), @@ -285,6 +306,10 @@ fn stronger_regularization_shrinks_the_fitted_coefficients() { assert!(coefficient_norm(&strong) < coefficient_norm(&weak)); } +/// A corpus with no coincident mass fails with `MissingClassMass` naming the class. +/// +/// A corpus with no mass on the coincident class fails with `PreparationError::MissingClassMass` +/// naming that class. #[test] fn fit_model_requires_complete_class_mass() { let mut corpus = Corpus::new(); @@ -292,7 +317,7 @@ fn fit_model_requires_complete_class_mass() { corpus.push(&[0.0, 1.0], [0.0, 0.5, 0.5], 1.0, b"two"); let gram = Gram::assemble(corpus.embeddings().as_raw(), &mut WorkCounters::default()); - // Every row assigned to fold 0: a raw fixture driving the fold fit directly. + // every row belongs to fold 0. Passing None fits the complete corpus. let error = FoldedTraining { training: corpus.training(), folds: IdSlice::from_raw(&[0, 0]), @@ -324,11 +349,13 @@ fn overconfident_logits_calibrate_above_one() { let temperature = calibration::fit_temperature(rows, logits); - // softmax([6, 0, 0] / T) equals the target at exp(6 / T) = 3, an - // interior optimum of the [0.05, 20] bracket. Near the optimum the - // cross-entropy is flat below f64 resolution over a relative - // window of √(2 · ε / 0.24) ~ 3e-8 in ln T, so the - // search cannot localize tighter than that. + // for x = ln T and a = 6e⁻ˣ, these identical rows have mean cross-entropy H(x) = ln(eᵃ + 2) − + // 0.6a near the optimum, where the probability floor is inactive. With p = eᵃ/(eᵃ + 2), Hₐ = p + // − 0.6 vanishes at a = ln 3. This gives T = 6/ln 3 inside [0.05, 20] and curvature Hₓₓ = + // 0.24(ln 3)². An absolute objective perturbation η gives the local scale √(2η/[0.24(ln 3)²]) + // in ln T, approximately the relative change in T. Taking η = 2⁻⁵² gives about 3.9 × 10⁻⁸. + // Therefore the 10⁻⁶ relative tolerance leaves margin over this illustrative scale, without + // treating η as a bound on the implementation's rounding error. let expected = 6.0 / 3.0_f64.ln(); assert!(temperature > 1.0); assert!((temperature - expected).abs() <= 1.0e-6 * expected); @@ -344,6 +371,10 @@ fn overconfident_logits_calibrate_above_one() { } } +/// Keeps the calibrated cross-entropy at or below the raw one. +/// +/// The fitted temperature's calibrated cross-entropy never exceeds the raw one, since the unit +/// temperature is always a candidate. #[test] fn calibration_never_worsens_cross_entropy() { let rows = [ @@ -370,6 +401,10 @@ fn calibration_never_worsens_cross_entropy() { assert!(metrics.calibrated_cross_entropy <= metrics.raw_cross_entropy); } +/// Reads cross-entropy `ln 3` and Brier score `2/3` for uniform logits against a one-hot target. +/// +/// Uniform logits against a one-hot target give cross-entropy `ln 3` and Brier score `2/3` within +/// `1e-15`. #[test] fn metrics_match_hand_computed_values() { let rows = [TrainingRow { @@ -382,7 +417,7 @@ fn metrics_match_hand_computed_values() { let metrics = calibration::metrics(IdSlice::from_raw(&rows), IdSlice::from_raw(&logits), 1.0) .expect("finite fixture rows have finite metrics"); - // Uniform probabilities: CE = ln 3, Brier = (2/3)^2 + 2 · (1/3)^2. + // uniform probabilities: CE = ln 3, Brier = (2/3)² + 2 · (1/3)². assert!((metrics.raw_cross_entropy.get() - 3.0_f64.ln()).abs() <= 1.0e-15); assert!((metrics.raw_brier.get() - 2.0 / 3.0).abs() <= 1.0e-15); } @@ -415,8 +450,8 @@ fn applicability_matches_hand_computed_values() { assert!((scales[0] - expected_scales.0).abs() <= 1.0e-9 * expected_scales.0); assert!((scales[1] - expected_scales.1).abs() <= 1.0e-9 * expected_scales.1); - // Both rows sit one leading unit from the mean, so their distances - // agree: √(scale^2 / dimensions). + // both rows differ from the mean by one unit in the leading coordinate. Their distances agree: + // √(scale² / dimensions). let expected_distance = (expected_scales.0 * expected_scales.0 / 3072.0).sqrt(); assert_eq!(fitted.distances.len(), 2); for &distance in &fitted.distances { @@ -424,6 +459,7 @@ fn applicability_matches_hand_computed_values() { } } +/// A corpus of identical embeddings standardizes to unit inverse scales and zero distances. #[test] fn constant_corpus_gets_unit_scales_and_zero_distances() { let mut corpus = Corpus::new(); @@ -444,6 +480,10 @@ fn constant_corpus_gets_unit_scales_and_zero_distances() { assert!(fitted.distances.iter().all(|distance| *distance == 0.0)); } +/// Maps each class's coefficient block and the trailing intercepts in `split_parameters`. +/// +/// `split_parameters` maps each class's coefficient block to its row and the trailing three entries +/// to the intercepts. #[test] fn split_parameters_places_rows_and_intercepts() { let mut parameters = Parameters::zero(); @@ -482,6 +522,11 @@ fn soft_corpus() -> Corpus { corpus } +/// Fits the separable soft corpus with each returned quantity inside its bounds. +/// +/// Fitting the separable soft corpus yields a temperature inside `(0.05, 20)`, a fold per row below +/// the fold count, finite out-of-fold logits, calibrated cross-entropy no worse than raw, weak +/// regularisation, and raw posteriors within `0.05` of every generating target. #[test] fn fit_recovers_the_generating_distributions() { let corpus = soft_corpus(); @@ -516,9 +561,8 @@ fn fit_recovers_the_generating_distributions() { assert!(fitted.evidence.regularization <= DPositive::ONE); assert!(fitted.evidence.iterations >= 1); - // The separable corpus rewards weak regularization out of fold, so the - // selection stays weak and the fitted raw posteriors reproduce the - // generating soft targets on the training rows. + // weak regularization minimizes out-of-fold loss for this separable corpus. The deployment + // fit's raw posteriors approximate the generating soft targets on its training rows. for (row, expected) in corpus.rows.iter_enumerated() { let prediction = fitted .classifier @@ -534,6 +578,10 @@ fn fit_recovers_the_generating_distributions() { } } +/// Picks the interior minimum, the improving end and the stronger tie in `regularization::winner`. +/// +/// `regularization::winner` picks the interior minimum, the last candidate of an improving curve, +/// and on an exact tie the stronger penalty. #[test] fn regularization_winner_takes_the_minimum_and_ties_prefer_the_stronger_penalty() { let reading = |regularization: f64, cross_entropy: f64| regularization::RegularizationReading { @@ -574,8 +622,9 @@ fn fit_selects_regularization_and_records_the_curve() { let winner = regularization::winner(curve); assert_eq!(fitted.evidence.regularization, curve[winner].regularization); - // The winner's reading and the reported raw metric are the same reduction - // over the same logits, so the equality is exact. + // identical arithmetic over identical inputs gives identical results. The winner's reading and + // the raw metric use the same cross-entropy reduction over the same rows and logits at T = 1. + // Therefore the equality is exact. assert_eq!( fitted.evidence.raw_cross_entropy, curve[winner].cross_entropy @@ -589,6 +638,7 @@ fn fit_selects_regularization_and_records_the_curve() { assert_eq!(again.evidence.selection, fitted.evidence.selection); } +/// A one-iteration outer budget fails the fit with `SolverFailure::OuterIterationBudget`. #[test] fn exhausted_outer_iteration_budget_is_an_error() { let corpus = soft_corpus(); @@ -650,6 +700,7 @@ struct RecordingProgress { } impl RecordingProgress { + /// The fold counts announced by `classifier_started`, in report order. fn announced(&self) -> Vec { self.announced .lock() @@ -657,7 +708,7 @@ impl RecordingProgress { .clone() } - /// The completed folds, ascending: the pool finishes them in its own order. + /// Returns completed folds in ascending order, independent of worker completion order. fn completed(&self) -> Vec { let mut folds = self .completed @@ -671,7 +722,6 @@ impl RecordingProgress { } impl Progress for RecordingProgress { - /// The fixture watches folds, so nothing crosses into owning machinery. type Detached = NoProgress; fn detach(&self) -> NoProgress { @@ -693,6 +743,10 @@ impl Progress for RecordingProgress { } } +/// Announces three folds once and completes each exactly once with no deployment completion. +/// +/// A three-fold fit announces three folds once and completes folds `0`, `1` and `2` exactly once +/// each, and the deployment fit adds no completion. #[test] fn every_cross_validation_fold_reports_once() { let corpus = soft_corpus(); @@ -708,12 +762,14 @@ fn every_cross_validation_fold_reports_once() { ) .expect("the separable corpus fits"); - // The fit trains four models. The fourth holds nothing out and is the deployment model rather - // than a fold, so the counter's ceiling is the announced three. + // each of the 13 candidate strengths fits three held-out models, followed by one full-corpus + // deployment fit: 13 × 3 + 1 = 40. Progress counts folds. Each reports once after its last + // candidate succeeds, and the deployment fit adds no fold completion. assert_eq!(progress.announced(), [3]); assert_eq!(progress.completed(), [0, 1, 2]); } +/// A fit whose solver cannot converge announces its folds but completes none of them. #[test] fn a_fit_that_never_converges_completes_no_fold() { let corpus = soft_corpus(); @@ -732,9 +788,8 @@ fn a_fit_that_never_converges_completes_no_fold() { ) .expect_err("one outer iteration cannot converge"); - // The announcement is the workload, not a promise it will land: a - // model that failed has not completed, so the bar stays empty - // rather than filling as the failures arrive. + // a fold reports completion only after all of its candidates succeed. This iteration budget + // prevents convergence, and a failed solve returns before decrementing the pending count. assert_eq!(progress.announced(), [2]); assert_eq!(progress.completed(), [0_usize; 0]); } diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/mod.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/mod.rs index 46c94c2a0c5..6a9be6c5868 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/mod.rs @@ -20,7 +20,7 @@ //! the sorted training distances, //! //! ```text -//! distance = √(mean(((e - mean) · inverse_scale)^2)) +//! distance = √(mean(((e - mean) · inverse_scale)²)) //! applicability = 1 - lower_bound(training_distances, distance) / N. //! ``` //! @@ -29,10 +29,10 @@ //! classifier's. //! //! [`fit()`] trains the model from a weighted soft-label corpus. Raw and calibrated posteriors stay -//! separate in [`Prediction`], so a caller cannot apply calibration twice. +//! separate in [`Prediction`], and a caller cannot apply calibration twice. //! -//! Inputs are `f32` data widened to `f64` at the arithmetic seams ([`AlignedVecN::dot_wide`]); -//! parameters and outputs live in `f64`. +//! Inputs are `f32` data widened to `f64` at the arithmetic boundaries ([`AlignedVecN::dot_wide`]). +//! Parameters and outputs live in `f64`. use core::{ error::Error, @@ -79,7 +79,7 @@ impl Error for PredictError {} /// The standardization the applicability distance measures under. /// -/// `inverse_scales` components are positive; [`fit()`] and validated artifact reads are the +/// `inverse_scales` components are positive. [`fit()`] and validated artifact reads are the /// construction sites. #[derive(Debug, Clone, PartialEq)] pub(crate) struct Standardization { @@ -88,10 +88,11 @@ pub(crate) struct Standardization { } impl Standardization { - /// Standardized diagonal-Mahalanobis distance of an embedding from the training distribution. + /// Computes an embedding's standardized distance from the training distribution. /// - /// Computes `√(mean(((e - mean) · inverse_scale)^2))`, accumulated in double precision over - /// two independent chains. Returns [`None`] when the reduction is not finite. + /// The distance is the diagonal Mahalanobis distance `√(mean(((e - mean) · inverse_scale)²))`, + /// accumulated in double precision over two independent chains. Returns [`None`] when the + /// reduction is not finite. fn distance(&self, embedding: &AlignedVecN) -> Option { let (embedding, embedding_rest) = embedding.lanes(); let (mean, mean_rest) = self.mean.lanes(); @@ -143,7 +144,7 @@ pub(crate) struct Prediction { /// The fitted policy classifier. /// -/// Coefficient rows follow class order. All parameters are finite and the temperature is positive; +/// Coefficient rows follow class order. All parameters are finite and the temperature is positive. /// [`fit()`] and validated artifact reads are the construction sites. #[derive(Debug, Clone, PartialEq)] pub(crate) struct Classifier { @@ -193,8 +194,8 @@ impl Classifier { .distances .partition_point(|training| *training < distance); - // `partition_point` keeps the insertion index at or below the length, so the ratio is a - // fraction by construction and only an empty distribution refuses. + // `partition_point` keeps the insertion index at or below the length. The ratio is + // therefore a fraction by construction, and only an empty distribution refuses. let applicability = UnitFraction::ratio(insertion as u64, self.applicability.distances.len() as u64) .expect("a fitted classifier holds a nonempty training distance distribution") diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/report/mod.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/report/mod.rs index f05fa1eca3a..4c28070ce10 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/report/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/report/mod.rs @@ -2,9 +2,10 @@ //! //! The report reconstructs the classifier training set from the generation's staged annotation //! artifacts (the [`replay`] facility), re-runs the full production fit under the echoed -//! configuration - fold assignment seeded by the echo, so the refit is deterministic - and records -//! whether the recomputed model reproduces the staged `.clsf` artifact byte-for-byte. A verified -//! bundle provably describes the deployed model, not a lookalike. +//! configuration - fold assignment seeded by the echo, and the refit is deterministic - and records +//! whether the recomputed model's SHA-256 reproduces the staged `.clsf` artifact's. A verified +//! bundle describes the deployed model rather than a lookalike, under the digest's collision +//! resistance. //! //! One JSON document carries everything a downstream renderer needs: the per-row records (identity, //! fold, soft target, weight, out-of-fold logits, raw and calibrated posteriors, and @@ -14,7 +15,7 @@ //! //! Failures panic with the failing step's error. A report run has no recovery path, and the error //! is the diagnosis. The byte certification is the exception: its verdict is the report's content, -//! so a digest mismatch compiles and serializes with its per-row evidence instead of panicking. +//! and a digest mismatch compiles and serializes with its per-row evidence instead of panicking. pub(crate) mod replay; @@ -38,7 +39,10 @@ struct Certification { staged: Sha256Digest, /// SHA-256 of the refit model's serialized bytes. recomputed: Sha256Digest, - /// Whether the digests agree, which certifies that the report describes the deployed model. + /// Whether the digests agree. + /// + /// Agreement certifies under SHA-256's collision resistance that the report describes the + /// deployed model. verified: bool, } @@ -109,8 +113,9 @@ pub(crate) struct ClassifierReport { } impl ClassifierReport { - /// Reconstructs the staged corpus and refits the deployed model, then certifies the bytes and - /// compiles the bundle. + /// Reconstructs the staged corpus and refits the deployed model. + /// + /// It then certifies the bytes and compiles the bundle. /// /// # Panics /// @@ -136,7 +141,7 @@ impl ClassifierReport { .write_into(std::io::sink()) .expect("writing to a sink performs no fallible IO"); // A digest mismatch serializes with its per-row evidence: the divergence is the most - // valuable thing this instrument can show. + // valuable thing this report can show. let temperature = refit.classifier.temperature(); let row_reports = rows @@ -166,8 +171,8 @@ impl ClassifierReport { // Finite by the refit's own certification: `certify` evaluated the objective over these // exact unscaled rows and refuses a non-finite value by name. The objective carries - // 0.5·λ·‖A‖² with λ positive by type, squares cannot cancel, so a non-finite row norm - // cannot reach a converged refit. + // 0.5·λ·‖A‖² with λ positive by type, and squares cannot cancel. A non-finite row norm + // therefore cannot reach a converged refit. let coefficient_norms = core::array::from_fn(|class| { refit.classifier.coefficients[class] .norm() @@ -200,12 +205,12 @@ impl ClassifierReport { } } - /// The reported training-row count. + /// Returns the reported training-row count. pub(crate) const fn row_count(&self) -> usize { self.rows.len() } - /// Whether the refit model reproduced the staged artifact bytes. + /// Returns whether the refit model reproduced the staged artifact bytes. pub(crate) const fn verified(&self) -> bool { self.certification.verified } diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/report/replay.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/report/replay.rs index 52f95de405a..75868a4b293 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/report/replay.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/report/replay.rs @@ -1,17 +1,17 @@ //! Reconstruction of a frozen classifier corpus from staged annotation artifacts. //! -//! The lab instruments (the fold probe, the classifier report) re-run classifier machinery over the -//! exact bytes a production fit consumed. The staged `annotation-corpus.json` document replays +//! The diagnostic tools (the fold probe, the classifier report) re-run classifier machinery over +//! the exact bytes a production fit consumed. The staged `annotation-corpus.json` document replays //! through the production assembly under the generation's echoed assembly configuration. The //! embedder answers every card text from the staged embedding table by text hash and refuses new //! embeddings. Reconstruction requires that serializing the reassembled table reproduce the staged -//! array files byte-for-byte under SHA-256 equality, which proves the replayed inputs are the bytes -//! the production fit consumed. +//! array files' SHA-256 digests, which establishes, under the digest's collision resistance, that +//! the replayed inputs are the bytes the production fit consumed. //! //! The artifacts come from a published generation ([`Frozen::load`]) or from a directory of //! supplied artifact files ([`Frozen::from_supplied`]). The supplied form exists for a fit that //! cannot publish. A failing fit stages no generation for probing, but its input artifacts exist on -//! disk, and the table certification holds either way; only a published generation records a +//! disk, and the table certification holds either way. Only a published generation records a //! document digest for the replay to check. //! //! Failures panic with the failing step's error. A replay has no recovery path, and the error is @@ -45,10 +45,12 @@ use crate::{ /// A card embedder answering from a staged embedding table by text hash. /// -/// The fingerprint is the staged table's, so reassembly reproduces the table identity; a text +/// The fingerprint is the staged table's, and reassembly reproduces the table identity. A text /// absent from the table panics, because the frozen corpus admits no new embeddings. struct TableEmbedder { + /// The staged table's embedding contract. fingerprint: EmbedderFingerprint, + /// Each staged row's embedding under its card text's digest. rows: std::collections::HashMap>, } @@ -88,6 +90,11 @@ impl CardEmbedder for TableEmbedder { self.fingerprint } + /// Looks up every text's embedding by its SHA-256 digest. + /// + /// # Panics + /// + /// Panics when a text's digest is absent from the staged table. fn embed<'text>( &self, texts: impl IntoIterator + Send, @@ -110,26 +117,37 @@ impl CardEmbedder for TableEmbedder { /// The document, staged table, and echoed configuration of one frozen generation. pub(crate) struct Frozen { + /// The validated corpus document with its exact wire bytes. supplied: SuppliedAnnotations, + /// The embedder answering from the staged table. embedder: TableEmbedder, - /// The manifest-recorded corpus document digest; a supplied-artifact corpus has no manifest - /// and records none. + /// The manifest-recorded corpus document digest. + /// + /// A supplied-artifact corpus has no manifest and records none. staged_document_digest: Option, + /// The digest of the loaded document bytes. document_digest: Sha256Digest, + /// The staged embedding array's digest, which the reassembled table must reproduce. staged_embeddings_digest: Sha256Digest, + /// The staged hash column's digest, which the reassembled table must reproduce. staged_hashes_digest: Sha256Digest, - /// The staged classifier artifact's recorded identity; a supplied-artifact corpus stages no - /// classifier and carries none. + /// The staged classifier artifact's recorded identity. + /// + /// A supplied-artifact corpus stages no classifier and carries none. staged_classifier_digest: Option, - /// The generation's echoed assembly configuration; the replay binds it, never the compiled - /// defaults. + /// The generation's echoed assembly configuration. + /// + /// The replay binds it, never the compiled defaults. assembly: AssemblyConfig, + /// The generation's echoed classifier fit configuration. fit: FitConfig, } impl Frozen { - /// The staged classifier artifact's recorded identity; [`None`] for supplied artifacts, which - /// stage no classifier. + /// Returns the staged classifier artifact's recorded identity. + /// + /// [`None`] for artifacts loaded through [`Self::from_supplied`], which come with no published + /// generation and record no classifier. pub(crate) const fn staged_classifier_digest(&self) -> Option { self.staged_classifier_digest } @@ -193,11 +211,12 @@ impl Frozen { /// /// `directory` holds the three artifact files under their staged names: /// `annotation-corpus.json`, `annotation-embeddings.arr`, and `annotation-hashes.arr`. Supplied - /// artifacts carry no configuration echo, so the assembly and fit configurations are the - /// compiled deployment defaults; [`reconstruct`](Self::reconstruct) still certifies the - /// reassembled table against the supplied bytes, so a default assembly that diverges from the - /// one that produced the artifacts fails the byte certification instead of probing a different - /// corpus. + /// artifacts carry no configuration echo, and the assembly and fit configurations are + /// therefore the compiled deployment defaults. [`reconstruct`](Self::reconstruct) still + /// certifies the reassembled table against the supplied bytes, which pins the rendered card + /// texts and their embeddings. The group budget, the one assembly setting, leaves those bytes + /// unchanged: a default that differs from the producing fit's changes the validation groups + /// without failing the certification. /// /// # Panics /// @@ -219,9 +238,9 @@ impl Frozen { std::fs::read(hashes_path.as_std_path()).expect("the supplied hash column reads"), ); - // Supplied artifacts name no embedder, so the fingerprint derives from the supplied table's - // own digest and labels the reassembled table's identity without entering the certified - // bytes. + // Supplied artifacts name no embedder. The fingerprint therefore derives from the supplied + // table's own digest and labels the reassembled table's identity without entering the + // certified bytes. let embedder = TableEmbedder::load( EmbedderFingerprint::new(embeddings_digest), &hashes_path, @@ -303,7 +322,7 @@ pub(crate) struct Reconstructed { impl Reconstructed { /// The trained prefix of the embedding table. /// - /// The trained rows lead the table, so the prefix keeps the corpus's card-row identities; the + /// The trained rows lead the table, and the prefix keeps the corpus's card-row identities. The /// pin claims that domain over the table's domain-neutral rows. pub(crate) fn trained_embeddings( &self, @@ -324,6 +343,7 @@ impl Reconstructed { #[cfg(test)] mod tests { + use std::fs; use camino::Utf8PathBuf; @@ -338,11 +358,14 @@ mod tests { salt::embedding::EmbedderFingerprint, }; + /// The record hash the fixture card and vote carry. const DIGEST: &str = "2a9934acae8bf210b6a3428e553b1bcc0e220a4de113940782cd573da1ea4f4b"; + /// The versioned HASH type URL of the fixture card. const EMPLOYED_BY: &str = "https://hash.ai/@h/types/entity-type/employed-by/v/1"; - /// Composes a minimal contract-conforming corpus document: one hash card carrying one - /// geometry vote. + /// Composes a minimal contract-conforming corpus document. + /// + /// One hash card carrying one geometry vote. fn document() -> String { json!({ "cards": [{ @@ -412,14 +435,16 @@ mod tests { dir } + /// Reads the three artifact files under their staged names and binds the supplied digests. + /// /// The constructor reads the three artifact files under exactly the staged names its /// documentation promises, and binds the supplied bytes' digests as the staged identities. #[test] fn from_supplied_reads_the_staged_artifact_names() { let dir = scratch("staged-names"); - // The literals spell the constructor's documented human contract; the constructor joins - // the pinned artifact names, so either side drifting fails this witness. + // The literals spell the constructor's documented human contract, and the constructor + // joins the pinned artifact names. Either side drifting therefore fails this witness. let document = document(); fs::write(dir.join("annotation-corpus.json"), &document) .expect("the corpus document should write"); @@ -450,8 +475,8 @@ mod tests { let frozen = Frozen::from_supplied(&dir); - // A supplied corpus records no manifest digest, so the recorded-digest check has no - // second opinion to forge. + // A supplied corpus records no manifest digest, and the recorded-digest check therefore + // has no second opinion to forge. assert_eq!(frozen.document_digest, Sha256Digest::of(&document)); assert!(frozen.staged_document_digest.is_none()); diff --git a/libs/@local/graph/atlas/src/salt/policy/classifier/tests.rs b/libs/@local/graph/atlas/src/salt/policy/classifier/tests.rs index 0c668cb2971..65a6df66a0f 100644 --- a/libs/@local/graph/atlas/src/salt/policy/classifier/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/classifier/tests.rs @@ -50,14 +50,20 @@ fn classifier( } } +/// `Posterior::softmax` over equal logits is exactly uniform. #[test] fn softmax_of_equal_logits_is_uniform() { let uniform = Posterior::softmax([0.0, 0.0, 0.0], 1.0).to_array(); assert_eq!(uniform, [1.0 / 3.0; 3]); } +/// Agrees with the textbook softmax within `1e-15` at three temperatures. +/// +/// The shift-stabilized softmax agrees with the textbook `exp / Σ exp` within `1e-15` at three +/// temperatures. #[test] fn softmax_matches_an_unshifted_reference() { + /// Computes the textbook softmax of `logits` at `temperature`, without the stabilizing shift. fn reference(logits: [f64; 3], temperature: f64) -> [f64; 3] { let exponentials = logits.map(|value| (value / temperature).exp()); let denominator = exponentials.iter().sum::(); @@ -74,6 +80,10 @@ fn softmax_matches_an_unshifted_reference() { } } +/// Keeps the exact degenerate distribution for a logit near `f64::MAX`. +/// +/// A logit near `f64::MAX` over a temperature below one still yields the exact degenerate +/// distribution, since the shift precedes the division. #[test] fn softmax_of_extreme_logits_stays_a_distribution() { // A logit near `f64::MAX` over a temperature below one overflows to `+∞` if divided @@ -83,6 +93,7 @@ fn softmax_of_extreme_logits_stays_a_distribution() { assert_eq!(posterior, [1.0, 0.0, 0.0]); } +/// A higher temperature lowers the top probability and raises the bottom one. #[test] fn temperature_flattens_the_distribution() { let raw = Posterior::softmax([2.0, 0.0, -1.0], 1.0).to_array(); @@ -91,6 +102,10 @@ fn temperature_flattens_the_distribution() { assert!(calibrated[2] > raw[2]); } +/// Reads the standardized distance `√(9 / 3072)` exactly for a one-component embedding. +/// +/// The standardized distance of a one-component embedding equals `√(9 / 3072)` exactly under zero +/// mean and unit inverse scales. #[test] fn standardized_distance_matches_the_definition() { let embedding = embedding(&[3.0]); @@ -103,13 +118,18 @@ fn standardized_distance_matches_the_definition() { let distance = standardization.distance(&embedding); - // 9 / 3072 is exactly representable, so both paths round identically. + // 9 / 3072 = 3 / 1024 is exactly representable: both paths round identically. assert_eq!( distance.expect("the fixture distance is finite"), (9.0_f64 / 3072.0).sqrt() ); } +/// Checks the `predict` outputs on a fixture whose distance ties the middle training distance. +/// +/// `predict` computes the logits as coefficient dot products plus intercepts, the raw and +/// temperature-calibrated posteriors, the standardized distance, and an applicability of `2/3` for +/// a distance tying the middle training distance. #[test] fn predict_computes_logits_posteriors_and_applicability() { let rows = [ @@ -148,6 +168,7 @@ fn predict_computes_logits_posteriors_and_applicability() { assert_eq!(prediction.applicability, 1.0 - 1.0 / 3.0); } +/// An embedding far beyond every training distance predicts with applicability zero. #[test] fn predict_ranks_an_outlier_inapplicable() { let model = classifier( @@ -163,6 +184,10 @@ fn predict_ranks_an_outlier_inapplicable() { assert_eq!(prediction.applicability, 0.0); } +/// Overflows the logit with `f64::MAX` against `f32::MAX` and returns `PredictError`. +/// +/// A coefficient of `f64::MAX` against an `f32::MAX` component overflows the logit and `predict` +/// returns `PredictError`. #[test] fn predict_rejects_overflow() { let model = classifier( diff --git a/libs/@local/graph/atlas/src/salt/policy/mod.rs b/libs/@local/graph/atlas/src/salt/policy/mod.rs index 4c3c9b3a7a7..4e1d52b91cb 100644 --- a/libs/@local/graph/atlas/src/salt/policy/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/mod.rs @@ -3,7 +3,7 @@ //! Every relation type in scope resolves to a distribution over the [`GeometryClass`]es, which //! downstream stages turn into attraction, protection, and admission decisions. The open-world //! [`classifier`] supplies the distribution for relation types without a higher-precedence explicit -//! policy record; [`precedence`] resolves the winning source per relation into the certified policy +//! policy record. [`precedence`] resolves the winning source per relation into the certified policy //! table. //! //! The classes describe geometric behaviour, never semantic valence: opposition, contradiction, and @@ -79,8 +79,8 @@ impl GeometryClass { clippy::cast_possible_truncation, reason = "the variant count is far below `u8::MAX`" )] - // SAFETY: the discriminants are the dense range `0..COUNT`, so every transmuted index is a - // declared `repr(u8)` variant. + // SAFETY: the discriminants are the dense range `0..COUNT`. Every transmuted index is + // therefore a declared `repr(u8)` variant. pub(crate) const VARIANTS: [Self; Self::COUNT] = core::array::from_fn(const |index| unsafe { core::mem::transmute::(index as u8) }); @@ -142,12 +142,14 @@ impl Posterior { /// Computes the temperature-scaled softmax of class logits. /// /// The logits shift by their maximum before the temperature division, then pass through the - /// max-shifted [`DVecN::softmax`], so finite logits and a positive finite temperature always - /// produce a valid distribution: components in the unit interval that sum to one. The order - /// matters. Dividing first can overflow one quotient to `+∞`, and a single infinite + /// max-shifted [`DVecN::softmax`]. Finite logits and a positive finite temperature therefore + /// always produce a valid distribution: components in the unit interval that sum to one. The + /// order matters. Dividing first can overflow one quotient to `+∞`, and a single infinite /// component then poisons the whole shifted vector with `∞ - ∞`. Shifting first is free, /// since softmax is shift-invariant. It also closes the overflow path: a pre-shifted logit - /// is never positive, so its quotient by any positive temperature never reaches `+∞`. + /// is never positive, and its quotient by any positive temperature therefore never reaches + /// `+∞`. A quotient can reach `−∞` under a small enough temperature. Its exponential is + /// then zero and the result is still a valid distribution. #[must_use] pub(crate) fn softmax(logits: [f64; GeometryClass::COUNT], temperature: f64) -> Self { let max = logits.iter().copied().fold(f64::NEG_INFINITY, f64::max); @@ -176,11 +178,11 @@ impl Posterior { /// The Coincident and Proximal components of a relation class distribution. /// -/// Overlay, the third class, carries no geometric weight, so the two stored components are the +/// Overlay, the third class, carries no geometric weight, and the two stored components are the /// distribution's entire geometric content. Each component lies in `0.0..=1.0`. -// The fields carry their own construction invariants, so the byte-level constructor is the -// validating try-cast derive: a candidate is a distribution pair exactly when both fields -// hold stored fractions. +// The fields carry their own construction invariants, and the byte-level constructor is therefore +// the validating try-cast derive. A candidate is a distribution pair exactly when both fields hold +// stored fractions. #[derive( Debug, Copy, @@ -221,7 +223,8 @@ impl ClassProbabilities { /// - Attraction weights come from the effective attraction distribution. /// - Protection masses come from the selected distribution and applicability. /// - The attraction group receives the strength multiplier unchanged. -// This type has no construction invariant of its own, so the derives admit byte-level construction. +// This type has no construction invariant of its own, and the derives therefore admit byte-level +// construction. // The `repr(C)` layout is the policy file's pinned wire row, checked field for field where the // artifact casts, and the try-cast derive validates every domain-typed field's bits at that cast. #[derive( @@ -250,14 +253,14 @@ pub(crate) struct RelationPolicy { pub applicability: UnitFraction, /// The frozen strength multiplier `h`, exactly 1 while the strength head is off. pub strength: NonNegative, - /// Layout filler pinning the tail padding; writers emit zero, readers ignore. + /// Layout filler pinning the tail padding. Writers emit zero, and readers ignore it. pub _pad: [u8; 4], } /// The certified policy table, strictly ascending by relation row. /// -/// [`resolve`] mints the table sorted with duplicate relations refused, so the order is a -/// construction fact. The checked door certifies tables assembled anywhere else. +/// [`resolve`] creates the table sorted with duplicate relations refused, and the order is a +/// construction fact. The checked constructor certifies tables assembled anywhere else. #[derive(Debug)] pub(crate) struct CertifiedPolicies(Vec); diff --git a/libs/@local/graph/atlas/src/salt/policy/precedence/mod.rs b/libs/@local/graph/atlas/src/salt/policy/precedence/mod.rs index 08a7990f5fc..470ac35e5d1 100644 --- a/libs/@local/graph/atlas/src/salt/policy/precedence/mod.rs +++ b/libs/@local/graph/atlas/src/salt/policy/precedence/mod.rs @@ -26,11 +26,11 @@ //! ``` //! //! A Coincident prediction that fails admission becomes Overlay, never Proximal. The policy row -//! keeps the Coincident and Proximal components alone. Overlay is the remainder, so the mix reduces -//! to scaling both stored components by `a`. An override asserts its distribution and has -//! applicability 1. [`Classification`] admits exactly one outcome, a classifier prediction, so -//! no fallback source sits below the classifier. -//! Strength is the unit multiplier while the strength head is off. +//! keeps the Coincident and Proximal components alone. Overlay is the remainder. The mix therefore +//! reduces to scaling both stored components by `a`. An override asserts its distribution and has +//! applicability 1. [`Classification`] admits exactly one outcome, a classifier prediction, and +//! no fallback source lies below the classifier. Strength is the unit multiplier while the +//! strength head is off. //! //! Resolution is where policy values leave the solver's double precision and narrow to //! working-precision data. @@ -130,7 +130,7 @@ pub(crate) struct PolicyOverride { /// The generation's global Coincident admission criteria. /// -/// Only generations that enable Coincident geometry enforce admission; unenforced, the attraction +/// Only generations that enable Coincident geometry enforce admission. Unenforced, the attraction /// distribution passes through the mix unchanged and the Coincident force coefficient governs /// downstream. The default thresholds are maximally conservative placeholders: a generation /// enforcing admission configures them from its precision release evidence. @@ -204,15 +204,15 @@ pub(crate) fn resolve( }) .collect(); - // Sorted above with duplicates refused, and the map preserves order, so the table is - // strictly ascending by construction. + // Sorted above with duplicates refused, and the map preserves order. The table is + // therefore strictly ascending by construction. Ok(CertifiedPolicies(policies)) } /// Applies the applicability mix and Coincident admission. /// -/// Overlay is the unstored remainder, so mixing toward it scales the stored components by `a`; -/// admission then reroutes a failing mixed Coincident mass to that remainder. +/// Overlay is the unstored remainder, and mixing toward it scales the stored components by `a`. +/// Admission then reroutes a failing mixed Coincident mass to that remainder. const fn attraction( selected: ClassProbabilities, applicability: UnitFraction, diff --git a/libs/@local/graph/atlas/src/salt/policy/precedence/tests.rs b/libs/@local/graph/atlas/src/salt/policy/precedence/tests.rs index 44ca6c37ba8..4d283194a82 100644 --- a/libs/@local/graph/atlas/src/salt/policy/precedence/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/precedence/tests.rs @@ -26,6 +26,7 @@ fn prediction(calibrated: [f64; 3], applicability: f64) -> Prediction { } } +/// An ontology row id from a literal. fn relation(row: u64) -> OntologyRowId { OntologyRowId::new(row) } @@ -67,6 +68,10 @@ fn prediction_resolves_through_the_applicability_mix() { assert_eq!(policy.strength, 1.0); } +/// Ranks a human override above the reviewed, synthetic and predicted sources. +/// +/// A human override outranks the reviewed and synthetic ones and the prediction, and asserted +/// records pass their distribution through the mix unchanged at applicability one. #[test] fn overrides_supersede_predictions_by_precedence() { let classifications = [( @@ -94,8 +99,8 @@ fn overrides_supersede_predictions_by_precedence() { let policies = resolve(&classifications, &overrides, CoincidentAdmission::default()) .expect("overrides resolve"); - // The human override wins; asserted records carry applicability 1, - // so the mix passes the distribution through unchanged. + // The human override wins. Asserted records carry applicability 1, + // and the mix passes the distribution through unchanged. let policy = policies[0]; assert_eq!( policy.selected, @@ -126,8 +131,8 @@ fn admission_reroutes_failing_coincident_mass() { relation(1), Classification::Predicted(prediction([0.25, 0.5, 0.25], 0.5)), ), - // Applicability 0.25 < 0.5 despite mixed Coincident 0.1875: - // rerouted. + // Applicability 0.25 < 0.5, and the mixed Coincident 0.75 · 0.25 = + // 0.1875 < 0.2 as well: rerouted on both criteria. ( relation(2), Classification::Predicted(prediction([0.75, 0.125, 0.125], 0.25)), @@ -146,6 +151,10 @@ fn admission_reroutes_failing_coincident_mass() { assert_eq!(policies[1].selected.coincident, 0.25); } +/// `resolve` refuses a duplicate, an ambiguous override and an unknown override. +/// +/// `resolve` fails with `DuplicateRelation` for a repeated classification, `AmbiguousOverride` for +/// two overrides of one source, and `UnknownOverride` for a relation with no classification. #[test] fn contract_violations_are_rejected() { let duplicate = resolve( @@ -219,6 +228,10 @@ fn contract_violations_are_rejected() { ); } +/// Orders resolution output by relation and passes certification unchanged. +/// +/// Resolution orders its output by relation regardless of input order, and the table passes +/// certification unchanged. #[test] fn resolution_feeds_the_certified_policy_table() { // Input order is irrelevant; the output is strictly ascending and diff --git a/libs/@local/graph/atlas/src/salt/policy/tests.rs b/libs/@local/graph/atlas/src/salt/policy/tests.rs index 1c4794d6384..a461436f29c 100644 --- a/libs/@local/graph/atlas/src/salt/policy/tests.rs +++ b/libs/@local/graph/atlas/src/salt/policy/tests.rs @@ -7,6 +7,10 @@ use zerocopy::TryFromBytes as _; use super::{GeometryClass, Posterior}; +/// Lists the three geometry classes in discriminant order under `VARIANTS`. +/// +/// `GeometryClass::VARIANTS` lists the three classes in discriminant order, matching `COUNT` and +/// each variant's declared discriminant. #[test] fn variants_enumerate_the_classes_in_class_order() { // Certifies the const-transmute derivation of `VARIANTS` against @@ -25,6 +29,7 @@ fn variants_enumerate_the_classes_in_class_order() { } } +/// Each declared discriminant byte parses to its class, and one past the last or `u8::MAX` refuses. #[test] fn wire_bytes_admit_only_declared_discriminants() { for class in GeometryClass::VARIANTS { @@ -40,6 +45,7 @@ fn wire_bytes_admit_only_declared_discriminants() { .expect_err("an undeclared discriminant must not parse"); } +/// `Posterior::new` accepts a distribution and reports each class's probability and the array. #[test] fn posterior_accepts_a_distribution() { let posterior = Posterior::new([0.5, 0.25, 0.25]).expect("a distribution should validate"); diff --git a/libs/@local/graph/atlas/src/salt/postings/artifact.rs b/libs/@local/graph/atlas/src/salt/postings/artifact.rs index 3e063ea7f96..e766f738f7a 100644 --- a/libs/@local/graph/atlas/src/salt/postings/artifact.rs +++ b/libs/@local/graph/atlas/src/salt/postings/artifact.rs @@ -1,4 +1,6 @@ -//! The postings archive and the membership views it serves. +//! Validation and mapped lookup of published type postings. +//! +//! The archive checks run ordering and domains before exposing membership and parent views. use core::ops::Range; @@ -11,7 +13,7 @@ use crate::{ runs::{RunsError, RunsView}, }; -/// An opened postings file does not hold a valid postings artifact. +/// A violation of the postings artifact's run or membership-count contract. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum InvalidPostingsFile { /// The list fenceposts break anchoring, ordering, or coverage at `position`. @@ -99,16 +101,13 @@ impl core::error::Error for InvalidPostingsFile {} /// A published postings artifact opened over its mapped file. /// -/// Construction checks the artifact contract once - fencepost anchoring/ordering/coverage in all -/// three fencepost regions, list ascent and domains, empty list runs for dense types, parent -/// ascent and domains, direct ascent and domains, and the pair count tying the direct map to the -/// membership total. An open postings therefore only serves valid runs and consumers re-validate -/// nothing. The bit set -/// frames were already validated when the file opened, where the format's geometry lives. The -/// archive holds the mapped file alone, and each lookup re-borrows its fencepost and items -/// regions as a [`RunsView`] pair the construction validated. A dense type's frame index is the -/// flag population below its row, read from the mapped flags frame at each lookup, so every -/// answer comes from file bytes and the regions stay in the page cache under memory pressure. +/// Construction validates fencepost anchoring, ordering and coverage in all three run regions. It +/// also checks strict ordering and domains in list, parent and direct runs, requires empty list +/// runs for dense types, and compares the direct entry count with the membership total. This count +/// check does not establish full transpose agreement between direct types and memberships. +/// +/// Lookups borrow the validated runs or dense frames from the mapped file without rebuilding them. +/// A dense type's frame index is the flag population below its row, computed at each lookup. #[derive(Debug)] pub(crate) struct PostingsArchive { file: PostingsFile, @@ -119,7 +118,7 @@ impl PostingsArchive { /// /// # Errors /// - /// Returns an error when the file violates the artifact contract. + /// Returns [`InvalidPostingsFile`] when the file violates the artifact contract. #[tracing::instrument(skip_all)] pub(crate) fn new(file: PostingsFile) -> Result { let types = file.types(); @@ -182,8 +181,8 @@ impl PostingsArchive { } } - // Every position-type pair appears once in each direction, so the direct entry count is - // the membership total: the list entries plus the dense populations. + // a transpose has one occurrence of each position-type pair in each direction. Compare the + // totals as a necessary condition, without reconstructing the full transpose. let dense_sets = file.dense_sets(); let membership = lists.items().len() as u64 + (0..dense_sets.len()) @@ -244,6 +243,9 @@ impl PostingsArchive { } /// Returns `position`'s direct type rows, strictly ascending, when the position is in domain. + // Production reads no direct types through the archive: construction validates the region + // against the membership total, and that is the region's whole production use. The postings + // tests read it to verify the written direct map restates the input type column. #[must_use] pub(crate) fn direct_types(&self, position: BasePosition) -> Option<&[OntologyRowId]> { let index = position.as_u64(); @@ -270,11 +272,12 @@ impl PostingsArchive { } } -/// Names the fencepost position a [`RunsError`] faults, for the per-region error variants. +/// Locates the invalid fencepost described by a [`RunsError`]. +/// +/// A missing column or broken anchor identifies the first post, an order violation identifies its +/// own index, and a closing mismatch identifies the last post. /// -/// A missing column and a broken anchor fault the first post, a break in the order faults its -/// own index, and a closing mismatch faults the last post - exactly the positions the archive -/// reported before the fencepost law moved into [`RunsView`]. +/// `posts` must be the length of the fencepost column that produced `error`. const fn post_position(error: RunsError, posts: usize) -> usize { match error { RunsError::Missing | RunsError::Anchor => 0, diff --git a/libs/@local/graph/atlas/src/salt/postings/build.rs b/libs/@local/graph/atlas/src/salt/postings/build.rs index 95ae795cb93..7a9d4200599 100644 --- a/libs/@local/graph/atlas/src/salt/postings/build.rs +++ b/libs/@local/graph/atlas/src/salt/postings/build.rs @@ -1,4 +1,6 @@ -//! The postings build turns the row-order type column into the file's regions. +//! Per-type membership derived from direct types in base delivery order. +//! +//! The direct map and its transpose are built together to keep both lookup directions consistent. use std::io; @@ -17,7 +19,7 @@ use crate::{ runs::{Runs, RunsBuilder}, }; -/// Building the postings failed. +/// An out-of-domain type reference encountered while building postings. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum PostingsError { /// A node row's direct types name an ontology row outside the type domain. @@ -48,25 +50,22 @@ impl core::error::Error for PostingsError {} /// The type postings of one generation, in writable form. /// -/// The direct map is the one stored relation - the row-order type column gathered into position -/// order - and the membership regions are its inversion, so the two directions agree by -/// construction. Construction picks each type's representation and lays every region out in the -/// file's order, with the fencepost columns at the build's native width; the writer persists -/// them little-endian as it streams. A type goes dense exactly when its dense set costs fewer -/// bytes than its list - [`DenseBitSlice::total_byte_len`] of the point domain against four -/// bytes per member. The choice therefore follows from the sizes alone and carries no tuning -/// knob. At equal cost the list wins because it reads without bit decoding. +/// The direct map is the row-order type column gathered into position order. Deriving membership by +/// inversion keeps the two directions consistent. A type uses a dense set exactly when that set +/// costs fewer bytes than its list: [`DenseBitSlice::total_byte_len`] of the point domain against +/// four bytes per member. The choice follows from the sizes alone and carries no tuning knob. At +/// equal cost the list wins because it reads without bit decoding. #[derive(Debug, PartialEq, Eq)] pub(crate) struct Postings { /// The types whose membership is a dense set. flags: Box>, /// Each list type's membership positions, ascending per type. A dense type's run is empty. lists: Runs, - /// The dense membership sets, one frame per dense type in ascending type order, each over - /// the point domain. + /// Dense membership frames in ascending type order, each over the point domain. dense_sets: Box>, - /// Each position's direct type rows, ascending per position. Its run count is the - /// base-position domain `N`. + /// Each position's direct type rows, ascending per position. + /// + /// Its run count is the base-position domain `N`. direct: Runs, /// Each type's direct parent rows, ascending per type. parents: Runs, @@ -76,13 +75,11 @@ impl Postings { /// Builds the postings over the finished lod permutation. /// /// `types` holds each node row's direct types in **row** order, exactly as the dataset streams - /// them (ascending, deduplicated); `row_of_position` is the lod's gather order, so the direct - /// map and the membership follow base delivery order. `parents` holds each ontology row's - /// direct parents in ontology-row order - the - /// [`Ontology::parents`](crate::dataset::Ontology::parents) contract, restated in file shape - - /// and its length is the type domain `T`. The build gathers the direct map first and derives - /// the membership regions from it by [`Inverse::new`], so every check of one direction binds - /// the other. + /// them (ascending, deduplicated). `row_of_position` must be a permutation of that row domain. + /// It gathers the direct map and membership into base delivery order. `parents` holds each + /// ontology row's direct parents in ontology-row order, following the + /// [`Ontology::parents`](crate::dataset::Ontology::parents) contract. Its length is the type + /// domain `T`. /// /// # Errors /// @@ -91,10 +88,9 @@ impl Postings { /// /// # Panics /// - /// This panics when `types` and `row_of_position` cover different row counts, and when a - /// row's direct types do not ascend strictly. The lod build already rejected mismatched - /// columns and the dataset contract promises ascending, deduplicated lists, so either - /// disagreement here is a producer bug. + /// This panics when `types` and `row_of_position` cover different row counts, when the + /// permutation names a row outside `types`, or when a row's direct types or a type's direct + /// parents do not ascend strictly. #[expect( clippy::panic_in_result_fn, reason = "the Result carries domain errors; mismatched columns and unsorted streams are \ @@ -114,9 +110,8 @@ impl Postings { let domain = parents.len(); - // The direct map is the gather itself: each position's run restates its row's type list - // verbatim, so the runs inherit the column's ascent and deduplication. The domain and - // ascent checks ride the gather. Every pass below trusts them. + // gather whole type lists into position order, checking ascent and domain as each list is + // read. Every pass below trusts these checks. let mut direct = RunsBuilder::with_capacity(row_of_position.len(), 0); for (_position, &row) in row_of_position.iter_enumerated() { let list = &types[row]; @@ -152,10 +147,7 @@ impl Postings { }) } - /// Measures the finished regions for the generation metadata. - /// - /// The measurements the manifest records so the representation split follows data rather than - /// taste: how many types went dense, and the region populations behind the artifact's size. + /// Counts the dense types and region populations behind the artifact's size. #[must_use] pub(crate) fn measurements(&self) -> PostingsMeasurements { PostingsMeasurements { @@ -181,16 +173,19 @@ struct Inverse { impl Inverse { /// Inverts the position-major direct map into the per-type membership regions. /// - /// This is the transpose: a type's membership holds exactly the positions whose direct runs - /// name the type, so the two directions carry one relation. Walking positions ascending makes - /// every list run sorted by construction: no sort pass exists. + /// A type's membership holds exactly the positions whose direct runs name the type. Walking + /// positions ascending produces sorted list runs without a sort pass. + /// + /// Every direct id must lie below `domain`. /// - /// Every direct id lies below `domain`. [`Postings::build`] validated that while gathering. + /// # Panics + /// + /// This panics when a direct id lies outside `domain`. fn new(direct: &Runs, domain: usize) -> Self { let points = direct.runs(); - // Member counts first: they pick each type's representation and become the fenceposts, so - // the fill pass below writes each entry at its final slot. + // member counts determine each type's representation and its final region. Their prefix + // sums place each list run for the fill pass. let mut counts = IdVec::from_elem(0_u64, domain); for &id in direct.items() { counts[id] += 1; @@ -203,8 +198,8 @@ impl Inverse { let dense_bytes = DenseBitSlice::::total_byte_len(points as u64); let is_dense = |count: u64| dense_bytes < count * size_of::() as u64; - // The dense count is known before the region exists, so the sets live in one allocation - // laid out exactly as the file stores them. + // counting dense types before allocation gives one region laid out exactly as the file + // stores it. let dense_count = counts.iter().filter(|&&count| is_dense(count)).count(); let mut dense_sets = DenseBitSliceArray::::new_empty(points, dense_count); @@ -228,8 +223,8 @@ impl Inverse { list_posts.push(total); } - // Fill in position order: each list run's cursor starts at its fencepost and ascending - // positions land ascending in place. Dense members insert into their type's set. + // fill in position order: each list run's cursor starts at its fencepost and writes + // ascending positions without a sort. Dense members insert into their type's set. let mut list_entries = vec![BasePosition::from_u32(0); total]; let mut cursors: IdVec = IdVec::from_raw(list_posts[..domain].to_vec()); @@ -270,14 +265,6 @@ impl WriteAs for Postings {} impl WriteInto for Postings { type Error = io::Error; - /// Writes the postings as a postings file. - /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. - /// - /// # Errors - /// - /// Returns an error when the underlying writer fails. fn write_into(&self, write: impl io::Write) -> io::Result { let mut writer = Writer { accumulator: Sha256::new(), @@ -308,7 +295,7 @@ impl WriteInto for Postings { pub(crate) struct PostingsMeasurements { /// Types in the domain. pub types: u64, - /// Types whose membership went dense under the size comparison. + /// Types whose membership uses a dense set under the size comparison. pub dense_types: u64, /// Entries in the list region: every list type's positions. pub list_entries: u64, @@ -320,16 +307,15 @@ pub(crate) struct PostingsMeasurements { /// Gathers the parent lists into their per-type runs. /// -/// The runs restate the dataset's stream. The domain check is the one condition the stream -/// cannot carry itself (parents may point forward). Ascent is the stream's own contract and is -/// asserted here, so a defective stream fails the build instead of publishing a file the next -/// open refuses. +/// Parent references may point forward. Their domain is the full length of `parents`. +/// +/// # Errors +/// +/// Returns [`PostingsError`] for an out-of-domain parent reference. /// /// # Panics /// -/// This panics when a type's direct parents do not ascend strictly. The -/// [`Ontology::parents`](crate::dataset::Ontology::parents) contract promises ascending, -/// deduplicated lists, so a violation here is a producer bug. +/// This panics when a type's direct parents do not ascend strictly. #[expect( clippy::panic_in_result_fn, reason = "the Result carries domain errors; an unsorted parent stream is a caller contract \ diff --git a/libs/@local/graph/atlas/src/salt/postings/closure.rs b/libs/@local/graph/atlas/src/salt/postings/closure.rs index aac8060dc82..d3072da4b50 100644 --- a/libs/@local/graph/atlas/src/salt/postings/closure.rs +++ b/libs/@local/graph/atlas/src/salt/postings/closure.rs @@ -1,14 +1,15 @@ -//! The type closure map gives each type a descendant bitset over the parent graph. +//! Inherited type memberships and icon sources derived from the parent graph. //! -//! Request-time inheritance expansion is one OR. A requested type's descendant row names every type -//! whose instances the request matches, so expanding a request is `OR` of the requested rows and -//! testing a type against a request is one bit read. The map derives at open from the published -//! parent edges, the one authority for inheritance, and lives on the heap: `T^2` bits stay in the -//! low megabytes while `T` stays in the low thousands. +//! The published parent edges are the authority for inheritance. For a domain of `T` types, a `T` +//! by `T` descendant matrix names every type whose instances a request naming an ancestor matches. +//! Each row folds into one membership bitset over base positions per type with descendants beyond +//! itself. A type whose only descendant is itself keeps no closure membership. Use its postings' +//! direct membership instead. The map retains the descendant matrix alongside these derived +//! memberships. //! -//! The same derivation resolves the icon memo: each type's nearest icon-bearing ancestor, settled -//! once at open, so a tile read costs one memo lookup per direct type. The memo records row -//! identities, and payload bytes resolve at read time against the table that owns them. +//! The topological order also resolves each type's nearest icon-bearing ancestor once at open. The +//! memo records row identities. Payload bytes resolve at read time against the table that owns +//! them. use hashql_core::id::{ Id as _, IdVec, @@ -17,10 +18,10 @@ use hashql_core::id::{ use crate::{identity::OntologyRowId, salt::postings::artifact::PostingsArchive}; -/// The parent graph holds a cycle, so no descendant order exists. +/// A cycle preventing a children-first ordering of the parent graph. /// -/// Type inheritance is acyclic at the source, so a cycle in published bytes means the generation's -/// ontology stream was defective. +/// Type inheritance requires an acyclic parent graph. This error reports a cycle in the supplied +/// parent edges. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub struct ParentCycle { /// Types entangled in cycles: every type whose descendant set never settled. @@ -50,37 +51,35 @@ pub(crate) struct IconSource { pub depth: u32, } -/// Descendant bitsets over the type domain, one row per type. +/// Inherited memberships and nearest icon sources for the type domain. /// -/// Row `t` marks every type whose instances a filter or coloring request naming `t` matches: `t` -/// itself and every type reaching `t` through parent edges - a `T` by `T` [`BitMatrix`], so a -/// request's expansion ORs whole rows word-wise. +/// A filter or coloring request naming `t` matches the instances of `t` itself and of every type +/// reaching `t` through parent edges. [`membership`](Self::membership) answers that set as one +/// bitset over base positions for types with descendants beyond themselves. For a type whose only +/// descendant is itself, use its direct postings membership. [`icon_source`](Self::icon_source) +/// resolves the memoized icon ancestor. #[derive(Debug, Clone)] pub(crate) struct ClosureMap { bits: BitMatrix, - /// The icon memo, one [`IconSource`] per type. - /// - /// [`None`] records an icon-free ancestor cone. icon_sources: IdVec>, } impl ClosureMap { /// Derives the closure map from the opened postings' parent graph. /// - /// The derivation walks types children-first in Kahn's ordering over the parent edges. A type's - /// settled descendant row ORs into each of its parents' rows, so every row settles in one pass - /// over the edges. + /// In an acyclic parent graph, a type's descendants are itself and the union of its children's + /// descendants. The derivation seeds the diagonal and processes types children-first in Kahn's + /// order, combining each settled row into its parents' rows with bitwise OR. Therefore every + /// row contains exactly its type's descendants once all of its children have settled. /// - /// `icons` names the type rows carrying their own icon, and a second pass unwinds the recorded - /// settle order, so every type follows its whole ancestor cone, resolving each type's - /// [`IconSource`]. An icon row resolves to itself at depth zero. Every other type takes the - /// shallowest parent resolution one edge deeper, and equal depths resolve to the earlier parent - /// in the run. + /// `icons` names the type rows carrying their own icon. A second pass reverses the settle order + /// to resolve each type's [`IconSource`] after its parents. Each icon row resolves to itself at + /// depth zero. Every other type takes the shallowest parent resolution one edge deeper. Equal + /// depths resolve to the earlier parent in the run. /// /// # Errors /// - /// Returns [`ParentCycle`] when the parent graph holds a cycle, in which case the generation's - /// ontology stream was defective. + /// Returns [`ParentCycle`] when the supplied parent graph holds a cycle. /// /// # Panics /// @@ -94,13 +93,12 @@ impl ClosureMap { let bound = OntologyRowId::from_usize(types); let mut bits = BitMatrix::new(types, types); - // Every type descends from itself: a request naming `t` - // matches instances of `t` directly. + // every type descends from itself: a request naming `t` matches instances of `t` directly. for type_row in OntologyRowId::MIN..bound { bits.insert(type_row, type_row); } - // Pending children per type. A type's row settles once it has absorbed every child's row. + // a type's row settles once it has absorbed every child's row. let mut pending: IdVec = IdVec::from_elem(0, types); for type_row in OntologyRowId::MIN..bound { let parents = postings @@ -140,16 +138,14 @@ impl ClosureMap { }); } - // An icon row is its own source at depth zero, and the resolution pass below leaves - // these seeded entries standing. + // an icon row is its own source at depth zero. The resolution pass preserves these entries. let mut icon_sources: IdVec> = IdVec::from_elem(None, types); for source in icons { icon_sources[source] = Some(IconSource { source, depth: 0 }); } - // The walk settled children before parents, so unwinding it hands every type its - // ancestors first: each type resolves against parents that already have. + // reversing the children-first order resolves every parent before its children. while let Some(r#type) = settled.pop() { if icon_sources[r#type].is_some() { continue; @@ -166,8 +162,8 @@ impl ClosureMap { }; let depth = depth + 1; - // Strictly-shallower replaces, so an equal-depth tie keeps the earlier parent - // in the run: ascending rows, the artifact contract. + // an equal-depth tie keeps the earlier parent in the run: ascending rows, the + // artifact contract. if best.is_none_or(|held| depth < held.depth) { best = Some(IconSource { source, depth }); } @@ -192,17 +188,10 @@ impl ClosureMap { (type_row.as_usize() < self.bits.row_domain_size()).then(|| self.bits.row(type_row)) } - /// Returns the [`IconSource`] `type_row` resolves to, or [`None`] for an icon-free cone. - /// - /// Equal-depth candidates resolve to the earlier parent in the run, so resolution is - /// deterministic under the artifact's ascending-row parent order. - /// - /// # Panics + /// Resolves the nearest icon-bearing ancestor within the closure. /// - /// This panics when `type_row` lies past the closure's type domain. The closure tabulates - /// the generation's own types, and a row the delta allocated past that bound resolves its - /// icon through the register's extension instead, so reaching here with one is a caller - /// routing bug rather than data. + /// Returns [`None`] outside the type domain or for an icon-free cone. Equal-depth candidates + /// resolve to the earlier parent in the artifact's ascending-row parent order. #[must_use] pub(crate) const fn icon_source(&self, type_row: OntologyRowId) -> Option { self.icon_sources[type_row] diff --git a/libs/@local/graph/atlas/src/salt/postings/mod.rs b/libs/@local/graph/atlas/src/salt/postings/mod.rs index 0763c6ef417..4862b086167 100644 --- a/libs/@local/graph/atlas/src/salt/postings/mod.rs +++ b/libs/@local/graph/atlas/src/salt/postings/mod.rs @@ -1,42 +1,35 @@ -//! The type postings. +//! Direct type memberships and the parent graph that defines their inheritance. //! -//! Per-type membership over the base delivery order, and the type graph it expands through. -//! -//! [`Postings`](build::Postings) is the filter contract's membership artifact. For every ontology -//! row it records which base delivery positions carry that type **directly**. A type filter ORs -//! requested rows' membership into one dense position bitmap, and the wire's `TYPE_MASK` column -//! slices membership over a tile's delivered runs. Inheritance never rides the membership. Requests -//! expand to descendant rows first through the [`ClosureMap`](closure::ClosureMap) derived from the -//! published parent edges, so the type graph stays the one authority for inheritance and no closure -//! is ever materialized on disk. +//! [`Postings`](build::Postings) records which base delivery positions carry each ontology row +//! **directly**. The published parent edges remain the authority for inheritance. +//! [`ClosureMap`](closure::ClosureMap) derives inherited memberships at open for types with +//! descendants beyond themselves. Requests borrow that derived membership when present and the +//! direct postings otherwise. Inheritance never changes the stored direct membership, and no +//! closure is materialized on disk. //! //! The file stores each type's membership in the cheaper of two representations. The writer chooses //! which one and the flags region records the choice: //! -//! - a **list**: the positions sorted ascending, `4` bytes each - the shape a run slice reads -//! linearly; -//! - a **dense set** over all `N` positions, one self-describing bit set frame - the shape the mega -//! types demand. Type volume is structurally skewed (one base type owns half of all instances in -//! the measured store), so an all-list format degenerates exactly on the types most worth -//! coloring by. +//! - a **list**: the positions sorted ascending, `4` bytes each, readable linearly; +//! - a **dense set** over all `N` positions, one self-describing bit set frame. Its size depends on +//! the point domain rather than the type's population. This bounds storage for heavily populated +//! types when membership volumes are skewed. //! //! Readers honor whichever representation the file records. The writer picks the cheaper one by -//! comparing byte costs - the frame against four bytes per member. The split therefore carries no +//! comparing byte costs - the frame against four bytes per member. The split carries no //! tuning knob and follows the data alone. //! //! Beside the membership the file stores its transpose, the **direct map** - each base position's //! direct type rows as one fencepost-delimited run per position. That is the position-scoped //! lookup - which types does this delivered position carry - answered from one run read. The //! build gathers the direct map from the row-order type column first and derives the membership -//! regions from it by inversion. Both directions therefore carry one relation and agree by -//! construction. -//! -//! It derives from the same row-order type column the quadtree consumes -//! ([`crate::salt::lod::quad::QuadTree::build`]'s `types` parameter), gathered through the lod's -//! permutation, and publishes as one [`crate::file::postings`] file; -//! [`PostingsArchive`](artifact::PostingsArchive) reopens the file over a whole-file mapping and -//! validates the artifact contract once, so lookups read from the page cache without holding -//! anything on the heap. +//! regions from it by inversion. Every gathered position-type pair is inserted into its type's +//! membership. Therefore both directions carry exactly the same relation. +//! +//! The postings publish as one [`crate::file::postings`] file. +//! [`PostingsArchive`](artifact::PostingsArchive) validates the artifact contract over a whole-file +//! mapping. Lookups borrow the mapped regions instead of copying the membership arrays onto the +//! heap. //! //! # Artifact contract //! diff --git a/libs/@local/graph/atlas/src/salt/postings/tests.rs b/libs/@local/graph/atlas/src/salt/postings/tests.rs index 79ba9fa1ae9..e576126c449 100644 --- a/libs/@local/graph/atlas/src/salt/postings/tests.rs +++ b/libs/@local/graph/atlas/src/salt/postings/tests.rs @@ -27,6 +27,11 @@ use crate::{ runs::Runs, }; +/// Creates a fresh per-process scratch directory under the system temp directory. +/// +/// # Panics +/// +/// This panics when the temporary path is not UTF-8 or the directory cannot be created. fn scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -39,10 +44,12 @@ fn scratch(name: &str) -> Utf8PathBuf { dir } +/// Converts a literal row number into an ontology id. fn id(row: u64) -> OntologyRowId { OntologyRowId::new(row) } +/// Builds a per-row type column from literal ontology row numbers. fn types(lists: &[&[u64]]) -> IdVec> { lists .iter() @@ -51,6 +58,10 @@ fn types(lists: &[&[u64]]) -> IdVec> { } /// Builds the dense set over `domain` positions admitting exactly `members`. +/// +/// # Panics +/// +/// This panics when a member lies outside `domain`. fn dense_set(domain: usize, members: &[u32]) -> Box> { let mut set = DenseBitSlice::new_empty(domain); for &member in members { @@ -64,31 +75,46 @@ fn le_posts(raw: &[u64]) -> IdVec> { IdVec::from_raw(raw.iter().copied().map(U64::new).collect()) } -/// Builds a lawful run structure for a fixture's regions. +/// Builds a validated run structure for a fixture's regions. +/// +/// # Panics +/// +/// This panics when the fenceposts violate the [`Runs`] contract. fn runs(posts: &[u64], items: Vec) -> Runs { Runs::from_parts(le_posts(posts), items).expect("the fixture posts satisfy the fencepost law") } -/// The hand fixture has eight rows over four types, gathered through a permutation. +/// The gather permutation for an eight-row fixture over four types. /// -/// Row-order direct types: `0:{0} 1:{0,2} 2:{1} 3:{2} 4:{0} 5:{1,2} 6:{0} 7:{0}`; -/// `row_of_position = [3, 1, 4, 0, 6, 2, 7, 5]`. Member positions per type, hand-derived: type 0 -/// `[1, 2, 3, 4, 6]`, type 1 `[5, 7]`, type 2 `[0, 1, 7]`, type 3 `[]`. Parents: `1 <- 0`, -/// `2 <- 0`, `3 <- {1, 2}`. +/// Row-order direct types: `0:{0} 1:{0,2} 2:{1} 3:{2} 4:{0} 5:{1,2} 6:{0} 7:{0}`. `row_of_position +/// = [3, 1, 4, 0, 6, 2, 7, 5]`. Member positions per type, hand-derived: type 0 `[1, 2, 3, 4, 6]`, +/// type 1 `[5, 7]`, type 2 `[0, 1, 7]`, type 3 `[]`. Parent-to-child edges: 0 → 1, 0 → 2, {1, 2} → +/// 3. /// -/// The split follows the size comparison alone: over eight points a dense set costs 16 bytes, so -/// five members (20 list bytes) go dense and three (12) stay a list. Type 0 is the fixture's dense -/// type. Every other type stays a list. +/// Over eight points a dense set costs 16 bytes. Type 0's five members cost 20 list bytes and +/// select the dense representation. The remaining lists use eight bytes for type 1, twelve for type +/// 2, and zero for type 3. const ROW_OF_POSITION: [u32; 8] = [3, 1, 4, 0, 6, 2, 7, 5]; +/// Builds the eight-row fixture type column. +/// +/// Type 0 occurs on five rows, type 1 on two, type 2 on three, and type 3 on none. fn fixture_types() -> IdVec> { types(&[&[0], &[0, 2], &[1], &[2], &[0], &[1, 2], &[0], &[0]]) } +/// Builds the fixture parent lists. +/// +/// Type 0 is the root, types 1 and 2 descend from it, and type 3 descends from both. fn fixture_parents() -> IdVec> { types(&[&[], &[0], &[0], &[1, 2]]) } +/// Writes `postings` to `name` under `dir` and reopens it as a validated mapped archive. +/// +/// # Panics +/// +/// This panics when the file cannot be written or reopened, or when artifact validation fails. fn mapped(dir: &Utf8PathBuf, name: &str, postings: &Postings) -> PostingsArchive { let path = dir.join(name); let mut file = fs::File::create(&path).expect("the fixture file should create"); @@ -101,7 +127,7 @@ fn mapped(dir: &Utf8PathBuf, name: &str, postings: &Postings) -> PostingsArchive .expect("the fixture postings should validate") } -/// The member-position count, over both encodings. +/// Counts member positions over either encoding. fn count(membership: &Membership<'_>) -> u64 { match membership { Membership::List(positions) => positions.len() as u64, @@ -109,6 +135,7 @@ fn count(membership: &Membership<'_>) -> u64 { } } +/// Collects member positions inside `range` in ascending order over either encoding. fn collect(membership: &Membership<'_>, range: core::ops::Range) -> Vec { let range = BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end); membership @@ -274,12 +301,9 @@ fn dense_iteration_crosses_word_boundaries() { assert_eq!(collect(&membership, 65..80), [79]); } -/// A membership whose list costs exactly the dense frame stays a list. -/// -/// Over five points a dense frame costs 16 bytes and so do four list members, so type 0 sits -/// exactly on the boundary the size comparison draws. The ruled tie keeps it a list, which reads -/// without bit decoding, while type 1's five members cost 20 list bytes and tip dense. A drift -/// from strict to inclusive comparison flips type 0 dense and fails both assertions. +// over five points a dense frame costs 16 bytes, equal to four list members. Type 0 is exactly at +// that boundary and keeps the list representation, which reads without bit decoding. Type 1's five +// members cost 20 list bytes and select the dense representation. #[test] fn equal_cost_membership_stays_a_list() { let dir = scratch("tie"); @@ -324,12 +348,10 @@ fn evidence_counts_the_split() { assert_eq!(evidence.direct_entries, 10); } -/// An eighty-point corpus drives a dense frame across the word boundary through the whole file. -/// -/// Type 0's eight members straddle position 64, so its frame holds two words and the path from -/// build through write to open exercises the multi-word header geometry, stride arithmetic, and -/// tail policing that the single-word fixtures never reach. Type 1 stays a two-member list whose -/// entries cross the same boundary. +// eighty points require two data words and one header word: 24 dense bytes. Type 0's eight members +// cost 32 list bytes and select a dense frame, including positions on both sides of 64. Type 1's +// two members cost eight list bytes and cross the same boundary. The final data word has 48 unused +// bits, which a single-word eight-point fixture never exercises at this stride. #[test] fn multi_word_dense_sets_roundtrip_through_the_file() { const MEMBERS: [u32; 8] = [0, 1, 62, 63, 64, 65, 78, 79]; @@ -415,8 +437,6 @@ fn build_rejects_out_of_domain_rows() { ); } -/// The gather asserts the dataset's ascent contract, so a defective stream fails the build -/// instead of publishing a file the next open refuses. #[test] #[should_panic(expected = "a row's direct types ascend strictly")] fn unsorted_direct_types_are_a_producer_bug() { @@ -427,7 +447,6 @@ fn unsorted_direct_types_are_a_producer_bug() { ); } -/// The parent regions assert the same contract for the type graph's stream. #[test] #[should_panic(expected = "a type's direct parents ascend strictly")] fn unsorted_parents_are_a_producer_bug() { @@ -467,7 +486,11 @@ fn empty_domains_roundtrip() { assert_matches!(membership, Membership::List(&[])); } -/// Writes lawful regions and returns the error their opening surfaces. +/// Writes regions with an artifact violation and returns the archive's rejection. +/// +/// # Panics +/// +/// This panics when file writing or opening fails, or when the archive accepts the regions. fn open_invalid(path: impl AsRef, regions: Regions<'_>) -> InvalidPostingsFile { let path = path.as_ref(); let mut file = fs::File::create(path).expect("the fixture file should create"); @@ -478,12 +501,15 @@ fn open_invalid(path: impl AsRef, regions: Regions<'_>) -> Inv .expect_err("the contract violation must surface") } -/// The helper writes lawful regions and overwrites one byte of the persisted file before returning -/// the error surfaced during reopening. +/// Overwrites one serialized byte and returns the archive's rejection. /// -/// The writer's fencepost columns arrive as validated [`Runs`], so a file with a broken fencepost -/// region can no longer be written; corrupting the bytes on disk is the remaining road to one, -/// and it is exactly the corruption class the open checks guard against. +/// The writer accepts validated [`Runs`]. Altering serialized bytes constructs invalid fenceposts +/// that this typed input cannot express. +/// +/// # Panics +/// +/// This panics when writing or opening the file fails, when `offset` lies outside the serialized +/// bytes, or when the archive accepts the altered regions. fn open_corrupted( path: impl AsRef, regions: Regions<'_>, @@ -507,8 +533,8 @@ fn open_rejects_membership_violations() { values.iter().copied().map(BasePosition::from_u32).collect() }; - // List posts not anchored at zero, then non-monotone. The writer's fencepost columns arrive - // as validated run structures, so these files exist only through byte corruption. + // invalid starts and non-monotone posts require byte corruption: the writer accepts only + // validated run structures. let flags = DenseBitSlice::new_empty(1); assert_eq!( open_corrupted( @@ -780,8 +806,7 @@ fn closure_expands_the_fixture_graph() { fn closure_rejects_parent_cycles() { let dir = scratch("cycle"); - // Types 0 and 1 parent each other. Type 2 stands free and - // settles, so exactly two types stay entangled. + // exactly types 0 and 1 form the cycle. Type 2 has no parent or child. let postings = Postings::build( &types(&[&[0]]), IdSlice::from_raw(&[0].map(NodeRowId::from_u32)), @@ -794,19 +819,12 @@ fn closure_rejects_parent_cycles() { assert_eq!(error.entangled, 2); } -/// The icon memo resolves the nearest icon-bearing ancestor, exactly where a request-time cache -/// went wrong. -/// -/// The graph is the counterexample that killed the cross-position icon cache: parents `1 <- 3`, -/// `3 <- {4, 5}`, `{2, 5} <- 0`, icons on 0 and 4. A walk from direct types `{1, 2}` finds 0's -/// icon after visiting 3, so a cache keyed by visited types would poison 3 with 0's icon - yet -/// 3's own nearest icon is 4's at depth one. The memo resolves each type over the whole graph, -/// so 3 reads 4. +/// Each type resolves its own nearest icon regardless of other types' traversal paths. #[test] fn icon_memo_resolves_the_nearest_ancestor_icon() { let dir = scratch("icon-memo"); - // Types 6 and 7 chain icon-free, so their cones record no source. + // types 6 and 7 form a separate icon-free chain. let postings = Postings::build( &types(&[&[1, 2], &[3]]), IdSlice::from_raw(&[0, 1].map(NodeRowId::from_u32)), @@ -821,7 +839,6 @@ fn icon_memo_resolves_the_nearest_ancestor_icon() { assert_eq!(closure.icon_source(id(0)), source(id(0), 0)); assert_eq!(closure.icon_source(id(1)), source(id(4), 2)); assert_eq!(closure.icon_source(id(2)), source(id(0), 1)); - // The cell the request-time cache poisoned: 3's nearest icon is 4's, not 0's. assert_eq!(closure.icon_source(id(3)), source(id(4), 1)); assert_eq!(closure.icon_source(id(4)), source(id(4), 0)); assert_eq!(closure.icon_source(id(5)), source(id(0), 1)); @@ -833,10 +850,9 @@ fn icon_memo_resolves_the_nearest_ancestor_icon() { /// Depth beats run order, and run order breaks equal-depth ties. /// -/// Type 3's earlier parent resolves deeper (0 through 1, depth two) than its later parent (2's -/// own icon, depth one), so the shallower source wins over the run order. Type 4's parents both -/// resolve at depth one, so the earlier parent in the run - ascending rows, the artifact -/// contract - decides. +/// Type 3 resolves to icon 2 at depth one despite its later parent position: the earlier parent 1 +/// reaches icon 0 at depth two. Type 4 reaches both icons at depth one and selects the earlier +/// parent in the ascending run, row 0. #[test] fn icon_memo_ties_resolve_by_depth_then_run_order() { let dir = scratch("icon-ties"); @@ -869,7 +885,13 @@ fn icon_memo_ties_resolve_by_depth_then_run_order() { ); } -/// Reference resolution: recurse over the raw parent lists with the memo's tie rule. +/// Resolves the nearest icon recursively over raw parent lists with the memo's tie rule. +/// +/// `parents` must be acyclic, with every parent reference inside its row domain. +/// +/// # Panics +/// +/// This panics when an inspected row lies outside `parents`. fn reference_icon_source( parents: &IdVec>, icons: &BTreeSet, @@ -899,10 +921,10 @@ fn reference_icon_source( /// The icon memo agrees with direct recursive resolution over random downward graphs. /// -/// Each type's parents draw from strictly smaller rows, so every graph is acyclic and every -/// parent run ascends by construction. The reference resolves each type recursively over the raw -/// lists with the same tie rule, so agreement pins the memo's topological pass and its archive -/// plumbing against an order-free restatement. +/// Strictly decreasing row ids cannot form a cycle. Every parent reference selects a smaller row +/// from a sorted set, making every generated graph acyclic and every parent run ascending. The +/// recursive reference uses raw lists with the same tie rule, independently of the memo's +/// topological order. #[property_test] fn icon_memo_matches_recursive_resolution( #[strategy = proptest::collection::vec(proptest::collection::btree_set(0_u64..12, 0..4), 1..12)] @@ -932,8 +954,8 @@ fn icon_memo_matches_recursive_resolution( let dir = scratch(&format!("icon-prop-{}", uuid::Uuid::now_v7())); let mapped = mapped(&dir, "icon-prop.post", &postings); - // The mapping keeps the unlinked file's bytes alive, so failing - // assertions cannot strand scratch files. + // on POSIX, the mapping retains access after unlink. Removing the directory before assertions + // prevents a failed assertion from retaining scratch files. fs::remove_dir_all(&dir).expect("the scratch directory is removable"); let closure = ClosureMap::new(&mapped, icons.iter().copied().map(OntologyRowId::new)) @@ -949,7 +971,11 @@ fn icon_memo_matches_recursive_resolution( } } -/// Reference membership: does `position`'s row carry `type_row` directly? +/// Tests whether `position`'s row carries `type_row` directly. +/// +/// # Panics +/// +/// This panics when `position` lies outside the permutation or its row lies outside `types`. fn reference_contains( types: &[SmallVec], row_of_position: &[u32], @@ -963,9 +989,9 @@ fn reference_contains( /// Built postings roundtrip through the file and agree with the row-order reference. /// -/// Agreement holds at every (type, position) pair. The size comparison picks each type's -/// representation from the drawn counts, so both representations recur across cases. Corpora -/// reach past one word of positions, so multi-word dense frames recur too. +/// Agreement holds at every (type, position) pair. Random member counts exercise both +/// representations under the byte-size comparison. Corpora extend past 64 positions to include +/// multi-word dense frames. #[property_test] fn built_postings_uphold_the_membership_contract( #[strategy = proptest::collection::vec(proptest::collection::btree_set(0_u64..5, 0..3), 0..100)] @@ -982,8 +1008,7 @@ fn built_postings_uphold_the_membership_contract( // A deterministic permutation other than the identity: reversal. let row_of_position: Vec = (0..points).rev().collect(); - // Parents point strictly downward, so the graph is acyclic by - // construction: a type's parent is its predecessor for odd rows. + // odd types have their even predecessor as parent. These disjoint one-edge chains are acyclic. let parents: IdVec> = (0..domain as u64) .map(|type_row| { if type_row & 1 == 1 { @@ -1012,14 +1037,14 @@ fn built_postings_uphold_the_membership_contract( let mapped = PostingsArchive::new(PostingsFile::open(&path).expect("the fixture file should open")) .expect("built postings always validate"); - // The mapping keeps the unlinked file's bytes alive, so failing - // assertions cannot strand scratch files. + // on POSIX, the mapping retains access after unlink. Removing the directory before assertions + // prevents a failed assertion from retaining scratch files. fs::remove_dir_all(&dir).expect("the scratch directory is removable"); prop_assert_eq!(mapped.points(), rows.len() as u64); - // The direct map restates each position's row types verbatim: the forward direction of the - // one relation the membership inverts, so agreement with the same reference pins both. + // check both directions against the same row-order relation: direct types here and per-type + // membership below. for position in 0..points { let expected: Vec = rows.as_raw() [row_of_position[position as usize] as usize] diff --git a/libs/@local/graph/atlas/src/salt/projector/artifact/mod.rs b/libs/@local/graph/atlas/src/salt/projector/artifact/mod.rs index af129a6171e..abb33c449fc 100644 --- a/libs/@local/graph/atlas/src/salt/projector/artifact/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/artifact/mod.rs @@ -1,5 +1,6 @@ -//! Checkpoint artifacts: the published model checkpoint, and the error vocabulary both checkpoint -//! flavours share. +//! Checkpoint artifacts. +//! +//! The published model checkpoint, and the error vocabulary both checkpoint flavours share. //! //! Both artifacts are burn's own named-MessagePack record format, written and parsed by the //! framework - the deliberate framework-parse exception to the crate's zerocopy mapping doctrine, @@ -12,7 +13,7 @@ //! on any backend for inference. It lives here as [`RecordedModel`] and [`open_model`]. The resume //! checkpoint is the fork point of the tuning protocol - the full training state at entry of the //! boundary step, from which a ladder segment resumes bit-equally on a deterministic backend - and -//! rides on the state it serializes, as +//! is defined on the state it serializes, as //! [`BoundaryState::write_checkpoint`](crate::salt::projector::train::BoundaryState::write_checkpoint) //! and //! [`BoundaryState::open_checkpoint`](crate::salt::projector::train::BoundaryState::open_checkpoint). @@ -123,7 +124,7 @@ impl From for CheckpointError { /// One recorded model checkpoint holding the framework's serialized bytes, ready to stage. /// /// The record-then-stage split keeps the two failure domains apart: recording fails only in the -/// framework's encoder while staging fails only in the writer, so neither error path has to +/// framework's encoder while staging fails only in the writer, and neither error path has to /// explain the other. Its writer marking admits the value as the published /// [`artifact::Projector`] entry. pub(crate) struct RecordedModel(Vec); @@ -131,14 +132,14 @@ pub(crate) struct RecordedModel(Vec); impl RecordedModel { /// Records the model's parameters as the checkpoint's byte form. /// - /// Consumes the model. Recording moves the parameters into the record, so a caller that - /// keeps its own copy clones at the call site where the copy is visible. + /// Consumes the model. Recording moves the parameters into the record. A caller that keeps + /// its own copy clones at the call site, where the copy is visible. /// /// # Errors /// /// Returns an error when the framework cannot encode the record. pub(crate) fn record(model: Projector) -> Result { - // Burn's "full" precision is f32 (as opposed to half); the model is f32 end to end, so + // Burn's "full" precision is f32 (as opposed to half). The model is f32 end to end, and // the recorder round-trips the parameters exactly. let recorder = NamedMpkBytesRecorder::::new(); let bytes = recorder.record(model.into_record(), ())?; diff --git a/libs/@local/graph/atlas/src/salt/projector/artifact/tests.rs b/libs/@local/graph/atlas/src/salt/projector/artifact/tests.rs index 323355408fd..51edcf27560 100644 --- a/libs/@local/graph/atlas/src/salt/projector/artifact/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/artifact/tests.rs @@ -3,7 +3,7 @@ //! Bit-exact round-trips across backends, and every open-path verification naming its failure. //! //! Forward-equality assertions are bit-exact by design. The record stores every f32 parameter at -//! full precision, so a round-tripped model must compute the identical function. Any deviation +//! full precision: a round-tripped model must compute the identical function, and any deviation //! breaks the round-trip rather than merely losing precision. use std::sync::LazyLock; @@ -22,6 +22,9 @@ use crate::{ salt::projector::model::{Architecture, Dimension, Layer, Projector, ProjectorInput}, }; +/// A small architecture for the checkpoint fixtures. +/// +/// Width 8, two residual blocks, six representation, four role and one condition dimension. fn architecture() -> Architecture { Architecture { width: nz!(8), @@ -32,8 +35,10 @@ fn architecture() -> Architecture { } } +/// The CPU device the checkpoint fixtures run on, resolved once. static DEVICE: LazyLock = LazyLock::new(|| Device::Cpu.pin(0).resolve()); +/// A fresh training projector of the fixture architecture initialized from `seed`. fn model(seed: u64) -> Projector { Projector::new( architecture(), @@ -69,6 +74,10 @@ fn probe>( .expect("projector outputs should convert to f32 values") } +/// Reopens a training-backend checkpoint on the inference backend and projects bit-identically. +/// +/// A model recorded on the training backend and reopened on the inference backend projects the +/// probe input bit-identically. #[test] fn model_checkpoint_round_trips_bit_exactly_across_backends() { let trained = model(7); @@ -86,6 +95,7 @@ fn model_checkpoint_round_trips_bit_exactly_across_backends() { ); } +/// Opening a checkpoint under a wider architecture fails with `Architecture` naming the mismatch. #[test] fn open_model_rejects_a_different_width() { let bytes = RecordedModel::record(model(7)) @@ -103,6 +113,10 @@ fn open_model_rejects_a_different_width() { assert_eq!(mismatch.actual, 8); } +/// A checkpoint opened under a deeper architecture fails before the record load. +/// +/// Opening a checkpoint under a deeper architecture fails with `Architecture` before the record +/// load, which would otherwise panic. #[test] fn open_model_rejects_a_different_depth_before_loading() { let bytes = RecordedModel::record(model(7)) @@ -111,9 +125,8 @@ fn open_model_rejects_a_different_depth_before_loading() { let mut deeper = architecture(); deeper.residual_blocks = nz!(3); - // A depth mismatch panics inside the framework's record zip, so - // this open returning an error at all certifies the pre-load - // check. + // A depth mismatch panics inside the framework's record zip: this open returning an error + // at all certifies the pre-load check. let error = open_model::(bytes.as_slice(), deeper, &*DEVICE) .expect_err("a depth mismatch should be rejected"); let CheckpointError::Architecture(mismatch) = error else { @@ -123,6 +136,7 @@ fn open_model_rejects_a_different_depth_before_loading() { assert_eq!(mismatch.dimension, Dimension::Depth); } +/// A checkpoint truncated to 100 bytes fails to open with `Record`. #[test] fn open_model_rejects_truncated_bytes() { let mut bytes = RecordedModel::record(model(7)) diff --git a/libs/@local/graph/atlas/src/salt/projector/band/enforce.rs b/libs/@local/graph/atlas/src/salt/projector/band/enforce.rs index 59e4181c5e3..fe604d6bc7f 100644 --- a/libs/@local/graph/atlas/src/salt/projector/band/enforce.rs +++ b/libs/@local/graph/atlas/src/salt/projector/band/enforce.rs @@ -13,7 +13,7 @@ use crate::math::{DNonNegative, DPositive, DVec2, DVec2x4T, Vec2}; /// One application's enforcement arithmetic, over the frozen constraint's derived readings. /// -/// The readings are copies taken from the projection itself, so no call site can pair them +/// The readings are copies taken from the projection itself, and no call site can pair them /// wrongly. The pass exists only after [`BandProjection::apply`]'s entry scan certifies every /// row finite, and its unchecked constructions consume that certificate. pub(super) struct EnforcementPass { @@ -46,8 +46,8 @@ impl EnforcementPass { /// Enforces one chunk, four rows at a time on SIMD lanes. /// - /// The widened distance kernel agrees bit for bit with the scalar form, so the batch and - /// remainder paths read identical displacements for identical rows. + /// The widened distance kernel agrees bit for bit with the scalar form. The batch and + /// remainder paths therefore read identical displacements for identical rows. pub(super) fn enforce_chunk( &self, rows: &mut [Vec2], @@ -99,8 +99,8 @@ impl EnforcementPass { /// Enforces one row from its widened squared displacement. /// - /// The running maximum updates before the clip test, so the record reads the pre-projection - /// displacement whether or not the radius binds. + /// The running maximum updates before the clip test, and the record therefore reads the + /// pre-projection displacement whether or not the radius binds. fn enforce_row( &self, row: &mut Vec2, @@ -112,7 +112,7 @@ impl EnforcementPass { let distance = square.sqrt(); // Proven finite: the entry scan and the freeze admit only finite rows and centres, and // the widest f32 displacement quotient by the smallest positive spread stays far inside - // the f64 range, so the re-entry needs no check. + // the f64 range. The re-entry needs no check. let normalized = (distance / self.spread_wide).finish_unchecked(); *maximum = (*maximum).max(normalized); @@ -150,12 +150,12 @@ pub(super) struct ChunkOutcome { } impl ChunkOutcome { - /// Rows this chunk clipped. + /// Returns the rows this chunk clipped. pub(super) const fn clipped(&self) -> u64 { self.clipped } - /// The chunk's largest normalized overshoot. + /// Returns the chunk's largest normalized overshoot. pub(super) const fn overshoot(&self) -> DNonNegative { self.overshoot } diff --git a/libs/@local/graph/atlas/src/salt/projector/band/mod.rs b/libs/@local/graph/atlas/src/salt/projector/band/mod.rs index 58739d0aa33..3352542d969 100644 --- a/libs/@local/graph/atlas/src/salt/projector/band/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/band/mod.rs @@ -1,5 +1,7 @@ -//! The per-row band projection enforces family (ii)'s constitutive constraint and keeps the -//! record that makes its non-binding claim evidence. +//! The per-row band projection over the zero field. +//! +//! It enforces the estimand's constitutive constraint on the zero field and keeps the record that +//! makes its non-binding claim evidence. //! //! The estimand is declared subject to `‖x₀(n) − x₀^ref(n)‖ ≤ band` for every node row `n` - //! each row of the live zero-condition field may move at most `band` world units from its @@ -8,15 +10,16 @@ //! global. Per row rather than in RMS because an RMS ball leaves a fixed-cardinality attack set //! per-row room that grows as `√N` with the corpus. The bound is enforced by //! projection rather than penalized: at every enforcement point a row past the radius moves -//! back to the ball around its own reference position, so the field the loss reads never exists -//! outside the constraint. The radius is the same-frame reconstruction `band = β · s_ref(Z_K)` - -//! `β` is the stage's declared dimensionless value (the target's `β_proj`, or a calibration -//! cohort's `β_cal`, assigned by the schedule, not by this module) and `s_ref` is the boundary -//! field's own RMS spread, so nothing but a dimensionless number ever crosses generations. +//! back to the ball around its own reference position, and the field the loss reads therefore +//! never exists outside the constraint. The radius is the same-frame reconstruction `band = β · +//! s_ref(Z_K)` - `β` is the stage's declared dimensionless value (the target's `β_proj`, or a +//! calibration cohort's `β_cal`, assigned by the schedule, not by this module) and `s_ref` is the +//! boundary field's own RMS spread, and nothing but a dimensionless number ever crosses +//! generations. //! //! The whole-field application is the only witness of what it censored in the constitutive -//! field, so that enforcing operation maintains the run's [`EnforcementRecord`] as it applies -//! and is the record's one writer. The record accumulates the clipped row-application count +//! field. That enforcing operation therefore maintains the run's [`EnforcementRecord`] as it +//! applies and is the record's one writer. The record accumulates the clipped row-application count //! (whose positivity is the `ever_clipped` bit - a clip is exactly a moved row, so within the //! record the bit and the count cannot disagree), the maximum pre-projection overshoot of the //! enforced radius in units of `s_ref`, and each row's running maximum normalized displacement @@ -30,7 +33,7 @@ //! The constraint also binds row values read through another realization of the same field. //! [`BandProjection::project`] applies the identical clip law to one such value and records //! nothing. On a backend whose kernels vary with the execution shape, two realizations of one -//! row can read different bytes, so a value near the radius can clip in the per-row form while +//! row can read different bytes, and a value near the radius can clip in the per-row form while //! the recorded field reads unclipped. That disagreement is the design: the record describes //! the constitutive field alone. In a run whose objective reads a second realization, a clean //! `ever_clipped` therefore does not certify that the objective ran unconstrained. The per-row @@ -40,9 +43,8 @@ //! //! The record's honesty rests on the arithmetic. A clipped row is placed at `band − margin` //! rather than exactly at the radius, with `margin` sized at the freeze to dominate every -//! narrowing -//! error of the stored f32 coordinates - that makes the projection idempotent in the stored -//! precision, so a parked row re-reads strictly inside and cannot re-clip on the next +//! narrowing error of the stored f32 coordinates. That makes the projection idempotent in the +//! stored precision: a parked row re-reads strictly inside and cannot re-clip on the next //! application to inflate the record by rounding alone. A freeze whose margin would consume the //! radius refuses, since at that magnitude the stored precision cannot represent the //! constraint's own boundary. And a non-finite row refuses before any byte moves: divergence @@ -85,36 +87,39 @@ const MARGIN_SCALE: DPositive = d_positive!(1.0 / 4_194_304.0); /// The landing margin's absolute floor, `2⁻¹⁴⁰`. /// /// Subnormal f32 components carry an absolute narrowing error up to `2⁻¹⁵⁰` regardless of the -/// extent, so an extent-scaled margin alone underestimates the error of a map whose coordinates -/// sit near the bottom of the f32 range. The floor keeps the idempotence argument valid there. +/// extent. An extent-scaled margin alone therefore underestimates the error of a map whose +/// coordinates sit near the bottom of the f32 range, and the floor keeps the idempotence argument +/// valid there. const MARGIN_FLOOR: DPositive = d_positive!(7.174_648_137_343_064e-43); /// The headroom the radius must keep over the landing margin: `margin · 1024 ≤ band`. /// -/// A clipped row lands within one `1024`th of the radius, so the landing stays a projection +/// A clipped row lands within one `1024`th of the radius, and the landing stays a projection /// onto the boundary rather than a shrink toward the centre. Without the headroom the stored /// f32 coordinates cannot express the constraint's boundary around the snapshot, and the /// freeze refuses. const MARGIN_HEADROOM: DPositive = d_positive!(1024.0); -/// The frozen constraint holds the projection centre `x₀^ref` beside its reconstructed radius. +/// The frozen constraint, pairing the projection centre `x₀^ref` with its reconstructed radius. /// /// The state is minimal: one radius and one margin. Every derived reading - the widened /// radius, its exact square, the landing radius, the widened spread - is an accessor over -/// them, so no cached projection of the radius can disagree with its source. +/// them, and no cached projection of the radius can disagree with its source. #[derive(Debug, PartialEq)] pub(crate) struct BandProjection { /// The boundary snapshot's zero field, each row's projection centre. centre: Box>, /// `β`: the stage's declared dimensionless radius. dimensionless_radius: Positive, - /// `s_ref(Z_K)`: the boundary field's RMS spread, the frame's unit carrier and the - /// normalizer of every record reading. + /// `s_ref(Z_K)`: the boundary field's RMS spread. + /// + /// The frame's unit carrier and the normalizer of every record reading. reference_spread: Positive, /// `band = β · s_ref` in the working precision: the enforced radius. radius: Positive, - /// The landing margin, sized at the freeze to dominate every narrowing error of the - /// stored f32 coordinates. + /// The landing margin. + /// + /// Sized at the freeze to dominate every narrowing error of the stored f32 coordinates. margin: DPositive, } @@ -147,7 +152,7 @@ where "the boundary snapshot should cover at least one row" ); - // The narrowed f32 product is the enforced radius; the refusal carries the exact + // The narrowed f32 product is the enforced radius. The refusal carries the exact // widened value. let radius = dimensionless_radius.checked_mul(reference_spread); let radius_exact = dimensionless_radius.mul_wide(reference_spread); @@ -190,15 +195,17 @@ where DPositive::from(self.radius) } - /// Returns the radius's f64 square, exact because an f32 significand squares within 53 - /// bits. The clip predicate compares squared displacements against it. + /// Returns the radius's f64 square, exact because an f32 significand squares within 53 bits. + /// + /// The clip predicate compares squared displacements against it. #[inline] const fn radius_squared(&self) -> DPositive { self.radius.square_wide() } - /// Returns the reference spread widened to f64, exactly. Every normalized reading divides - /// by it. + /// Returns the reference spread widened to f64, exactly. + /// + /// Every normalized reading divides by it. #[inline] const fn spread_wide(&self) -> DPositive { // Positive with no check: the exact widening of a positive f32 stays positive. @@ -207,12 +214,12 @@ where /// Returns `band − margin`, where a clipped row lands. /// - /// The landing sits one narrowing allowance inside the radius, so a clipped row re-reads - /// strictly inside and the projection is idempotent in the stored precision. + /// The landing sits one narrowing allowance inside the radius. A clipped row therefore + /// re-reads strictly inside, and the projection is idempotent in the stored precision. #[inline] const fn landing_radius(&self) -> DPositive { // Positive with no check: the freeze's headroom keeps the margin at or below a - // 1024th of the radius, so the landing keeps at least 1023/1024 of it. + // 1024th of the radius, and the landing keeps at least 1023/1024 of it. DPositive::new_unchecked(self.radius_wide() - self.margin) } @@ -230,10 +237,11 @@ where /// /// Every row's pre-projection displacement updates its running maximum first, then a row /// whose displacement exceeds the radius moves to the landing radius along its own - /// direction from the centre. Untouched rows keep their exact bytes, so an unclipped + /// direction from the centre. Untouched rows keep their exact bytes, and an unclipped /// application leaves the field bit-identical - the coincidence the non-binding claim /// reads. Rows enforce in parallel over fixed chunks, and the partial reductions combine by - /// integer sum and maximum, so the result is bit-deterministic under any thread schedule. + /// integer sum and maximum. The result is therefore bit-deterministic under any thread + /// schedule. /// /// The field and the record share the centre's row domain, and enforcement points arrive /// in step order - wiring contracts checked in debug builds, since all three come from one @@ -291,8 +299,9 @@ where &self.centre } - /// Consumes the projection into its centre, without a copy, for the evidence record that - /// outlives the constraint. + /// Consumes the projection into its centre, without a copy. + /// + /// For the evidence record that outlives the constraint. #[must_use] pub(crate) fn into_centre(self) -> Box> { self.centre @@ -392,15 +401,15 @@ where /// /// The floor sits two margins inside the radius. A clipped row lands one margin inside, at /// the landing radius, and re-reads within one narrowing allowance of it, and the margin - /// dominates that allowance by construction, so every clipped-in-place row's squared - /// displacement stays at or above this floor. The saturation reading that consumes it - /// therefore counts every row the projection is actively holding, together with unclipped + /// dominates that allowance by construction. Every clipped-in-place row's squared + /// displacement therefore stays at or above this floor, and the saturation reading that + /// consumes it counts every row the projection is actively holding, together with unclipped /// rows within two margins of the boundary. #[inline] #[must_use] pub(crate) fn saturation_floor_squared(&self) -> DPositive { // Positive with no check: the freeze's headroom keeps the margin at or below a 1024th - // of the radius, so the floor keeps at least 1022/1024 of it and stays positive. + // of the radius, and the floor keeps at least 1022/1024 of it and stays positive. let floor = (self.landing_radius() - self.margin).get(); // Total: the floor is at most the f32-born radius and at least 1022/1024 of a radius diff --git a/libs/@local/graph/atlas/src/salt/projector/band/record.rs b/libs/@local/graph/atlas/src/salt/projector/band/record.rs index 7c212533bdb..8f6b786a446 100644 --- a/libs/@local/graph/atlas/src/salt/projector/band/record.rs +++ b/libs/@local/graph/atlas/src/salt/projector/band/record.rs @@ -10,7 +10,7 @@ use crate::math::{DNonNegative, DPositive, DVec2, Vec2}; /// The clip moves a row along `x ↦ centre + landing·u(x)` with `u` the unit displacement from /// the centre, whose Jacobian at the pre-projection position is `factor·(I − uuᵀ)` with /// `factor = landing/‖x − centre‖`: the radial component of a perturbation dies and the -/// tangential component scales down to the landing sphere. The matrix is symmetric, so the +/// tangential component scales down to the landing sphere. The matrix is symmetric, and the /// transpose the chain rule needs is the matrix itself. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct ClipJacobian { @@ -25,7 +25,7 @@ impl ClipJacobian { /// /// Returns the landed value with the applied derivative where `square` exceeds /// `radius_squared`, and [`None`] where the squared displacement sits at or inside it. - /// Every projection path shares this predicate and this landing arithmetic, so a value + /// Every projection path shares this predicate and this landing arithmetic, and a value /// clips identically wherever it is read. /// /// The readings must be one consistent set: `square` is the widened squared distance of @@ -55,7 +55,7 @@ impl ClipJacobian { let displacement = DVec2::from(value) - DVec2::from(centre); // In `(0, 1)` on this branch: the distance exceeds the radius and the landing sits - // strictly inside it, so the quotient of two finite positives is finite and positive. + // strictly inside it. The quotient of two finite positives is finite and positive. let factor = DPositive::new_unchecked((landing_radius / distance).into_raw()); let jacobian = Self { direction: displacement / distance, @@ -85,21 +85,25 @@ impl ClipJacobian { /// every [`BandProjection::apply`](super::BandProjection::apply), and /// never reset: the type has no operation that shrinks a field, which is the no-reset rule made /// structural. The calibration protocol derives a run's non-binding verdict from these readings, -/// so every one is taken pre-projection by the enforcing operation itself. +/// and every one is taken pre-projection by the enforcing operation itself. #[derive(Debug, PartialEq)] pub(crate) struct EnforcementRecord { - /// `u(n)` per row: the running maximum normalized pre-projection displacement over every - /// application so far. + /// `u(n)` per row: the running maximum normalized pre-projection displacement. + /// + /// The maximum runs over every application so far. row_maxima: Box>, /// The cumulative count of row-applications the projection moved. clipped_row_applications: u64, - /// The largest excess of any row's pre-projection normalized displacement over the enforced - /// radius. Zero while nothing has bound. + /// The largest excess of any row's pre-projection normalized displacement over the radius. + /// + /// Zero while nothing has bound. max_overshoot: DNonNegative, /// The boundary step the record opened at, which starts the accumulation interval. opened_at: usize, - /// The last enforcement point applied, or [`None`] before the first. Together with - /// [`opened_at`](Self::opened_at) this is the interval's persisted endpoint pair. + /// The last enforcement point applied, or [`None`] before the first. + /// + /// Together with [`opened_at`](Self::opened_at) this is the interval's persisted endpoint + /// pair. last_application: Option, } @@ -144,8 +148,8 @@ where /// Returns whether any projection application moved any row. /// - /// Derived from the count: a clip is exactly a moved row, so the bit and the count cannot - /// disagree. + /// Derived from the count: a clip is exactly a moved row, and the bit and the count therefore + /// cannot disagree. #[inline] #[must_use] pub(crate) const fn ever_clipped(&self) -> bool { @@ -173,8 +177,9 @@ where &self.row_maxima } - /// Consumes the record into its per-row maxima, without a copy, for the evidence record - /// that closes over it. + /// Consumes the record into its per-row maxima, without a copy. + /// + /// For the evidence record that closes over it. #[must_use] pub(crate) fn into_row_maxima(self) -> Box> { self.row_maxima diff --git a/libs/@local/graph/atlas/src/salt/projector/band/refusal.rs b/libs/@local/graph/atlas/src/salt/projector/band/refusal.rs index 9d588396370..8514f85b723 100644 --- a/libs/@local/graph/atlas/src/salt/projector/band/refusal.rs +++ b/libs/@local/graph/atlas/src/salt/projector/band/refusal.rs @@ -1,7 +1,7 @@ //! The freeze refuses a constraint the stored precision cannot carry. //! -//! The live field arrives as a proven-finite point field, so divergence refuses at the field's own -//! construction, and what remains here is the constraint that does not exist over the stored +//! The live field arrives as a proven-finite point field, and divergence refuses at the field's +//! own construction. What remains here is the constraint that does not exist over the stored //! coordinates: a radius outside the working precision, or an extent past the finite range. A //! radius below the landing margin's headroom refuses the same way. @@ -9,28 +9,31 @@ use core::{error::Error, fmt}; use crate::math::{DPositive, Positive}; -/// An invalid constraint refuses before training, and every freeze-time failure is this one -/// refusal class. +/// The band freeze's one refusal class, carrying the failed check's reading. /// -/// The declared constraint does not exist over the stored coordinates, so no fit starts. The -/// variants carry the failed check's reading and nothing branches on them: there is no -/// degraded mode. +/// The declared constraint does not exist over the stored coordinates. The freeze measures the +/// boundary field, which exists only after the opening segment has trained, and its refusal ends +/// the run before the target phase starts. The variants carry the failed check's reading and +/// nothing branches on them: there is no degraded mode. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum BandRefusal { - /// The reconstructed radius `β · s_ref` is not a strictly positive f32, so the constraint - /// has no enforceable size in the working precision. + /// The reconstructed radius `β · s_ref` is not a strictly positive f32. + /// + /// The constraint has no enforceable size in the working precision. RadiusOutOfDomain { /// The product in double precision, where it is exact. radius: DPositive, }, - /// The snapshot's coordinate extent plus the radius leaves the finite f32 range, so a - /// projected row could narrow to infinity. + /// The snapshot's coordinate extent plus the radius leaves the finite f32 range. + /// + /// A projected row could narrow to infinity. RepresentationCeiling { /// The extent as measured, in double precision. extent: DPositive, }, - /// The radius sits below the landing margin's headroom, so the stored precision cannot - /// represent the constraint's boundary around the snapshot. + /// The radius sits below the landing margin's headroom. + /// + /// The stored precision cannot represent the constraint's boundary around the snapshot. RepresentationFloor { /// The enforced radius in the working precision. radius: Positive, diff --git a/libs/@local/graph/atlas/src/salt/projector/band/tests.rs b/libs/@local/graph/atlas/src/salt/projector/band/tests.rs index 6991cfa9d16..06338730191 100644 --- a/libs/@local/graph/atlas/src/salt/projector/band/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/band/tests.rs @@ -1,8 +1,8 @@ //! Certificates for the band projection and its enforcement record. //! -//! Dyadic fixtures make every pre-projection reading exact - distances, normalized maxima, and -//! overshoots are exactly representable - the record therefore asserts exact contracts. The landing -//! point itself carries the documented margin, so its assertions bound rather than pin. +//! Dyadic fixtures make every pre-projection reading exact (distances, normalized maxima and +//! overshoots are exactly representable), and the record therefore asserts exact contracts. The +//! landing point itself carries the documented margin: its assertions bound rather than pin. #![expect( clippy::float_cmp, @@ -28,13 +28,15 @@ fn field(points: &mut [Vec2]) -> &mut FinitePointField { FinitePointField::new_unchecked_mut(IdSlice::from_raw_mut(points)) } +/// Builds a [`Positive`] from a literal test value. fn positive(value: f32) -> Positive { Positive::new(value).expect("test value is positive") } -/// The centres reach an extent of `8.5`, making the margin `17 · 2⁻²³`. That clears the -/// headroom bar with room to spare, and the freeze admits the radius `0.5` (`β = 0.25`, -/// `s_ref = 2`). +/// The centres reach an extent of `8.5`, making the margin `17 · 2⁻²³`. +/// +/// That clears the headroom bar with room to spare, and the freeze admits the radius `0.5` +/// (`β = 0.25`, `s_ref = 2`). const CENTRES: [Vec2; 6] = [ Vec2::new(0.0, 0.0), Vec2::new(8.0, 0.0), @@ -44,9 +46,11 @@ const CENTRES: [Vec2; 6] = [ Vec2::new(8.0, 8.0), ]; -/// One live row per centre. Rows 0 and 4 sit past the radius (displacements `1.25` and -/// `0.625`), row 1 sits exactly on it (legal, unclipped), and the rest sit inside. The batch -/// path covers rows 0 through 3 and the remainder path rows 4 and 5, with a clip on each. +/// One live row per centre. +/// +/// Rows 0 and 4 lie past the radius (displacements `1.25` and `0.625`), row 1 lies exactly on it +/// (legal, unclipped), and the rest lie inside. The four-lane batch path covers rows 0 through 3 +/// and the remainder path rows 4 and 5, with a clip on each. const LIVE: [Vec2; 6] = [ Vec2::new(0.75, 1.0), Vec2::new(8.0, 0.5), @@ -56,6 +60,7 @@ const LIVE: [Vec2; 6] = [ Vec2::new(8.25, 8.0), ]; +/// The band projection frozen over the fixture centres with inner radius `0.25` and outer `2.0`. fn fixture() -> BandProjection { BandProjection::freeze( boxed_field(Box::new(CENTRES)), @@ -65,13 +70,16 @@ fn fixture() -> BandProjection { .expect("the fixture is a valid constraint") } +/// The Euclidean distance from `row` to `centre`, accumulated in `f64`. fn displacement(row: Vec2, centre: Vec2) -> f64 { let along_x = f64::from(row.x()) - f64::from(centre.x()); let along_y = f64::from(row.y()) - f64::from(centre.y()); along_x.hypot(along_y) } -/// The freeze lands every declared constant exactly, and a fresh record is born zero over the +/// Reproduces every declared constant at the freeze and opens a fresh record at zero. +/// +/// The freeze reproduces every declared constant exactly, and a fresh record opens at zero over the /// whole row domain. #[test] fn freezes_the_reconstruction_and_opens_a_zero_record() { @@ -97,9 +105,10 @@ fn freezes_the_reconstruction_and_opens_a_zero_record() { ); } -/// One application clips exactly the two out-of-band rows and records their exact -/// pre-projection readings. Every in-band row keeps its bytes, the row exactly on the radius -/// included. +/// Clips exactly the two out-of-band rows and keeps every in-band row's bytes. +/// +/// One application clips exactly the two out-of-band rows and records their exact pre-projection +/// readings. Every in-band row keeps its bytes, the row exactly on the radius included. #[test] fn enforces_per_row_and_records_the_pre_projection_readings() { let projection = fixture(); @@ -141,15 +150,17 @@ fn enforces_per_row_and_records_the_pre_projection_readings() { let post_x = f64::from(live[row].x()) - f64::from(centre.x()); let post_y = f64::from(live[row].y()) - f64::from(centre.y()); // Narrowing the landed point moves each component by up to half an f32 ulp of its own - // magnitude (about `5e-7` at this fixture's coordinates near 8), so the cross reading + // magnitude (about `5e-7` at this fixture's coordinates near 8): the cross reading // bounds rather than pins. let cross = pre_y.mul_add(-post_x, pre_x * post_y).abs(); assert!(cross <= 1e-6, "row {row} left its direction by {cross}"); } } +/// Moves nothing on a second application over the projected field. +/// /// A second application over the projected field moves nothing: the landing margin keeps every -/// clipped row strictly inside, so re-enforcement cannot inflate the record by rounding. +/// clipped row strictly inside, and re-enforcement therefore cannot inflate the record by rounding. #[test] fn a_projected_field_re_reads_as_inside() { let projection = fixture(); @@ -163,8 +174,8 @@ fn a_projected_field_re_reads_as_inside() { projection.apply(field(&mut live), 10, &mut record); assert_eq!(live, projected); - // The once-clipped rows re-read as interior through the per-row form too, so a deposit - // after this application composes through the identity. + // The once-clipped rows re-read as interior through the per-row form too: a deposit after + // this application composes through the identity. for (row, &value) in projected.iter().enumerate() { let (unmoved, clip) = projection.project(NodeRowId::from_usize(row), value); assert_eq!(unmoved, value, "row {row}"); @@ -176,8 +187,10 @@ fn a_projected_field_re_reads_as_inside() { assert_eq!(record.row_maxima().as_raw(), &*maxima_after_first); } -/// The running maxima keep a mid-run excursion that returns before the end: the reading a -/// radius would censor, visible although the final position sits inside. +/// Keeps a mid-run excursion in the running maxima after the row returns inside. +/// +/// The running maxima keep a mid-run excursion that returns before the end: the reading a radius +/// would censor, visible although the final position lies inside. #[test] fn running_maxima_keep_the_transient_excursion() { let projection = BandProjection::freeze( @@ -210,8 +223,10 @@ fn running_maxima_keep_the_transient_excursion() { assert_eq!(record.last_application(), Some(6)); } -/// A non-finite row refuses at the field's construction, naming the smallest offender, so a -/// diverged field never reaches enforcement and the record stays untouched. +/// Refuses a non-finite row at the field's construction and leaves the record untouched. +/// +/// A non-finite row refuses at the field's construction, naming the smallest offender: a diverged +/// field never reaches enforcement, and the record stays untouched. #[test] fn non_finite_row_refuses_at_construction() { let projection = fixture(); @@ -235,9 +250,10 @@ fn non_finite_row_refuses_at_construction() { ); } -/// The per-row projection and the whole-field application share one clip law. Clip decisions -/// match row for row and a clipped row reads identical bytes through both, with the derivative -/// tied to the landing the row actually took. +/// The per-row projection and the whole-field application share one clip law. +/// +/// Clip decisions match row for row and a clipped row reads identical bytes through both, with the +/// derivative tied to the landing the row actually took. #[test] fn the_per_row_projection_matches_the_application_row_for_row() { let projection = fixture(); @@ -281,8 +297,10 @@ fn the_per_row_projection_matches_the_application_row_for_row() { ); } -/// The clip Jacobian is the exact derivative of the applied map: on an axis-aligned clip the -/// radial force dies to exactly zero and a tangential force scales by exactly the factor. +/// Matches the clip Jacobian to the exact derivative on an axis-aligned clip. +/// +/// The clip Jacobian is the exact derivative of the applied map: on an axis-aligned clip the radial +/// force dies to exactly zero and a tangential force scales by exactly the factor. #[test] fn the_clip_jacobian_kills_radial_force_exactly() { let projection = BandProjection::::freeze( @@ -312,8 +330,10 @@ fn the_clip_jacobian_kills_radial_force_exactly() { assert!((landing - f64::from(landed.y())).abs() < 1e-6); } -/// The Jacobian matches finite differences of the map it claims to differentiate, on a -/// generic non-axis-aligned clip. +/// Matches the Jacobian to finite differences on a generic non-axis-aligned clip. +/// +/// The Jacobian matches finite differences of the map it claims to differentiate, on a generic +/// non-axis-aligned clip. #[test] fn the_clip_jacobian_matches_the_map_derivative() { let projection = fixture(); @@ -321,7 +341,7 @@ fn the_clip_jacobian_matches_the_map_derivative() { let (_, clip) = projection.project(NodeRowId::new(0), LIVE[0]); // Row 0: centre (0, 0), pre-clip position (0.75, 1.0), ‖d‖ = 1.25 exactly. The mirror - // states the applied map with the landing the state itself names, so the derivative under + // states the applied map with the landing the state itself names: the derivative under // test is the map's own. let state = clip.expect("row 0 clipped"); let landing = state.factor * 1.25; @@ -365,6 +385,8 @@ fn the_clip_jacobian_matches_the_map_derivative() { } } +/// Refuses an overflowing and an underflowing radius product, each carrying its exact value. +/// /// The reconstructed radius must be a strictly positive f32: an overflowing product and an /// underflowing one both refuse, each carrying the exact double-precision value. #[test] @@ -396,10 +418,12 @@ fn a_radius_outside_the_value_domain_refuses() { ); } -/// An extent past the finite f32 range is refused, because a projected row there could narrow -/// to infinity. The radius must be commensurate with the largest centre to trip it - a smaller -/// excess is absorbed by the f64 sum, and an absorbable excess sits provably below the half-ulp -/// that narrowing to `f32::MAX` tolerates. +/// Refuses an extent past the finite f32 range with a radius commensurate with the centre. +/// +/// An extent past the finite f32 range is refused, because a projected row there could narrow to +/// infinity. The radius must be commensurate with the largest centre to trip it - a smaller excess +/// is absorbed by the f64 sum, and an absorbable excess lies provably below the half-ulp that +/// narrowing to `f32::MAX` tolerates. #[test] fn an_extent_past_the_finite_range_refuses() { let refused = BandProjection::::freeze( @@ -417,6 +441,8 @@ fn an_extent_past_the_finite_range_refuses() { ); } +/// Refuses a radius below the landing margin's headroom at centres of `2²⁰`. +/// /// A radius below the landing margin's headroom refuses: centres at `2²⁰` and a radius of `128` /// leave the margin `2⁻²² · (2²⁰ + 128)`, whose headroom `256.03125` exceeds the radius. #[test] @@ -439,9 +465,11 @@ fn a_radius_below_the_margin_headroom_refuses() { ); } -/// Near the bottom of the f32 range the absolute margin floor takes over from the extent scale: -/// a subnormal radius of `2⁻¹³⁵` refuses against the floor's headroom `2⁻¹³⁰`, which the -/// extent-scaled margin alone would have admitted. +/// Refuses a subnormal radius under the absolute margin floor near the bottom of the range. +/// +/// Near the bottom of the f32 range the absolute margin floor takes over from the extent scale: at +/// centres of `2⁻¹²⁰` the extent-scaled margin would be about `2⁻¹⁴²`, below the floor `2⁻¹⁴⁰`, and +/// a subnormal radius of `2⁻¹³⁵` refuses carrying the floor's headroom `2⁻¹³⁰`. #[test] fn the_margin_floor_binds_for_subnormal_radii() { let refused = BandProjection::::freeze( @@ -466,8 +494,9 @@ fn the_margin_floor_binds_for_subnormal_radii() { ); } -/// The saturation floor recovers the freeze's exact margin and sits two of them inside the -/// radius. Every quantity in this fixture is dyadic, so the assert is an exact contract. +/// The saturation floor recovers the freeze's exact margin and lies two of them inside the radius. +/// +/// Every quantity in this fixture is dyadic, and the assert is therefore an exact contract. #[test] fn saturation_floor_sits_two_margins_inside_the_radius() { let projection = fixture(); diff --git a/libs/@local/graph/atlas/src/salt/projector/bench/live.rs b/libs/@local/graph/atlas/src/salt/projector/bench/live.rs index 048db3eeba5..073fca8d1d6 100644 --- a/libs/@local/graph/atlas/src/salt/projector/bench/live.rs +++ b/libs/@local/graph/atlas/src/salt/projector/bench/live.rs @@ -5,14 +5,14 @@ //! burn tensor work (forward, surrogate, backward, optimizer), how much is the crate's hand-rolled //! field evaluation, and how much is the CPU batch pipeline (draw, assemble, input //! materialization). The same decomposition prices the batch pipeline's allocator arena and any -//! per-phase optimization argument, so one fixture feeds three decisions. +//! per-phase optimization argument, and one fixture therefore feeds three decisions. //! //! [`Fixture::build`] synthesizes a corpus at the trainer's shape: a symmetric semantic graph, //! typed relation instances over sixteen relations, unit-norm representations, local scales, a //! landmark pool, and a mined frame produced by the real miner over a synthetic coordinate frame. //! Draws run the production [`BatchSampler`] at the ratified [`BatchPlan`] with every family -//! populated (relation at the lens's active extreme), so a measured step carries the full composite -//! objective, not a placeholder loss. +//! populated (relation at the lens's active extreme). A measured step therefore carries the full +//! composite objective, not a placeholder loss. //! //! Values are synthetic and costs are real. Every phase runs the production code path with the //! production types, and the numbers mean shape and traversal, never convergence. @@ -62,7 +62,8 @@ use crate::{ /// The relation-type count of the synthetic corpus. /// -/// Comfortably above the ratified per-step draw of twelve, so type selection stays a real draw. +/// Comfortably above the ratified per-step draw of twelve, which keeps type selection a real +/// draw. const RELATION_TYPES: usize = 16; /// The landmark pool size the ratified draw of 512 samples from. @@ -70,14 +71,23 @@ const LANDMARK_POOL: usize = 4096; /// One synthesized corpus at the trainer's live shape. pub struct Fixture { + /// The corpus row count. rows: usize, + /// The symmetric semantic graph. graph: SemanticGraph, + /// The attraction and protection indexes over the synthetic relation instances. indexes: RelationIndexes, + /// The unit-norm representations, one row per corpus row. representations: MatrixN, + /// The cycling node roles, one per corpus row. roles: Vec, + /// The synthetic local scales, one per corpus row. scales: LocalScales, + /// The landmark anchor pool. landmarks: Vec>, + /// The mined hard negatives over a synthetic coordinate frame. mined: MinedFrame, + /// The ratified batch plan the draws run at. plan: BatchPlan, } @@ -97,7 +107,9 @@ impl Assembled { /// The production sampler bound over the fixture, opaque to the bench target. pub struct Sampler<'fixture> { + /// The production sampler over the fixture's graph and indexes. sampler: BatchSampler<'fixture, NodeRowId, EdgeRowId>, + /// The fixture the draws read their mined frame and landmarks from. fixture: &'fixture Fixture, } @@ -129,8 +141,8 @@ impl Fixture { let plan = crate::salt::fit::ProjectorOptions::ratified().plan; // The mined frame comes from the production miner over a synthetic - // coordinate frame: pooled hard negatives at the real quota, so the - // hard family draws and evaluates at its live shape. + // coordinate frame: pooled hard negatives at the real quota. The + // hard family therefore draws and evaluates at its live shape. let coordinates: Vec = core::iter::repeat_with(|| { Vec2::new( rng.random_range(-1.0..=1.0_f32), @@ -224,16 +236,20 @@ impl Sampler<'_> { /// The per-backend training state. struct Live> { + /// The training-decorated model, taken out for the optimizer step and put back. model: Option>>, + /// The Adam optimizer over the model's parameters. optimizer: burn::optim::adaptor::OptimizerAdaptor< burn::optim::Adam, Projector>, Autodiff, >, + /// The device every tensor phase runs on. device: B::Device, } impl> Live { + /// Builds the seeded model and a fresh Adam optimizer on `device`. fn build(device: B::Device, seed: u64) -> Self { Self { model: Some(Projector::new( @@ -246,6 +262,11 @@ impl> Live { } } + /// Materializes the batch's model input on the device and waits for the queue to drain. + /// + /// # Panics + /// + /// Panics when the device fails to complete its queue. fn input(&self, batch: &Assembled, evaluation: &Evaluation<'_, NodeRowId>) { let input = batch .0 @@ -254,6 +275,7 @@ impl> Live { B::sync(&self.device).expect("the measured device should complete its queue"); } + /// Runs the training-path forward and reads back the output sum, dropping the recorded graph. fn forward(&self, batch: &Assembled, evaluation: &Evaluation<'_, NodeRowId>) -> f32 { let model = self.model.as_ref().expect("the model is always present"); let input = batch @@ -262,6 +284,7 @@ impl> Live { model.forward(input).sum().into_scalar() } + /// Runs the plain-backend forward of the validation model and reads back the output sum. fn refresh(&self, batch: &Assembled, evaluation: &Evaluation<'_, NodeRowId>) -> f32 { let model = self .model @@ -272,6 +295,14 @@ impl> Live { model.forward(input).sum().into_scalar() } + /// Evaluates the full composite objective and returns its loss total. + /// + /// The return waits for the device queue. + /// + /// # Panics + /// + /// Panics when the objective leaves the finite domain or the device fails to complete its + /// queue. fn objective(&self, batch: &Assembled, evaluation: &Evaluation<'_, NodeRowId>) -> f32 { let model = self.model.as_ref().expect("the model is always present"); let mut metrics = BudgetBreakdown::default(); @@ -284,6 +315,14 @@ impl> Live { total } + /// Runs one training step and returns the loss total. + /// + /// The step is objective, backward and optimizer, and the return waits for the device queue. + /// + /// # Panics + /// + /// Panics when the objective leaves the finite domain or the device fails to complete its + /// queue. fn step(&mut self, batch: &Assembled, evaluation: &Evaluation<'_, NodeRowId>) -> f32 { let model = self.model.take().expect("the model is always present"); let mut metrics = BudgetBreakdown::default(); @@ -304,7 +343,9 @@ impl> Live { /// context - columns, numerical contract, decile axis - binds once at build, as the session binds /// it once per run. The timed phases never pay setup. pub struct Stepper<'fixture> { + /// The model, optimizer, and device. live: Live, + /// The evaluation context bound once at build. evaluation: Evaluation<'fixture, NodeRowId>, } @@ -328,26 +369,30 @@ impl<'fixture> Stepper<'fixture> { } } - /// Materializes the batch's model input on the device, fenced. + /// Materializes the batch's model input on the device, then waits for the device queue. pub fn input(&self, batch: &Assembled) { self.live.input(batch, &self.evaluation); } - /// Runs the training-path forward (autodiff graph recorded), fenced by a scalar readback. + /// Runs the training-path forward with the autodiff graph recorded. + /// + /// A scalar readback synchronizes it. + /// + /// This drops the recorded graph without running backward. + /// + /// # Advice /// - /// This drops the recorded graph unconsumed: no backward ever runs. On a pooled asynchronous - /// device a tight loop of these outruns buffer reclamation and exhausts memory, so the - /// decomposition phases are a synchronous-backend instrument; the production forward motion is - /// [`refresh`](Self::refresh). + /// Use this decomposition phase only on the CPU backend. For production-style refresh, use + /// [`refresh`](Self::refresh), which runs the validation model on the plain backend. #[must_use] pub fn forward(&self, batch: &Assembled) -> f32 { self.live.forward(batch, &self.evaluation) } - /// Runs the refresh forward on the plain backend, fenced by a scalar readback. + /// Runs the refresh forward on the plain backend, synchronized by a scalar readback. /// - /// This records no autodiff graph: it is the per-step refresh motion as production performs it, - /// safe to loop on any backend. + /// This runs the validation model without recording an autodiff graph, as production refresh + /// does. #[must_use] pub fn refresh(&self, batch: &Assembled) -> f32 { self.live.refresh(batch, &self.evaluation) @@ -356,8 +401,8 @@ impl<'fixture> Stepper<'fixture> { /// Runs input, forward, and the full composite objective, returning the loss total. /// /// This is everything a step does before its backward pass. It covers the readback, the - /// hand-rolled budget-family fields, the clip, the surrogate construction, and the support - /// terms. + /// hand-rolled field evaluation, the budget measurement, the surrogate construction, and the + /// support terms. #[must_use] pub fn objective(&self, batch: &Assembled) -> f32 { self.live.objective(batch, &self.evaluation) diff --git a/libs/@local/graph/atlas/src/salt/projector/bench/mod.rs b/libs/@local/graph/atlas/src/salt/projector/bench/mod.rs index 7d533ce562f..6377c2cf9c4 100644 --- a/libs/@local/graph/atlas/src/salt/projector/bench/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/bench/mod.rs @@ -9,8 +9,8 @@ //! //! This module synthesizes batches at the corpus shape the trainer feeds: unit-norm 512-wide //! representations, mixed roles, a width-1 `[eta]` condition. The backward pass drives a -//! mean-coordinate loss; gradient values are meaningless, but the traversal is the full autodiff -//! graph the composite objective shares, so its wall time is the decision's number. +//! mean-coordinate loss. Gradient values are meaningless, but the traversal is the full autodiff +//! graph the composite objective shares, and its wall time is the decision's number. use burn::{ DispatchDevice, @@ -32,7 +32,7 @@ const ARCHITECTURE: Architecture = Architecture::default(); /// One synthesized batch at the trainer's input shape. /// -/// Holds the raw columns; tensors materialize per run so device transfer and graph construction +/// Holds the raw columns. Tensors materialize per run so device transfer and graph construction /// stay inside the timed region, exactly as they recur per training step. pub struct Batch { rows: usize, @@ -136,6 +136,7 @@ pub struct Model { } impl Model { + /// Builds models with matching parameters from one seed on the chosen device. #[must_use] pub fn build(device: PinnedDevice, seed: u64) -> Self where @@ -150,6 +151,9 @@ impl Model { } } + /// Runs the plain forward pass and returns the output sum. + /// + /// The sum's readback synchronizes the device. pub fn forward(&self, batch: &Batch) -> f32 { let output = self .projector diff --git a/libs/@local/graph/atlas/src/salt/projector/bench/tests.rs b/libs/@local/graph/atlas/src/salt/projector/bench/tests.rs index 5ca7005b2e2..344e556c65f 100644 --- a/libs/@local/graph/atlas/src/salt/projector/bench/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/bench/tests.rs @@ -9,6 +9,7 @@ use rand_xoshiro::Xoshiro256PlusPlus; use super::{Batch, Model}; use crate::device::Device; +/// A synthetic batch of `rows` rows from the fixed seed 7. fn batch(rows: usize) -> Batch { Batch::new::(rows, 7) } @@ -30,6 +31,11 @@ fn batches_are_unit_norm_and_deterministic() { } } +/// Compares the bench model's forward `sum / count` with its forward-backward mean. +/// +/// Both are finite and agree as `sum / count` within `1e-4`. +/// +/// The count is the output's element count: 16 rows of planar coordinates give 32. #[test] fn autodiff_numerical_stability() { let model = Model::build::(Device::Cpu.pin(0), 42); @@ -46,10 +52,12 @@ fn autodiff_numerical_stability() { ); } -/// The live fixture draws, assembles, and steps every phase to finite numbers. +/// Runs the live fixture's four phases to finite numbers. +/// +/// The fixture draws, assembles, and runs the input, forward, objective and step phases. /// -/// A small corpus keeps the smoke fast; the phases exercised are exactly the ones the bench target -/// times, so a fixture defect fails here instead of in a wall-time run. +/// A small corpus keeps the smoke test fast. The phases are the ones the bench target times, apart +/// from `refresh`, and a fixture defect in them therefore fails here instead of in a wall-time run. #[test] fn live_fixture_steps_every_phase() { let fixture = super::live::Fixture::build(256, 11); diff --git a/libs/@local/graph/atlas/src/salt/projector/budget/mod.rs b/libs/@local/graph/atlas/src/salt/projector/budget/mod.rs index c59e957d076..7a8b44ceace 100644 --- a/libs/@local/graph/atlas/src/salt/projector/budget/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/budget/mod.rs @@ -9,11 +9,14 @@ //! ratio = ‖relation‖ / baseline //! ``` //! -//! The floor keeps the baseline positive where the semantic gradient vanishes, so the recorded -//! ratios stay finite and comparable across runs. +//! The floor keeps the baseline positive where the semantic gradient vanishes. The ratios +//! accumulate in double precision, where every quotient of a finite `f32` norm by a positive `f32` +//! baseline is finite, and they stay comparable across runs. The narrowed `f32` mean an accessor +//! returns can still overflow ([`BudgetSummary::mean_ratio`]). //! -//! The relation gradients re-enter the parameter graph through [`surrogate`]: one backward pass -//! through the returned scalar deposits exactly the requested per-node coordinate gradient. +//! The combined hand-gradient field, semantic and relation together, re-enters the parameter +//! graph through [`surrogate`]: one backward pass through the returned scalar deposits exactly +//! the requested per-node coordinate gradient. #[cfg(test)] mod tests; @@ -63,8 +66,8 @@ pub(crate) struct BudgetOutcome { /// Streaming aggregation of budget outcomes for the training metrics. /// -/// One summary aggregates the nodes recorded into it; the training loop keeps one per reporting -/// bucket (overall, per relation type, per degree decile) and records each node's outcome into +/// One summary aggregates the nodes recorded into it. The training metrics keep one per reporting +/// bucket (overall, per relation type, per degree decile) and record each node's outcome into /// every bucket it belongs to. The summary accumulates the ratio mean in double precision. #[derive(Debug, Default)] pub(crate) struct BudgetSummary { @@ -113,6 +116,15 @@ impl BudgetSummary { } /// Returns the mean relation-to-baseline norm ratio. + /// + /// # Warning + /// + /// The mean accumulates in `f64` and narrows to `f32` here, and the narrowing can overflow. + /// The maximum is not the cutoff. A double mean at or above `f32::MAX` and below + /// `f32::MAX + 2¹⁰³` (half an `f32` ulp above the maximum) rounds down to the finite + /// `f32::MAX`. A mean at or beyond that tie returns as `+∞`. Valid inputs reach the overflow: a + /// zero semantic gradient against the floor `2⁻¹⁴⁹` and a unit relation gradient give the + /// finite double ratio `2¹⁴⁹`, whose narrowing overflows. #[expect( clippy::cast_precision_loss, reason = "node counts stay far below the f64 integer bound" @@ -143,10 +155,11 @@ impl BudgetSummary { /// /// Its backward pass carries the per-node coordinate gradients into the model parameters. /// -/// The returned value is `Σ_i ⟨coordinates[i], gradient[i]⟩`: its gradient with respect to -/// `coordinates` is exactly `gradient`, so a single backward pass propagates the caller's per-node -/// vectors through the projector's Jacobian. `gradient` lives on the inner backend and enters the -/// graph as a constant - the model cannot differentiate through the hand-gradient field. +/// The returned value is `Σ_i ⟨coordinates[i], gradient[i]⟩`. Its gradient with respect to +/// `coordinates` is exactly `gradient`. Therefore a single backward pass propagates the caller's +/// per-node vectors through the projector's Jacobian. `gradient` lives on the inner backend and +/// enters the graph as a constant - the model cannot differentiate through the hand-gradient +/// field. pub(crate) fn surrogate( coordinates: Tensor, gradient: Tensor, diff --git a/libs/@local/graph/atlas/src/salt/projector/budget/tests.rs b/libs/@local/graph/atlas/src/salt/projector/budget/tests.rs index 3f8a5b41999..aa16927cc47 100644 --- a/libs/@local/graph/atlas/src/salt/projector/budget/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/budget/tests.rs @@ -1,7 +1,7 @@ //! Certificates for the budget diagnostics and the gradient surrogate. //! //! The measurement assertions are bit-exact where every intermediate is dyadic. The surrogate -//! certificates establish the seam the training loop depends on: one backward pass through the +//! certificates establish the contract the training loop depends on: one backward pass through the //! surrogate deposits exactly the requested coordinate gradient, both at a detached coordinate leaf //! and through the full model Jacobian. @@ -22,8 +22,13 @@ use crate::{ salt::projector::model::{Architecture, Projector, ProjectorInput}, }; +/// The CPU device the surrogate tensors live on, resolved once. static DEVICE: LazyLock = LazyLock::new(|| Device::Cpu.pin(0).resolve()); +/// Records both norms in `measure` and lets the floor bind under a small semantic gradient. +/// +/// `measure` records the semantic and relation norms and takes the semantic norm as baseline unless +/// it falls under the floor, which then binds, including for a vanished gradient. #[test] fn measure_records_the_baseline_convention() { let budget = Budget { @@ -63,6 +68,7 @@ fn summary_reports_hand_computed_ratios() { assert_eq!(summary.mean_ratio(), Some(8.03125)); } +/// A fresh summary has no nodes and no mean ratio. #[test] fn summary_is_empty_before_any_record() { let summary = BudgetSummary::new(); @@ -71,6 +77,7 @@ fn summary_is_empty_before_any_record() { assert_eq!(summary.mean_ratio(), None); } +/// Backpropagating the surrogate at a coordinate leaf deposits the requested gradient bit for bit. #[test] fn surrogate_deposits_exactly_the_requested_gradient_at_a_leaf() { let device = &*DEVICE; @@ -98,7 +105,7 @@ fn surrogate_deposits_exactly_the_requested_gradient_at_a_leaf() { /// Nudges every parameter off its initialization. /// /// The identity-contract layers initialize to zero and would block gradient flow into the deep -/// block parameters, leaving the surrogate certificate comparing zeros with zeros; a deterministic +/// block parameters, leaving the surrogate certificate comparing zeros with zeros. A deterministic /// ramp makes every parameter's gradient generically nonzero. struct Perturb; @@ -122,9 +129,8 @@ impl ModuleMapper for Perturb { let shape = tensor.shape(); let device = tensor.device(); let ramp = Tensor::from_data(TensorData::new(ramp, shape), &device); - // The sum is an interior autodiff node; re-rooting it as a - // required-gradient leaf is what lets gradients accumulate at - // the perturbed parameter. + // The sum is an interior autodiff node. Re-rooting it as a required-gradient leaf is + // what lets gradients accumulate at the perturbed parameter. Param::from_mapped_value(id, (tensor + ramp).detach().require_grad(), mapper) } } @@ -149,6 +155,7 @@ impl ModuleVisitor for GradientCollector<'_> { } } +/// Collects every parameter's gradient from `gradients` keyed by parameter id, in id order. fn parameter_gradients( model: &Projector, gradients: &::Gradients, @@ -161,13 +168,16 @@ fn parameter_gradients( collector.collected } +/// Reproduces the direct backpropagation gradient through the surrogate on all 23 parameters. +/// +/// Handing the detached coordinate gradient of `Σ y²` to the surrogate produces the same gradient +/// on all 23 trainable parameters as backpropagating the loss through the model directly. #[test] fn surrogate_matches_ordinary_autodiff_through_the_model() { - // Reference: L(y) = sum(y · y) has coordinate gradient 2 · y. Path - // A backpropagates L through the model directly; path B evaluates - // the same coordinate gradient detached and hands it to the - // surrogate. Equal parameter gradients certify that one surrogate - // backward deposits J^T g for the full FiLM-residual Jacobian. + // Reference: L(y) = Σ y·y has coordinate gradient 2·y. Path A backpropagates L through the + // model directly. Path B evaluates the same coordinate gradient detached and hands it to + // the surrogate. Equal parameter gradients certify that one surrogate backward deposits + // Jᵀ·g for the full FiLM-residual Jacobian. let device = &*DEVICE; let architecture = Architecture { width: nz!(8), diff --git a/libs/@local/graph/atlas/src/salt/projector/evidence/mod.rs b/libs/@local/graph/atlas/src/salt/projector/evidence/mod.rs index 671b55b2fa8..2a93c325118 100644 --- a/libs/@local/graph/atlas/src/salt/projector/evidence/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/evidence/mod.rs @@ -1,10 +1,10 @@ //! The per-evaluation evidence the target objective must emit. //! -//! The estimand reads through frozen references, so every published claim about a fit rests on -//! readings that let a later audit rebuild the frame arithmetic. Each evaluation assembles one -//! [`EvaluationEvidence`]. The record holds the live alignment beside the +//! The estimand reads through frozen references. Every published claim about a fit therefore +//! rests on readings that let a later audit rebuild the frame arithmetic. Each evaluation +//! assembles one [`EvaluationEvidence`]. The record holds the live alignment beside the //! reference-configuration pair `s_K`/`r_K` - the same closed-form fit with the boundary -//! snapshot `Z_K` in the zero slot. The whole-corpus alignment rides beside them, since its +//! snapshot `Z_K` in the zero slot. The record also holds the whole-corpus alignment, since its //! composition with the gauge fit is the frame bridge, and the zero-field common-mode //! similarity onto `Z_K` reads the uniform mode the per-row band admits. The affine component //! on the gauge population follows with its normalized residual. The gauge displacement @@ -14,7 +14,7 @@ //! corpus is the alignment channel's signature, and it is readable exactly because the sequence //! survives. //! -//! The generation-level constants ride once in a [`RulerIdentity`], so every reading stays +//! The generation-level constants appear once in a [`RulerIdentity`], and every reading stays //! reproducible against the exact ruler that produced it. Everything here is aggregate, and no //! pair or row identity persists in any record. @@ -47,27 +47,30 @@ hashql_core::id::newtype! { pub(crate) struct StratumId(u32) } -/// A refused evidence reading names the fit that could not be made. +/// The fit a refused evidence reading could not make. /// /// Every variant leaves the evaluation without its declared evidence, and an evaluation that /// cannot state its evidence publishes nothing. Nothing branches on the variants and there is no /// partial record. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum EvidenceRefusal { - /// The whole-corpus alignment fit refused, over coincident canonical rows or an exactly - /// cancelling covariance. Non-finite coordinates never reach the reading: the readback - /// boundary refuses them naming the row. + /// The whole-corpus alignment fit refused. + /// + /// The cause is coincident canonical rows or an exactly cancelling covariance. Non-finite + /// coordinates never reach the reading: the readback boundary refuses them naming the row. CorpusAlignment, /// The zero-field common-mode fit onto the boundary snapshot refused. ZeroCommonMode, /// The gauge similarity fit over the whole-field anchor constellations refused. Gauge, - /// The reference-configuration fit of the canonical gauge rows onto the frozen `Z_K` - /// anchors refused. + /// The reference-configuration fit refused. + /// + /// The fit places the canonical gauge rows onto the frozen `Z_K` anchors. ReferenceConfiguration, - /// The affine fit over the gauge population refused: the anchors' canonical scatter is - /// degenerate beyond what the similarity fit tolerates, since an affine solve needs both - /// axes of its source. + /// The affine fit over the gauge population refused. + /// + /// The anchors' canonical scatter is degenerate beyond what the similarity fit tolerates, since + /// an affine solve needs both axes of its source. Affine, } @@ -92,7 +95,7 @@ impl Error for EvidenceRefusal {} /// The declared and measured scalar constants of one ruler freeze. /// -/// The fields are the scalar constants every per-evaluation reading normalizes against, so a +/// The fields are the scalar constants every per-evaluation reading normalizes against, and a /// reading replayed later resolves against the exact ruler that produced it. The trainer fills /// the record at the freeze, where each source object is in hand. The boundary field and the /// ruler's two tables travel beside this record as typed run evidence, and the writer that @@ -103,8 +106,9 @@ pub(crate) struct RulerIdentity { pub boundary_step: usize, /// `s_ref(Z_K)`: the boundary field's frozen spread, the frame's unit carrier. pub reference_spread: Positive, - /// `spread_G(Z_K)`: the gauge anchors' frozen spread, the denominator of every normalized - /// residual. + /// `spread_G(Z_K)`: the gauge anchors' frozen spread. + /// + /// The denominator of every normalized residual. pub gauge_spread: Positive, /// `ε_rel`: the declared dimensionless regularizer. pub epsilon_rel: Positive, @@ -112,8 +116,9 @@ pub(crate) struct RulerIdentity { pub epsilon_abs: Positive, /// `β_proj`: the declared dimensionless projection radius. pub dimensionless_radius: Positive, - /// `band_proj = β_proj · s_ref(Z_K)`: the enforced world-unit radius, recorded beside its - /// dimensionless source so the freeze-time domain check stays auditable. + /// `band_proj = β_proj · s_ref(Z_K)`: the enforced world-unit radius. + /// + /// Recorded beside its dimensionless source so the freeze-time domain check stays auditable. pub radius: Positive, } @@ -156,7 +161,7 @@ impl EnforcementSummary { /// The gauge displacement family of one covariate stratum. /// /// The displacement is each anchor's world-unit zero-field distance from its frozen `Z_K` -/// position. The family keeps the control evidence's exact aggregate shape, so the two collateral +/// position. The family keeps the control evidence's exact aggregate shape, and the two collateral /// readings compare like for like. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct DisplacementStratum { @@ -195,38 +200,46 @@ pub(crate) struct EvaluationEvidence { /// The objective-shape fitted scale, the `s` the step's estimand actually descended. /// /// Its forwards run in the pass's own padded shape, which an autotuned backend may realize - /// differently from the whole fields, so this reading stands alone and bridges nothing. + /// differently from the whole fields. This reading therefore stands alone and bridges nothing. pub objective_scale: Positive, /// The objective-shape fit's normalized residual. pub objective_residual: DNonNegative, /// `n_eff_G`: the effective anchor count over duplicate classes. pub effective_count: DNonNegative, - /// The gauge similarity, canonical onto zero over the whole-field realization: one end of - /// the frame bridge. + /// The gauge similarity, canonical onto zero over the whole-field realization. + /// + /// One end of the frame bridge. pub gauge_similarity: Similarity, - /// `s_K`: the scale of the canonical gauge rows fitted onto the frozen `Z_K` anchors, the - /// gauge-path envelope's first input. The live `s` need not equal it once the zero field - /// has moved, because the live fit reads current against current and this fit reads - /// current against frozen. + /// `s_K`: the scale of the canonical gauge rows fitted onto the frozen `Z_K` anchors. + /// + /// The gauge-path envelope's first input. The live `s` need not equal it once the zero field + /// has moved, because the live fit reads current against current and this fit reads current + /// against frozen. pub reference_scale: Positive, - /// `r_K`: the reference-configuration fit's normalized residual, the envelope's second - /// input. + /// `r_K`: the reference-configuration fit's normalized residual. + /// + /// The envelope's second input. pub reference_residual: DNonNegative, - /// The whole-corpus similarity, canonical onto zero: the published field's alignment and - /// the frame bridge's other end. + /// The whole-corpus similarity, canonical onto zero. + /// + /// The published field's alignment and the frame bridge's other end. pub corpus_similarity: Similarity, - /// `s_z` with its rotation and translation: the similarity fit of the live zero field onto - /// `Z_K`. A uniform shrink of every row is per-row legal and invisible to the band, and - /// `|log s_z|` reads exactly that common mode. + /// `s_z` with its rotation and translation: the live zero field's similarity fit onto `Z_K`. + /// + /// A uniform shrink of every row is per-row legal and invisible to the band, and `|log s_z|` + /// reads exactly that common mode. pub zero_similarity: Similarity, - /// The fitted affine component on the gauge population, canonical onto zero. Its - /// non-similarity part carries the anisotropic deformation the residual `r` prices. + /// The fitted affine component on the gauge population, canonical onto zero. + /// + /// Its non-similarity part carries the anisotropic deformation the residual `r` prices. pub affine: Transform, - /// The affine fit's root-mean-square residual, normalized against the frozen gauge spread: - /// the movement no affine map explains. + /// The affine fit's root-mean-square residual, normalized against the frozen gauge spread. + /// + /// The movement no affine map explains. pub affine_residual: DNonNegative, - /// The gauge displacement families, one per populated covariate stratum in ascending - /// stratum order. + /// The gauge displacement families. + /// + /// One per populated covariate stratum in ascending stratum order. pub displacement: Vec, /// The saturation tallies, one per populated covariate stratum in ascending stratum order. pub saturation: Vec, @@ -238,7 +251,8 @@ pub(crate) struct EvaluationEvidence { /// /// The gauge constellation and the band constraint are frozen at the boundary, the strata ride /// the admitted inputs, and the enforcement record accumulates through the run. A reading -/// borrows them as one value, so the reference set the evidence derives from is named once. +/// borrows them as one value, and the reference set the evidence derives from is therefore named +/// once. pub(crate) struct EvidenceReferences<'run, N> { /// The frozen gauge constellation. pub anchors: &'run GaugeAnchors, @@ -253,11 +267,11 @@ pub(crate) struct EvidenceReferences<'run, N> { impl EvaluationEvidence { /// Assembles one evaluation's evidence reading. /// - /// The live fields arrive proven from their readback boundaries, so every reading - - /// the whole-field fits and each per-row family - derives from proven-finite + /// The live fields arrive proven from their readback boundaries. Every reading - the + /// whole-field fits and each per-row family - therefore derives from proven-finite /// coordinates. The gauge, reference-configuration, and affine fits read the anchor - /// constellations gathered from those same fields, so every end of the recorded frame - /// bridge derives from one field realization and the composition is exact on it. The + /// constellations gathered from those same fields. Every end of the recorded frame bridge + /// therefore derives from one field realization, and the composition is exact on it. The /// objective-shape fit arrives as its own recorded reading and enters no bridge. The /// displacement and saturation families fold per row in one serial pass each - the reading /// runs per evaluation, far off the per-step enforcement path. @@ -307,8 +321,8 @@ impl EvaluationEvidence { let gauge_spread = anchors.frozen_spread().widen(); // Total: an rms residual over f32-born fields stays below ~1.2e87 by its own totality - // theorem, and the smallest positive f32 spread is 2⁻¹⁴⁹, so a normalized residual - // sits more than five hundred exponent shells inside the f64 range. + // theorem, and the smallest positive f32 spread is 2⁻¹⁴⁹. A normalized residual + // therefore lies more than five hundred exponent shells inside the f64 range. let normalized = |rms: DNonNegative| (rms / gauge_spread).finish_unchecked(); let gauge = Similarity::fit_uniform_par(&canonical_anchors, &zero_anchors) @@ -332,8 +346,8 @@ impl EvaluationEvidence { - DVec2::from(references.projection.centre()[row])) .norm_squared(); // Finite with no check. The field proofs above certified every coordinate - // finite, and a widened f32 difference squares within the f64 range, so the root - // is finite too. + // finite, and a widened f32 difference squares within the f64 range. The root is + // therefore finite too. families .entry(references.strata[row]) .or_default() diff --git a/libs/@local/graph/atlas/src/salt/projector/evidence/tests.rs b/libs/@local/graph/atlas/src/salt/projector/evidence/tests.rs index 1ace232fd5d..2d0880d794b 100644 --- a/libs/@local/graph/atlas/src/salt/projector/evidence/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/evidence/tests.rs @@ -1,9 +1,9 @@ //! Certificates for the per-evaluation evidence reading. //! -//! Dyadic fixtures land every asserted reading on an exactly representable value, from the -//! fitted scales through the displacement quantiles, so each assert is an exact contract. The -//! clip-then-read certificate exercises the one place production arithmetic rounds - a clipped -//! row's landing coordinate - and asserts the boolean the floor was sized for. +//! Dyadic fixtures make every asserted reading exactly representable, from the fitted scales +//! through the displacement quantiles, and each assert is therefore an exact contract. The +//! clip-then-read certificate exercises the one place production arithmetic rounds, a clipped +//! row's landing coordinate, and asserts the boolean the saturation floor was sized for. #![expect( clippy::float_cmp, @@ -23,13 +23,15 @@ use crate::{ }, }; +/// Builds a [`Positive`] from a literal test value. fn positive(value: f32) -> Positive { Positive::new(value).expect("test value is positive") } -/// The frame bridge, gauge frame to corpus frame: undoing the gauge fit and applying the corpus -/// fit. Composition can leave the representable coefficient range, in which case the two recorded -/// ends remain the complete evidence. +/// Composes the frame bridge from the gauge frame to the corpus frame. +/// +/// It undoes the gauge fit and applies the corpus fit. Composition can leave the representable +/// coefficient range, in which case the two recorded ends remain the complete evidence. fn bridge(evidence: &EvaluationEvidence) -> Option { evidence .gauge_similarity @@ -37,9 +39,10 @@ fn bridge(evidence: &EvaluationEvidence) -> Option { .then(evidence.corpus_similarity) } -/// The boundary snapshot is an exact square of gauge anchors at rows 1, 2, 4, 5 with two far -/// fillers, centred so every fitted translation vanishes. The anchors' frozen spread is exactly -/// 1, so normalized residuals read unscaled. +/// The boundary snapshot is an exact square of gauge anchors with two far fillers. +/// +/// The anchors are at rows 1, 2, 4, 5, centred so every fitted translation vanishes. The anchors' +/// frozen spread is exactly 1: normalized residuals read unscaled. const SNAPSHOT: [Vec2; 6] = [ Vec2::new(8.0, 8.0), Vec2::new(1.0, 0.0), @@ -49,8 +52,9 @@ const SNAPSHOT: [Vec2; 6] = [ Vec2::new(0.0, -1.0), ]; -/// Rows 0 through 2 sit in stratum 0 and rows 3 through 5 in stratum 1, so each stratum holds -/// one filler and two anchors. +/// Rows 0 through 2 form stratum 0 and rows 3 through 5 stratum 1. +/// +/// Each stratum holds one filler (rows 0 and 3) and two anchors. const STRATA: [StratumId; 6] = [ StratumId::new(0), StratumId::new(0), @@ -65,6 +69,7 @@ fn frame(points: &[Vec2]) -> &FinitePointField { FinitePointField::new_unchecked(IdSlice::from_raw(points)) } +/// The gauge anchors frozen over the square fixture's rows 1, 2, 4 and 5 with one class each. fn gauge() -> GaugeAnchors { GaugeAnchors::freeze( Box::new([1, 2, 4, 5].map(NodeRowId::new)), @@ -86,6 +91,7 @@ fn projection(dimensionless_radius: f32) -> BandProjection { .expect("the fixture is a valid constraint") } +/// Fits the gauge live over the anchor rows of the canonical and zero fields. fn live_fit( anchors: &GaugeAnchors, canonical: &[Vec2; 6], @@ -100,10 +106,13 @@ fn live_fit( .expect("the fixture fits") } +/// The field with every point scaled by `factor`. fn scaled(field: &[Vec2; 6], factor: f32) -> [Vec2; 6] { field.map(|point| Vec2::new(point.x() * factor, point.y() * factor)) } +/// Separates a uniform shrink of the zero field exactly in every reading. +/// /// A uniform shrink of the whole zero field is per-row legal away from the fillers, and every /// reading separates it exactly: the live scale leaves the reference-configuration scale, the /// common-mode fit reads the shrink as `s_z = 2`, the anchors' displacement family reads the @@ -136,8 +145,8 @@ fn a_common_shrink_separates_every_reading() { assert_eq!(evidence.effective_count, DNonNegative::from_usize(4)); // The whole-field gauge fit reads current against current: canonical 2x onto zero 0.5x. - // The objective-shape fit read the same constellations in this fixture, so its own - // recorded reading agrees. + // The objective-shape fit read the same constellations in this fixture as the whole-field + // fit. assert_eq!(evidence.scale.get(), 0.25); assert_eq!(evidence.residual, 0.0); assert_eq!(evidence.objective_scale.get(), 0.25); @@ -146,7 +155,7 @@ fn a_common_shrink_separates_every_reading() { assert_eq!(evidence.reference_scale.get(), 0.5); assert_eq!(evidence.reference_residual, 0.0); - // The whole corpus moved as one similarity, so both bridge ends agree and the bridge is + // The whole corpus moved as one similarity: both bridge ends agree and the bridge is // exactly the identity. assert_eq!(evidence.corpus_similarity, evidence.gauge_similarity); assert_eq!(bridge(&evidence), Some(Similarity::IDENTITY)); @@ -156,7 +165,7 @@ fn a_common_shrink_separates_every_reading() { assert_eq!(evidence.zero_similarity.scale().get(), 2.0); assert_eq!(evidence.zero_similarity.translation(), Vec2::new(0.0, 0.0)); - // The gauge constellation stayed similar, so the affine component is the plain scale. + // The gauge constellation stayed similar, and the affine component is the plain scale. assert_eq!( evidence.affine, Transform::from_scale(Vec2::new(0.25, 0.25)) @@ -175,7 +184,7 @@ fn a_common_shrink_separates_every_reading() { } // The anchors stay far inside the radius 2 band. Each stratum's far filler moved by the - // illegal 4·sqrt(2) and reads saturated. The per-row constraint sees the fillers alone, + // illegal 4√2 and reads saturated. The per-row constraint sees the fillers alone, // while the common-mode scale above reads the shrink itself. assert_eq!(evidence.saturation.len(), 2); for (tally, stratum) in evidence.saturation.iter().zip([0, 1]) { @@ -192,6 +201,8 @@ fn a_common_shrink_separates_every_reading() { assert_eq!(evidence.enforcement.last_application, None); } +/// Splits an anisotropic gauge deformation exactly between the similarity and affine fits. +/// /// An anisotropic gauge deformation splits the decomposition exactly: the similarity residual /// prices the deformation at `r = 0.75` while the affine fit absorbs it whole, and the /// reference-configuration fit stays at the frozen identity. @@ -240,9 +251,11 @@ fn the_affine_component_absorbs_what_the_similarity_prices() { } } -/// A non-finite coordinate anywhere in either field dies at the whole-corpus fit before any -/// per-row reading, and a collinear canonical gauge constellation admits a similarity fit -/// while refusing the affine one. Production data reaches both arms. +/// Refuses a non-finite coordinate at the field's proof and a collinear gauge's affine fit alone. +/// +/// A non-finite coordinate is refused at the field's own proof, naming its row, and never reaches +/// the reading. A collinear canonical gauge constellation admits a similarity fit while refusing +/// the affine one. #[test] fn refusals_name_the_fit_that_could_not_be_made() { let anchors = gauge(); @@ -251,7 +264,8 @@ fn refusals_name_the_fit_that_could_not_be_made() { let zero = SNAPSHOT; // A non-finite coordinate never reaches the reading: the readback boundary's proof - // refuses it naming the row, so the reading's own refusals cover fit degeneracy alone. + // refuses it naming the row. The reading's own refusals therefore cover fit degeneracy + // alone. let mut poisoned = scaled(&SNAPSHOT, 2.0); poisoned[0] = Vec2::new(f32::NAN, 16.0); assert_eq!( @@ -286,9 +300,11 @@ fn refusals_name_the_fit_that_could_not_be_made() { ); } -/// A row the projection actually clipped reads as saturated through the stored `f32` bytes: -/// the landing coordinate rounds, the floor sits a full margin below the landing radius, and -/// the enforcement summary copies the record's cumulative story at the evaluation point. +/// Reads a clipped row as saturated through the stored `f32` bytes. +/// +/// A row the projection actually clipped reads as saturated through the stored `f32` bytes: the +/// landing coordinate rounds, the saturation floor lies a full margin below the landing radius, and +/// the enforcement summary copies the record's cumulative readings at the evaluation point. #[test] fn a_clipped_row_reads_saturated_with_its_record() { let anchors = gauge(); @@ -328,7 +344,7 @@ fn a_clipped_row_reads_saturated_with_its_record() { assert_eq!(evidence.saturation[0].saturated, 1); assert_eq!(evidence.saturation[1].saturated, 0); - // The anchors never moved, so the gauge displacement family stays at zero. + // The anchors never moved, and the gauge displacement family reads zero. for family in &evidence.displacement { assert_eq!(family.displacement.q50.get(), 0.0); assert_eq!(family.displacement.mean.get(), 0.0); @@ -342,8 +358,10 @@ fn a_clipped_row_reads_saturated_with_its_record() { assert_eq!(evidence.enforcement.last_application, Some(9)); } -/// An objective-shape fit disagreeing with the whole-field realization stands as its own -/// reading and enters no bridge end: both recorded ends derive from the fields alone. +/// Keeps an objective-shape fit out of the bridge, whose ends derive from the fields alone. +/// +/// An objective-shape fit disagreeing with the whole-field realization stands as its own reading +/// and enters no bridge end: both recorded ends derive from the fields alone. #[test] fn the_objective_reading_enters_no_bridge_end() { let anchors = gauge(); @@ -377,8 +395,10 @@ fn the_objective_reading_enters_no_bridge_end() { assert_eq!(bridge(&evidence), Some(Similarity::IDENTITY)); } -/// The bridge converts a gauge-frame reading into the corpus frame by recorded arithmetic -/// alone: undoing a pure gauge scale of 2 against a corpus identity halves the reading. +/// Converts a gauge-frame reading into the corpus frame by recorded arithmetic alone. +/// +/// The bridge converts a gauge-frame reading into the corpus frame by recorded arithmetic alone: +/// undoing a pure gauge scale of 2 against a corpus identity halves the reading. #[test] fn the_bridge_composes_the_two_recorded_ends() { let gauge_similarity = Similarity::new( diff --git a/libs/@local/graph/atlas/src/salt/projector/gauge/mod.rs b/libs/@local/graph/atlas/src/salt/projector/gauge/mod.rs index 68aefbb13c6..c7526cb19b1 100644 --- a/libs/@local/graph/atlas/src/salt/projector/gauge/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/gauge/mod.rs @@ -2,7 +2,7 @@ //! //! The contrast compares distances in the zero-condition frame. The zero side is read directly, //! and the canonical side is read through a similarity fitted over the gauge anchors - rows drawn -//! disjoint from movement participants, held-out pair endpoints, and matched controls, so the +//! disjoint from movement participants, held-out pair endpoints, and matched controls, and the //! optimizer cannot own the frame it is measured in. Rotation and translation cancel in pair //! distances, which concentrates the whole alignment in the fitted scale `s`. The fit is live: //! `s` carries a derivative into both fields' anchor coordinates, and hiding either path would @@ -18,20 +18,20 @@ //! - zero: `∂s/∂x₀(g) = (R·u(g)) / D` //! //! Centring makes each raw-coordinate derivative the centred one minus the mean of all centred -//! derivatives, and both means vanish over centred sums (`Σu = Σv = 0`), so the forms above are -//! exact in the raw coordinates. The tests pin the Euler laws this buys: `Σ u·∂s/∂x_c = −s` and -//! `Σ v·∂s/∂x₀ = +s`, because the scale is degree −1 in the canonical constellation and degree -//! +1 in the zero one. The per-anchor magnitude falls as `1/(|G|·spread)`, which is what makes -//! the gauge channel a reading rather than a lever. +//! derivatives, and both means vanish over centred sums (`Σu = Σv = 0`). The forms above are +//! therefore exact in the raw coordinates. Because the scale is degree −1 in the canonical +//! constellation and degree +1 in the zero one, the Euler laws `Σ u·∂s/∂x_c = −s` and +//! `Σ v·∂s/∂x₀ = +s` follow. The per-anchor magnitude falls as `1/(|G|·spread)`, which is what +//! makes the gauge channel a reading rather than a lever. //! -//! Each fit rule is a shape fixed by the derivation with an owner-valued number, and a rule -//! whose number is not yet declared does not bind. The minimum spread (`spread_G/band ≥ κ`) and -//! the minimum effective count (the Kish form over anchors deduplicated by duplicate class) bind -//! at the freeze. The maximum normalized residual binds at every fit. Any degeneracy - the -//! closed form's own refusals or a residual above its bar - lands in the one refusal class, -//! [`GaugeRefusal`], whose outcome is fixed: publish no activation candidate and record the -//! failed reading. The fit always runs on all of the anchors, never a subsample, because the -//! estimator's contract needs `s` to be a function of the fields alone. +//! Each fit rule is a shape fixed by the derivation with a declared number, and a rule whose +//! number is not declared does not bind. The minimum spread (`spread_G/band ≥ κ`) and the +//! minimum effective count (the Kish form over anchors deduplicated by duplicate class) bind at +//! the freeze. The maximum normalized residual binds at every fit. Any degeneracy - the closed +//! form's own refusals or a residual above its bar - is one refusal class, [`GaugeRefusal`], +//! whose outcome is fixed: publish no activation candidate and record the failed reading. The +//! fit always runs on all of the anchors, never a subsample, because the estimator's contract +//! needs `s` to be a function of the fields alone. mod refusal; #[cfg(test)] @@ -50,8 +50,9 @@ hashql_core::id::newtype! { } hashql_core::id::newtype! { - /// The split's duplicate-class covariate, under which byte-identical embedding rows share - /// one class. + /// The split's duplicate-class covariate. + /// + /// Byte-identical embedding rows share one class under it. /// /// The draw machinery assigns the ids. This module consumes them for the effective count, /// where duplicates of one class are the same evidence and count once. @@ -62,29 +63,33 @@ hashql_core::id::newtype! { /// The band-conditioned minimum-spread rule. /// /// Present when the replicate-band artifact exists. A frame whose defining spread is commensurate -/// with the band has noise-owned units, so the anchors' frozen spread must satisfy +/// with the band has noise-owned units. The anchors' frozen spread must therefore satisfy /// `spread_G / band ≥ κ`. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct SpreadFloor { - /// `κ`: the owner's spread factor. Its value is an open owner decision. Its role is not. + /// `κ`: the declared spread factor. Its value remains an open choice. Its role is fixed. pub kappa: Positive, /// The constraint radius in world units: the same-frame reconstruction `β_proj · s_ref`. pub band: Positive, } -/// The frozen gauge population holds the anchor rows with their duplicate classes beside the -/// frozen spread and the effective count. +/// The frozen gauge population. +/// +/// It pairs the anchor rows and their duplicate classes with the frozen spread and the effective +/// count. #[derive(Debug, PartialEq)] pub(crate) struct GaugeAnchors { /// The anchor rows in draw order. rows: Box>, /// Each anchor's duplicate class, aligned with `rows`. classes: Box>, - /// `spread_G(Z_K)`: the anchors' centred RMS spread in the boundary snapshot, the frozen - /// denominator of every normalized residual and the minimum-spread rule's reading. + /// `spread_G(Z_K)`: the anchors' centred RMS spread in the boundary snapshot. + /// + /// The frozen denominator of every normalized residual and the minimum-spread rule's reading. frozen_spread: Positive, - /// `n_eff_G`: the Kish effective count over anchors deduplicated by duplicate class. With - /// equal per-anchor weights this is the distinct class count, and a stratified-weighted + /// `n_eff_G`: the Kish effective count over anchors deduplicated by duplicate class. + /// + /// With equal per-anchor weights this is the distinct class count, and a stratified-weighted /// draw would supply its own weights. effective_count: DNonNegative, } @@ -170,10 +175,9 @@ where /// Fits the alignment over pre-gathered anchor constellations in draw order. /// - /// The trainer's per-step evaluation holds the anchors' coordinates in a batch-local frame - /// rather than in whole-corpus fields, so this entry takes the two constellations already - /// gathered - `source` the anchors' canonical coordinates and `target` their zero-frame - /// coordinates, both in draw order. + /// This entry takes the two constellations already gathered - `source` the anchors' canonical + /// coordinates and `target` their zero-frame coordinates, both in draw order - for a caller + /// that holds the anchors in a batch-local frame rather than in whole-corpus fields. /// /// Both constellations cover the anchor draw - a wiring contract checked in debug builds, /// since the gather and the frozen draw come from one gauge. @@ -246,22 +250,28 @@ where } } -/// One evaluation's fitted alignment carries the similarity and its scale beside the normalized -/// residual and the scale's exact adjoints into both fields' anchor coordinates. +/// One evaluation's fitted alignment. +/// +/// It holds the similarity and its scale, the normalized residual, and the scale's exact adjoints +/// into both fields' anchor coordinates. #[derive(Debug, PartialEq)] pub(crate) struct GaugeFit { /// The fitted similarity, canonical onto zero: the evidence bridge between frames. similarity: Similarity, /// The fitted scale `s`, the one live alignment quantity pair distances consume. scale: Positive, - /// `RMS(S(x_c(g)) − x₀(g)) / spread_G(Z_K)`: the non-similarity deformation of the gauge - /// constellation, recorded at every fit and bounded by the bar when one is declared. + /// `RMS(S(x_c(g)) − x₀(g)) / spread_G(Z_K)`: the constellation's non-similarity deformation. + /// + /// Recorded at every fit and bounded by the bar when one is declared. residual: DNonNegative, - /// `∂s/∂x_c(g)` per anchor: the adjoint that fans the objective's pull on `s` into the - /// canonical anchor coordinates. + /// `∂s/∂x_c(g)` per anchor: the adjoint into the canonical anchor coordinates. + /// + /// It fans the objective's pull on `s` into those coordinates. canonical_adjoints: Box>, - /// `∂s/∂x₀(g)` per anchor: the zero-field twin, present for the same reason the contrast's - /// zero slope is - hiding a real path would misstate the derivative. + /// `∂s/∂x₀(g)` per anchor: the zero-field twin of the canonical adjoint. + /// + /// Present for the same reason the contrast's zero slope is - hiding a real path would + /// misstate the derivative. zero_adjoints: Box>, } @@ -303,7 +313,7 @@ type AdjointFields = ( /// Evaluates both adjoint fields at the fitted optimum, per anchor in parallel. /// -/// The rotation and scale re-widen from the fitted f32 coefficients, so the adjoints +/// The rotation and scale re-widen from the fitted f32 coefficients. The adjoints therefore /// differentiate the alignment the forward pass actually uses. `D` re-accumulates in f64 through /// the deterministic chunked reduction. fn adjoints( @@ -312,7 +322,7 @@ fn adjoints( similarity: Similarity, scale: Positive, ) -> Result { - // The fields carry the finiteness proof, so the statistics evaluate with no scan. + // The fields carry the finiteness proof, and the statistics evaluate with no scan. let source_centre = source.centroid(); let target_centre = target.centroid(); diff --git a/libs/@local/graph/atlas/src/salt/projector/gauge/refusal.rs b/libs/@local/graph/atlas/src/salt/projector/gauge/refusal.rs index df028dbbabe..d56c4671222 100644 --- a/libs/@local/graph/atlas/src/salt/projector/gauge/refusal.rs +++ b/libs/@local/graph/atlas/src/salt/projector/gauge/refusal.rs @@ -7,9 +7,9 @@ use crate::math::{DNonNegative, DPositive, Positive}; /// A refused gauge publishes no activation candidate and records the failed reading. /// -/// Every variant is the failure table's alignment-degeneracy row. Nothing branches on them and -/// there is no degraded mode: a refused freeze has no gauge, a refused fit has no scale, and -/// training refuses the step rather than descending through a degenerate frame. +/// Every variant is an alignment degeneracy. Nothing branches on them and there is no degraded +/// mode. A refused freeze has no gauge, a refused fit has no scale, and training refuses the step +/// rather than descending through a degenerate frame. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum GaugeRefusal { /// Fewer than two anchors: no constellation to fit a frame on. @@ -17,33 +17,37 @@ pub(crate) enum GaugeRefusal { /// The anchors supplied. count: usize, }, - /// The anchors' frozen spread is not a strictly positive f32: the constellation is - /// coincident, or past the working precision. + /// The anchors' frozen spread is not a strictly positive f32. + /// + /// The constellation is coincident, or past the working precision. DegenerateSpread { /// The spread as measured, in double precision. spread: f64, }, - /// The frozen spread sits below the declared band floor, so the frame's units are - /// noise-owned. + /// The frozen spread sits below the declared band floor: the frame's units are noise-owned. SpreadBelowFloor { /// `spread_G / band`. ratio: DPositive, /// The declared `κ`. kappa: Positive, }, - /// The effective anchor count falls below the declared minimum, so the fitted scale's - /// stability has no sample behind it. + /// The effective anchor count falls below the declared minimum. + /// + /// The fitted scale's stability has no sample behind it. UndersizedEffectiveCount { /// The Kish effective count over duplicate classes. effective: DNonNegative, /// The declared minimum. minimum: Positive, }, - /// The closed form refused, over coincident anchors, an exactly cancelling covariance, a - /// non-finite coordinate, or a fitted coefficient outside the accepted range. + /// The closed form refused. + /// + /// The cause is coincident anchors, an exactly cancelling covariance, a non-finite coordinate, + /// or a fitted coefficient outside the accepted range. FitRefused, - /// The normalized residual exceeds the declared bar: the gauge constellation deformed - /// beyond similarity, and the alignment is not a measurement. + /// The normalized residual exceeds the declared bar. + /// + /// The gauge constellation deformed beyond similarity, and the alignment is not a measurement. ResidualAboveBar { /// `RMS(S(x_c(g)) − x₀(g)) / spread_G`. residual: DNonNegative, diff --git a/libs/@local/graph/atlas/src/salt/projector/gauge/tests.rs b/libs/@local/graph/atlas/src/salt/projector/gauge/tests.rs index a94058f41fe..be48e6b0cfc 100644 --- a/libs/@local/graph/atlas/src/salt/projector/gauge/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/gauge/tests.rs @@ -1,9 +1,9 @@ //! Certificates for the gauge alignment. //! //! A square constellation under scale 2 and a right-angle rotation, translated by integers, -//! lands every fit coefficient and both adjoint fields on exactly representable values, and the -//! Euler sums with them, so the recovery fixture asserts exact contracts. The finite-difference -//! certificate uses a generic constellation and an f64 mirror of the closed form. +//! makes every fit coefficient, both adjoint fields and the Euler sums exactly representable, and +//! the recovery fixture therefore asserts exact contracts. The finite-difference certificate uses +//! a generic constellation and an f64 mirror of the closed form. #![expect( clippy::float_cmp, @@ -38,16 +38,19 @@ fn fit_fields( ) } +/// A duplicate class id from a literal. fn class(id: u32) -> DuplicateClassId { DuplicateClassId::new(id) } +/// Builds a [`Positive`] from a literal test value. fn positive(value: f32) -> Positive { Positive::new(value).expect("test value is positive") } -/// A gauge over corpus rows 1, 2, 4, 5 with distinct duplicate classes, frozen against -/// `snapshot`. +/// Builds a gauge over corpus rows 1, 2, 4, 5 frozen against `snapshot`. +/// +/// The anchors carry distinct duplicate classes. fn square_gauge(snapshot: &[Vec2]) -> GaugeAnchors { GaugeAnchors::freeze( Box::new([1, 2, 4, 5].map(NodeRowId::new)), @@ -79,8 +82,10 @@ const ZERO: [Vec2; 6] = [ Vec2::new(3.0, -2.0), ]; -/// Every fit coefficient lands exactly on this fixture, and so do the adjoint fields and the -/// Euler sums. Anchors sit at non-contiguous corpus rows, so the gather is also under test. +/// Reads every fit coefficient, adjoint field and Euler sum exactly on the square fixture. +/// +/// Every fit coefficient is exact on this fixture, and so are the adjoint fields and the Euler +/// sums. Anchors lie at non-contiguous corpus rows: the gather is also under test. #[test] fn recovers_an_exact_similarity_with_exact_adjoints() { let gauge = square_gauge(&ZERO); @@ -110,7 +115,7 @@ fn recovers_an_exact_similarity_with_exact_adjoints() { ); // Euler laws: the scale is degree −1 in the canonical constellation and degree +1 in the - // zero one, so the centred dot of each field with its adjoints reads ∓s exactly. + // zero one: the centred dot of each field with its adjoints reads ∓s exactly. let rows = [1_usize, 2, 4, 5]; let canonical_euler: f32 = rows .iter() @@ -167,6 +172,8 @@ fn mirror_scale(source: &[(f64, f64)], target: &[(f64, f64)]) -> f64 { dot.hypot(perp) / variance } +/// Matches both adjoint fields to central finite differences on a generic constellation. +/// /// Both adjoint fields match central finite differences of the closed form on a generic /// constellation. #[test] @@ -254,6 +261,8 @@ fn adjoints_match_finite_differences() { } } +/// Moves the fitted translation alone when the zero field translates. +/// /// Translating the zero field moves the fitted translation alone: the centred quantities and /// therefore both adjoint fields are bit-identical. #[test] @@ -279,7 +288,7 @@ fn adjoints_are_invariant_under_zero_field_translation() { /// The minimum-spread rule binds at the freeze, inclusively at its edge. #[test] fn the_spread_floor_binds_at_the_freeze() { - // Frozen spread is exactly 2, so a band of 0.5 reads a ratio of exactly 4. + // Frozen spread is exactly 2: a band of 0.5 reads a ratio of exactly 4. GaugeAnchors::freeze( Box::new([1, 2, 4, 5].map(NodeRowId::new)), Box::new([class(0), class(1), class(2), class(3)]), @@ -337,8 +346,10 @@ fn the_effective_count_deduplicates_by_class() { assert_eq!(admitted.effective_count(), DNonNegative::from_usize(3)); } -/// A deformation orthogonal to the fit's normal equations leaves the similarity exactly in -/// place and lands whole in the residual, so the bar's reading is exact. +/// Leaves the similarity in place under a deformation orthogonal to the normal equations. +/// +/// A deformation orthogonal to the fit's normal equations leaves the similarity exactly in place +/// and enters the residual whole: the bar's reading is exact. #[test] fn the_residual_bar_binds_at_the_fit() { let canonical = [ @@ -347,8 +358,8 @@ fn the_residual_bar_binds_at_the_fit() { Vec2::new(0.0, 1.0), Vec2::new(0.0, -1.0), ]; - // Identity similarity plus the orthogonal deformation (0, ±0.5): Σe = 0, Σu·e = 0, - // Σu⊥·e = 0, so the fit stays the identity and the residual RMS is exactly 0.5. + // Identity similarity plus the orthogonal deformation (0, ±0.5): Σe = 0, Σu·e = 0 and + // Σu⊥·e = 0, hence the fit stays the identity and the residual RMS is exactly 0.5. let zero = [ Vec2::new(1.0, 0.5), Vec2::new(-1.0, 0.5), diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/contrast.rs b/libs/@local/graph/atlas/src/salt/projector/loss/contrast.rs index 38b44a754bd..dc7634fc49c 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/contrast.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/contrast.rs @@ -4,30 +4,39 @@ //! d₀)/σ₀ + m`. The aligned canonical distance is compared against the zero distance in units of //! the pair's frozen ruler `σ₀`, and the margin `m` offsets the comparison. Contraction means `v` //! falls. Equality of the two distances is a failure to contract, and a non-negative margin keeps -//! it one. The penalty applied to `v` and the per-pair weight are the batch term's to fold - this -//! module owns the violation and its live partials, evaluated fused so the pair loops stay pure -//! plumbing. +//! it one. The penalty applied to `v` and the per-pair weight are the batch term's to fold. This +//! module owns the violation and its live partials, evaluated fused, and the pair loops apply +//! them without deriving anything themselves. //! //! The fitted scale `s` is live. It comes from the similarity alignment of the canonical field -//! onto the zero field, refit whenever the fields move, so it carries a derivative: rotation and -//! translation cancel in pair distances, which concentrates the whole alignment orbit in this one -//! scalar and makes the violation invariant in value and derivative under translation, rotation, -//! and uniform scaling of the canonical field. The ruler `σ₀` is a declared constant of the -//! estimand, measured once on the zero-condition snapshot taken before the objective's first -//! gradient and frozen with the generation. No gradient exists through it, because nothing live -//! enters it. A live ruler would hand the optimizer its own unit of account, and a detached copy -//! of a live quantity would lie about the derivative. A frozen constant does neither. +//! onto the zero field, refit whenever the fields move, and it carries a derivative. Rotation and +//! translation cancel in pair distances, and the refit scale absorbs a uniform scaling, which +//! concentrates the whole alignment orbit in this one scalar. The violation's value is therefore +//! invariant under translation, rotation, and uniform scaling of the canonical field, and its +//! scalar partials in the distances and the scale are invariant under translation and rotation, +//! while the coordinate gradients transform with the coordinates. Under a uniform scaling of the +//! canonical field by `c > 0`, with the zero field and the ruler fixed, `d_c` becomes `c·d_c` and +//! the refit scale becomes `s/c`: the product `s·d_c` and `v` remain unchanged, `∂v/∂d_c = s/σ₀` +//! divides by `c`, `∂v/∂d₀ = −1/σ₀` remains unchanged, and `∂v/∂s = d_c/σ₀` multiplies by `c`. +//! These laws hold in exact real arithmetic, and a finite-precision refit and evaluation can round +//! differently after the transformation. //! -//! The zero-side partial `∂v/∂d₀ = −1/σ₀` is negative, so the optimizer is paid to inflate a +//! The ruler `σ₀` is a declared constant of the estimand, measured once on the zero-condition +//! snapshot taken before the objective's first gradient and frozen with the generation. No +//! gradient exists through it, because nothing live enters it. A live ruler would hand the +//! optimizer its own unit of account, and a detached copy of a live quantity would lie about the +//! derivative. A frozen constant does neither. +//! +//! The zero-side partial `∂v/∂d₀ = −1/σ₀` is negative: the optimizer is paid to inflate a //! violating pair's zero distance. The reward is real and stays in the gradient. What holds it is //! the per-row band projection on the zero field, never the derivative's absence, and an //! implementation that detaches or drops the zero-side path optimizes a different objective whose //! constraint claim is false. The hand derivation exists to keep these signs exact. //! -//! At coincidence either distance's direction vector is undefined, and the coordinate fold treats -//! the contribution as zero: the value still counts, the pull has nowhere to point. The slopes -//! this module returns are direction-free scalars. Zeroing the fold at coincidence is the batch -//! term's contract, stated on [`ContrastEnergy::evaluate`]. +//! At coincidence either distance's direction vector is undefined, and the coordinate fold +//! deposits a zero contribution for that distance while the value still counts. The slopes this +//! module returns are direction-free scalars. Zeroing the fold at coincidence is the batch term's +//! contract, stated on [`ContrastEnergy::evaluate`]. use crate::math::{DNonNegative, DPositive, Negative, NonNegative, Positive}; @@ -35,12 +44,12 @@ use crate::math::{DNonNegative, DPositive, Negative, NonNegative, Positive}; /// /// The fitted scale is the similarity alignment's scalar for the evaluation being scored, and /// the margin is the violation's offset at distance equality. Both are constant across the pairs of -/// one evaluation, so the per-pair loop carries one copy. +/// one evaluation. /// /// The margin's domain is non-negative because equality must stay a failure to contract: a /// negative margin would score `s·d_c = d₀` as satisfied, and the objective's product meaning is -/// that an uncontracted pair is never satisfied. The margin's value is an open owner decision; -/// its domain is not. +/// that an uncontracted pair is never satisfied. The margin's value remains an open choice. Its +/// domain does not. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct ContrastEnergy { fitted_scale: Positive, @@ -49,24 +58,46 @@ pub(crate) struct ContrastEnergy { /// One pair's violation and its three live partial derivatives, fused. /// -/// Every field is finite by construction of the inputs: the ruler is strictly positive, the -/// distances are finite and non-negative, and the fitted scale is finite and strictly positive. +/// The widened slopes carry an unconditional finite-domain guarantee, the zero slope a conditional +/// one, and the violation none. The canonical and fitted-scale slopes are widened quotients of +/// in-domain `f32`-born values and never leave their domains. The zero slope is a negated `f32` +/// reciprocal, in domain for every ruler above `2⁻¹²⁸`. The violation is a raw `f32` that overflows +/// at inputs inside every documented domain: a fitted scale of `f32::MAX` with a canonical distance +/// of `2` under a ruler of `1` reads `+∞` with the zero distance and the margin at `0`. +/// [`ContrastEnergy::evaluate`] states the ranges. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct ContrastEvaluation { - /// The violation `v = (s·d_c − d₀)/σ₀ + m`, raw: the aligned product of unbounded - /// working-precision factors can overflow, and the penalty folds the reading under the - /// objective's own gate. + /// The violation `v = (s·d_c − d₀)/σ₀ + m`, raw: it can overflow to either infinity. + /// + /// Every operation is `f32` arithmetic rounded to nearest, and an operation overflows when + /// its rounded result exceeds `f32::MAX` in magnitude: an exact result less than half an ulp + /// (`2¹⁰³`) beyond `f32::MAX` rounds back to it and remains finite, and one at least that far + /// rounds to the infinity of its sign. Three of the four operations can overflow. The aligned + /// product `s·d_c` overflows to `+∞`. The quotient by `σ₀` overflows to `+∞` or `−∞`, + /// following the sign of the difference `s·d_c − d₀`, and with a finite product that needs a + /// ruler below one. The final addition of the margin overflows to `+∞`: at `s = 1`, + /// `d_c = m = f32::MAX`, `d₀ = 0` and `σ₀ = 1` the product and the quotient read `f32::MAX` + /// and the sum reads `+∞`. The subtraction cannot overflow, because the difference of two + /// non-negative values never exceeds the larger of them in magnitude. Every input is finite. + /// The reading is therefore never NaN: an overflow yields an infinity, and every later operand + /// is finite. pub violation: f32, - /// `∂v/∂d_c = s/σ₀`, strictly positive: shrinking the aligned canonical distance is the - /// objective's productive direction. Total: the widened quotient of two `f32`-born - /// positives never leaves the domain. + /// The canonical slope `∂v/∂d_c = s/σ₀`, strictly positive. + /// + /// Shrinking the aligned canonical distance is the objective's productive direction. The + /// slope is total: the widened quotient of two `f32`-born positives never leaves the domain. pub canonical_slope: DPositive, - /// `∂v/∂d₀ = −1/σ₀`, strictly negative: the reward for inflating the zero distance, present - /// and honest, held by the band projection rather than hidden from the gradient. + /// The zero slope `∂v/∂d₀ = −1/σ₀`, strictly negative. + /// + /// The reward for inflating the zero distance, present and honest, held by the band + /// projection rather than hidden from the gradient. In domain for every ruler above `2⁻¹²⁸`. + /// A ruler at or below it overflows the `f32` reciprocal, and the slope leaves its domain, as + /// [`ContrastEnergy::evaluate`] states. pub zero_slope: Negative, - /// `∂v/∂s = d_c/σ₀`, non-negative, zero exactly at canonical coincidence. Its adjoint fans - /// into the gauge anchors' canonical coordinates through the alignment fit. Total by the - /// same widened quotient. + /// The fitted-scale slope `∂v/∂s = d_c/σ₀`, zero exactly at canonical coincidence. + /// + /// Non-negative, and total by the same widened quotient as the canonical slope. Its adjoint + /// fans into the gauge anchors' canonical coordinates through the alignment fit. pub fitted_scale_slope: DNonNegative, } @@ -82,14 +113,37 @@ impl ContrastEnergy { /// Evaluates one pair's violation and its partials in the two distances and the scale. /// - /// `ruler` is the pair's frozen `σ₀`, strictly positive by construction of the frozen table, - /// so every quotient here is total. `canonical_distance` and `zero_distance` are the raw - /// (unaligned) canonical and zero-frame pair distances. + /// `ruler` is the pair's frozen `σ₀`, strictly positive by construction of the frozen table. + /// `canonical_distance` and `zero_distance` are the raw (unaligned) canonical and zero-frame + /// pair distances, finite and non-negative in their typed domain. A distance whose `f32` + /// computation escaped to `+∞` lies outside that domain, and the violation then follows the + /// other operand. An infinite canonical distance reads an infinite fitted-scale slope and, + /// against a finite zero distance, a `+∞` violation. An infinite zero distance against a + /// finite aligned distance `s·d_c` reads a `−∞` violation. An infinite aligned distance + /// against an infinite zero distance reads NaN, the difference `∞ − ∞`, whether the canonical + /// distance escaped or the product `s·d_c` alone overflowed. The guarantees below, the + /// violation's never-NaN reading included, hold inside the domain. /// - /// The returned slopes are scalars in the distances. Folding them into coordinate gradients - /// multiplies by the pair's unit direction vectors, and at coincidence of either field's - /// endpoints that direction is undefined: the caller folds a zero contribution there, keeping - /// the value and dropping the pull, which is the continuous limit. + /// Positivity defines every quotient here and makes the two widened ones total: `s/σ₀` and + /// `d_c/σ₀` divide `f32`-born values in `f64` and stay in their domains at every input, with + /// no condition on the ruler. The other two divisions run in `f32`, where positivity is not + /// enough. The formula `−1/σ₀` is finite for every positive real ruler, and + /// [`Positive::recip`] forms it as an `f32` division, which overflows for a ruler at or below + /// `2⁻¹²⁸` (below `1/f32::MAX`, a subnormal such as `f32::from_bits(1)`). The zero slope's + /// guarantee is therefore conditional. A ruler above `2⁻¹²⁸`, where every normal `f32` lies, + /// keeps it in domain, and a ruler at or below that bound leaves the returned slope outside + /// the [`Negative`] domain. The violation is raw at every ruler, as its field states. + /// + /// The returned slopes are scalars in the distances. A coordinate gradient multiplies a slope + /// by the gradient of the pair distance `d = ‖y_source − y_target‖` between the endpoint + /// coordinates, the unit direction `(y_source − y_target)/d`, defined wherever `d > 0`. At + /// coincidence, `d = 0`, the norm has no gradient and its subdifferential is the closed unit + /// ball: every vector of norm at most one is a subgradient. The caller folds the zero vector + /// there, the symmetric choice among those subgradients rather than a continuous limit, + /// keeping the value and dropping that distance's pull. The distance's scalar slope remains + /// `s/σ₀` or `−1/σ₀` at coincidence, and the pull the caller deposits also carries the slope + /// `φ′(v)` of the penalty `φ` applied to `v`, which the quadratic hinge sets to zero at or + /// below a zero violation. #[must_use] pub(crate) const fn evaluate( self, @@ -99,12 +153,16 @@ impl ContrastEnergy { ) -> ContrastEvaluation { let scale = self.fitted_scale; // The aligned product and the violation are raw: two unbounded working-precision - // factors can overflow, and the difference crosses signs. The penalty owns the check. + // factors can overflow, and the difference crosses signs. The estimator's fold finish + // is the check. let aligned = scale.get() * canonical_distance; ContrastEvaluation { violation: (aligned - zero_distance.get()) / ruler + self.margin.get(), canonical_slope: scale.div_wide(ruler), + // `Positive::recip` divides in `f32` and only debug-asserts finiteness: a ruler at or + // below 2⁻¹²⁸ panics there with debug assertions enabled and reads `−∞` inside + // `Negative` without them. zero_slope: -ruler.recip(), fitted_scale_slope: canonical_distance.div_wide(ruler), } @@ -113,15 +171,16 @@ impl ContrastEnergy { #[cfg(test)] mod tests { + use super::{ContrastEnergy, ContrastEvaluation}; use crate::math::{NonNegative, Positive}; - /// Central finite difference of `function` at `at` with the given step. + /// Estimates `function`'s derivative at `at` by a central difference of half-width `step`. fn central_difference(function: impl Fn(f64) -> f64, at: f64, step: f64) -> f64 { (function(at + step) - function(at - step)) / (2.0 * step) } - /// The violation stated verbatim, for finite differences. + /// Computes the violation in `f64` from its defining expression, for finite differences. #[expect( clippy::suboptimal_flops, reason = "the mirror states the defining expression verbatim" @@ -130,6 +189,13 @@ mod tests { (scale * canonical - zero) / ruler + margin } + /// Builds a contrast energy from raw values and evaluates it at the given ruler and distances. + /// + /// # Panics + /// + /// This panics when `scale` or `ruler` is not finite and positive, or when `margin`, + /// `canonical` or `zero` is not finite and non-negative: the typed constructors refuse the + /// value, and the `expect` names the argument. fn evaluate( scale: f32, margin: f32, @@ -148,6 +214,13 @@ mod tests { ) } + /// The slopes agree with central finite differences of the violation within `1e-6`. + /// + /// The violation is affine in each argument varied. The central difference is therefore the + /// exact partial up to `f64` rounding. The implementation evaluates at the `f32`-rounded + /// constants and the reference at the unrounded ones: at most two roundings of relative size + /// `2⁻²⁴` on slopes below `3.5` keep every compared slope within `5·10⁻⁷`, inside the + /// tolerance. #[test] #[expect( clippy::cast_possible_truncation, @@ -186,6 +259,7 @@ mod tests { assert!((f64::from(evaluation.fitted_scale_slope) - scale_reference).abs() < 1e-6); } + /// The signs follow the formula: `s/σ₀ > 0`, `−1/σ₀ < 0` and `d_c/σ₀ > 0` off coincidence. #[test] fn sign_structure_holds() { let evaluation = evaluate(1.25, 0.1, 0.7, 2.4, 3.1); @@ -206,11 +280,12 @@ mod tests { #[test] fn equality_reads_the_margin() { - // s·d_c = 2.0 = d₀, so the violation is exactly the margin. + // s·d_c = 2.0 = d₀: the violation is exactly the margin. let evaluation = evaluate(0.5, 0.25, 0.8, 4.0, 2.0); assert!((evaluation.violation - 0.25).abs() < 1e-6); } + /// Doubling the ruler halves every slope within `1e-7`. #[test] fn ruler_denominated_slopes_halve_when_the_ruler_doubles() { let narrow = evaluate(1.25, 0.1, 0.7, 2.4, 3.1); @@ -227,6 +302,7 @@ mod tests { ); } + /// Scaling `d_c` by `c` and `s` by `1/c` leaves the violation unchanged. #[test] fn violation_value_is_invariant_along_the_scaling_orbit() { // A uniform canonical scaling by c with the fitted scale refit to s/c reads the same diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/energy.rs b/libs/@local/graph/atlas/src/salt/projector/loss/energy.rs index 36fcd69f10f..e6e157a06b9 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/energy.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/energy.rs @@ -26,8 +26,8 @@ impl AffinityEnergy { /// Returns [`None`] unless the curve's exponent satisfies `b ≥ 0.5`. The offset keeps the /// attraction value finite for far pairs and bounds the repulsion gradient for near pairs. /// The exponent bound keeps the coordinate gradient finite at coincidence, where its - /// magnitude scales as `d^(2b - 1)` (fitted curves land well inside the bound - rejecting - /// the rest makes gradient boundedness a property of the type, not of the corpus). + /// magnitude scales as `d^(2b - 1)`. Fitted curves lie well inside the bound. Rejecting the + /// rest makes gradient boundedness a property of the type rather than of the corpus. #[must_use] pub(crate) fn new(curve: AffinityCurve, epsilon: Positive) -> Option { (curve.b() >= 0.5).then_some(Self { curve, epsilon }) @@ -71,7 +71,7 @@ impl AffinityEnergy { (value, derivative) } - /// Computes the shared derivative mass `a b u^(b - 1) q^2`. + /// Computes the shared derivative mass `a b u^(b - 1) q²`. /// /// `-q'(u)` in both derivatives; the callers divide by their respective logarithm arguments and /// choose the sign. @@ -95,7 +95,7 @@ impl AffinityEnergy { /// radius and stays positive at every finite distance, asymptotically a factor of `e` per /// temperature of depth inside, with residual `sigmoid(-radius / temperature)` at coincidence. /// -/// The energy is strictly increasing, so coincidence is its unique minimum. That residual and the +/// The energy is strictly increasing, and coincidence is its unique minimum. That residual and the /// competing terms jointly set a pair's equilibrium distance. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct ProximalEnergy { @@ -240,9 +240,10 @@ impl RelationEnergy { /// /// The mixture scales each class energy by its weight, and the derivative is the matching /// weighted sum of class derivatives. The fold widens the f32-born readings once and runs - /// in double width, and a product of unbounded weights and saturated energies can still - /// overflow, so the pair rides as unclaimed derivations and each consumer folds the - /// reading under its own check. + /// in double width, where a product of two in-domain `f32` operands lies far inside the + /// `f64` range and cannot overflow. The types carry no such bound. The pair therefore + /// returns as unclaimed [`Derivation`]s, and each consumer chooses its own exit: a checked + /// finish or a raw fold. pub(crate) fn mixture( self, normalized: NonNegative, diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/mod.rs b/libs/@local/graph/atlas/src/salt/projector/loss/mod.rs index 2183acb85a8..76970323b8e 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/mod.rs @@ -1,26 +1,37 @@ //! The composite training objective over a prepared batch. //! -//! The objective splits along the hand-gradient seam. The hand-gradient terms - semantic -//! attraction, ordinary and hard-negative repulsion, and relation attraction - evaluate value and -//! coordinate gradient in one fused pass over their edge lists, with every derivative hand-derived -//! in [`energy`] and certified against finite differences; their gradients accumulate into -//! [`GradientField`]s the budget measures per node before the combined field reaches shared -//! parameters. The support term rides ordinary autodiff on the coordinate tensor, so nothing needs -//! its gradient ahead of the backward pass. +//! The objective splits into hand-gradient terms and an autodiff term. The hand-gradient terms - +//! semantic attraction, ordinary and hard-negative repulsion, and relation attraction - evaluate +//! value and coordinate gradient in one fused pass over their edge lists, with every derivative +//! hand-derived in [`energy`] and certified against finite differences. Their gradients accumulate +//! into [`GradientField`]s the budget measures per node before the combined field reaches shared +//! parameters. The support term takes its gradient from ordinary autodiff on the coordinate +//! tensor, because no consumer reads that gradient ahead of the backward pass. //! //! Every term takes a premultiplied `scale`: the term's loss coefficient times any estimator //! normalization (the semantic term's total-weight-over-batch-size factor, the relation term's lens -//! factor). The terms speak the batch-local row domain: pairs, edges, and anchors carry +//! factor). The terms index the batch-local row domain: pairs, edges, and anchors carry //! [`BatchRowId`] positions into the coordinate slice each term evaluates. That key is distinct //! from the corpus's [`NodeRowId`](crate::identity::NodeRowId) by design. The assembly that //! re-indexes corpus draws into a batch owns the conversion, and the type system keeps the two //! domains apart. //! //! Pairs at exactly zero distance contribute their value but no gradient: a coincident pair has no -//! direction to move along. Coincidence is the attraction and relation energies' minimum and the -//! repulsion energy's maximum - a stationary point whose coordinate gradient vanishes as `d^(2b - -//! 1)` under the curve's `b ≥ 1/2` construction bound, so the zero is the continuous limit and any -//! separation restores the outward push. +//! direction to move along. For the affinity terms (semantic attraction and both repulsions) +//! coincidence is the energy's minimum or maximum in the distance, and the coordinate gradient's +//! magnitude scales as `d^(2b - 1)`. [`AffinityEnergy::new`]'s bound `b ≥ 1/2` keeps that magnitude +//! bounded at coincidence. Every `b > 1/2` sends it to zero, which makes the zero contribution the +//! continuous limit there. At the admitted endpoint `b = 1/2` the magnitude has a finite nonzero +//! limit (with `a = 1` and offset one, `q(d) = 1 / (1 + d)`, the attraction energy's right-hand +//! radial derivative at zero is `1/2` and the repulsion energy's is `-1`), the direction has no +//! limit, and the energy's explicit zero branch selects the zero vector. The relation energies +//! split the same way. +//! [`CoincidentEnergy`]'s radial derivative has limit zero at coincidence, also at a zero radius +//! under its Huber branch, and the zero contribution is its limit. [`ProximalEnergy`]'s radial +//! slope at zero distance is `sigmoid(-radius / temperature)`, positive in the real model, and its +//! coordinate gradient has no unique direction at that point. The relation fold's zero contribution +//! is an explicit rule for that case, the symmetric choice among the directions, rather than a +//! limit. Any separation restores the term's push or pull. mod contrast; mod energy; @@ -54,7 +65,7 @@ use crate::{ hashql_core::id::newtype! { /// A batch-local row position. /// - /// Batch assembly re-indexes one step's drawn corpus rows into a dense local domain. This key names positions in that domain and nothing else. It is distinct by design from the corpus's `NodeRowId`: a corpus row and its batch-local position are different keys, and confusing them is the wiring defect this type exists to prevent. The `u32` width is a representation bound because a batch indexes one step's participating rows. + /// Batch assembly re-indexes one step's drawn corpus rows into a dense local domain. This key names positions in that domain and nothing else. It is distinct by design from the corpus's [`NodeRowId`](crate::identity::NodeRowId): a corpus row and its batch-local position are different keys, and confusing them is the wiring defect this type exists to prevent. The `u32` width is a representation bound because a batch indexes one step's participating rows. pub(crate) struct BatchRowId(u32) } @@ -91,11 +102,11 @@ pub(crate) struct RelationEdge { /// A per-node coordinate gradient accumulator. /// -/// One field accumulates every term on one side of the budget boundary; the budget then clips the -/// relation field against the semantic field node by node. Contributions arrive in either -/// precision and accumulate in double precision; consumers narrow once where a total leaves the -/// field for the working precision. Reset and reuse the field across steps rather than -/// reallocating. +/// One field accumulates every term on one side of the budget boundary. The budget then measures +/// the relation field against the semantic field node by node, and the two apply whole. +/// Contributions arrive in either precision and accumulate in double precision. Consumers narrow +/// once where a total leaves the field for the working precision. [`take`](Self::take) zeroes the +/// entry it reads, which lets one scratch field serve several passes within a step. #[derive(Debug)] pub(crate) struct GradientField(Box>); @@ -139,13 +150,13 @@ where /// Evaluates the semantic attraction term over weighted positive pairs. /// /// Adds `scale · weight · -ln(q(d^2) + ε)` per pair to the returned value and the matching -/// hand-derived gradients to `field`. Weight-proportional sampling emits unit weights; the weight +/// hand-derived gradients to `field`. Weight-proportional sampling emits unit weights. The weight /// slot exists for capped explicit weights. /// /// # Panics /// /// This panics when a pair references a row outside `coordinates` or `field`. Pairs and coordinates -/// come from one batch assembly, so a mismatch is a wiring defect. +/// come from one batch assembly, and a mismatch is therefore a wiring defect. pub(crate) fn attraction_term( coordinates: &FinitePointField, pairs: impl IntoIterator, f32)>, @@ -164,13 +175,13 @@ where /// Evaluates a repulsion term over weighted negative pairs. /// /// Adds `scale · weight · -ln(1 - q(d^2) + ε)` per pair to the returned value and the matching -/// hand-derived gradients to `field`. Ordinary negatives carry unit weights; mined hard negatives +/// hand-derived gradients to `field`. Ordinary negatives carry unit weights. Mined hard negatives /// carry their bounded rank weights. /// /// # Panics /// /// This panics when a pair references a row outside `coordinates` or `field`. Pairs and coordinates -/// come from one batch assembly, so a mismatch is a wiring defect. +/// come from one batch assembly, and a mismatch is therefore a wiring defect. pub(crate) fn repulsion_term( coordinates: &FinitePointField, pairs: impl IntoIterator, f32)>, @@ -209,8 +220,8 @@ where total = f64::from(factor).mul_add(f64::from(value), total); - // d(d^2)/dy_left = 2 · (y_left - y_right). The pair energy supplies its derivative in the - // squared distance, so no division by the distance occurs and coincident pairs need no + // d(d²)/dy_left = 2 · (y_left - y_right). The pair energy supplies its derivative in the + // squared distance. No division by the distance occurs, and coincident pairs need no // branch beyond the energy's own zero-derivative contract. let gradient = difference * (2.0 * factor * derivative); field.accumulate(left, gradient); @@ -229,13 +240,13 @@ where /// /// Per instance the contribution is `scale · confidence · normalization · strength` times the /// weighted class mixture at the locally normalized distance `z = d / √((ρ_i + ε)(ρ_j + ε))`. The -/// local scales enter as detached measurements. The gradient flows through `d` only, so `dz/dd` is -/// a per-pair constant. +/// local scales enter as detached measurements, and the gradient flows through `d` only: +/// `dz/dd = 1 / √((ρ_i + ε)(ρ_j + ε))` is a per-pair constant. /// /// # Panics /// /// This panics when an edge references a row outside the frame. The batch and the frame come -/// from one assembly, so a mismatch is a wiring defect. +/// from one assembly, and a mismatch is therefore a wiring defect. pub(crate) fn relation_term( frame: ScaledFrame<'_, N>, batch: &[RelationEdges], @@ -305,7 +316,7 @@ pub(crate) struct BatchAnchor { /// Validated support-term constants. /// -/// `threshold` is the Huber threshold on the normalized residual; `epsilon` both guards the radius +/// `threshold` is the Huber threshold on the normalized residual. `epsilon` both guards the radius /// division and smooths the distance at coincidence. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct SupportOptions { @@ -387,6 +398,7 @@ impl SupportTargets { .iter() .map(|anchor| anchor.weight) .collect::>(); + Some(Self { rows: Tensor::from_data(TensorData::new(rows, [count]), device), targets: Tensor::from_data(TensorData::new(targets, [count, 2]), device), @@ -401,13 +413,13 @@ impl SupportTargets { /// The value is `scale · Σ_i weight_i · huber(‖y_i - target_i‖ / (radius_i + ε), threshold)`, /// differentiable through `coordinates`. /// -/// Smoothing replaces the Euclidean distance with `√(d^2 + ε^2) - ε`. The smoothed form is exact at +/// Smoothing replaces the Euclidean distance with `√(d² + ε²) - ε`. The smoothed form is exact at /// zero and stays within `ε` of the true distance everywhere. Its gradient is well defined and zero -/// at coincidence. Anchored nodes start exactly on their targets, so the unsmoothed square root -/// would differentiate at its singular point on the first step. +/// at coincidence. Anchored nodes start exactly on their targets, where the unsmoothed square root +/// is singular, and the first step would differentiate it there. /// -/// Every anchor row must index into `coordinates`; anchors and coordinates come from one batch -/// assembly, so an out-of-range row is a wiring defect the backend's row selection rejects. +/// Every anchor row must index into `coordinates`. Anchors and coordinates come from one batch +/// assembly, and an out-of-range row is a wiring defect. pub(crate) fn support_term( coordinates: &Tensor, targets: &SupportTargets, diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/objective/mod.rs b/libs/@local/graph/atlas/src/salt/projector/loss/objective/mod.rs index 51d73ff966d..395732da4f5 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/objective/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/objective/mod.rs @@ -15,16 +15,31 @@ //! strength multiplier. The class masses stay out because they weight the penalty family's class //! energies rather than unit mass. The composite objective's term coefficient stays out because //! composition carries no per-unit structure. The step factor selects the canonical condition and -//! never multiplies, so the canonical coordinates handed to this term are the canonical step's -//! field and no lens factor exists here. The force-pruning threshold decides population +//! never multiplies: the canonical coordinates handed to this term are the canonical step's +//! field, and no lens factor exists here. The force-pruning threshold decides population //! membership before any unit reaches this module. //! //! The treatment activation scales every force this term emits and never its reading. At zero -//! activation the same arithmetic runs over the same units and adds exactly zero to every -//! gradient, which is what lets a reference replicate run the same law with the target code path -//! live rather than removed. +//! activation the same arithmetic runs over the same units and adds exactly zero to every gradient +//! channel whenever every unit's mass, penalty slope, zero slope and fitted-scale slope are finite, +//! which is what lets a reference replicate run the same law with the target code path live rather +//! than removed. The other factors the force meets are the canonical slope, the distances and the +//! coordinate differences. The canonical slope is finite at every input. The fields prove each +//! coordinate finite and nothing about the differences: the estimator subtracts in `f32` and +//! squares and sums the differences in `f32`, and endpoints such as `(f32::MAX, 0)` and +//! `(-f32::MAX, 0)` read an infinite difference and an infinite computed distance. The +//! fitted-scale slope `d_c/σ₀` is infinite for an overflowed canonical distance, and the +//! differences, the distances and that slope are finite for finite computed distances. The force +//! is the activation times the unit's mass times the penalty slope, multiplied left to right, and +//! a zero activation against an infinite hinge slope or an infinite mass reads `0 · ∞`, which is +//! NaN rather than zero. The scale pull multiplies the force by the fitted-scale slope, and a zero +//! force against an infinite one reads NaN while the canonical branch skips that overflowed +//! distance. The zero-side deposit, taken for a finite positive computed zero distance, multiplies +//! the force by the zero slope, which is `−∞` for a ruler at or below `2⁻¹²⁸`, the bound +//! [`TargetUnit::ruler`] states: a zero force against it reads NaN as well. The direct coordinate +//! deposits skip a zero or an overflowed distance outright, at any force. //! -//! The fitted alignment scale is live, so its adjoint is real force. The term accumulates the +//! The fitted alignment scale is live, and its adjoint is real force. The term accumulates the //! pull on the scale across units and returns it, and [`fan_scale_pull`] carries that pull into //! the gauge anchors' coordinates through the fit's exact adjoints. The direct coordinate //! channels and the scale channel together are the estimator's whole derivative, and dropping @@ -54,8 +69,9 @@ use crate::{ /// variant here and its derivations at the match arms the compiler names. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum UnitLaw { - /// One unit per admitted force-bearing link instance, weighted by the released census - /// factors - the released pipeline's own unit of account. + /// One unit per admitted force-bearing link instance, weighted by the released census factors. + /// + /// This is the released pipeline's own unit of account. #[cfg_attr( not(test), expect( @@ -69,7 +85,7 @@ pub(crate) enum UnitLaw { /// One declared unit of the target estimand, drawn into a batch. /// -/// The endpoint rows speak the coordinate domain the term evaluates against. The ruler is the +/// The endpoint rows index the coordinate domain the term evaluates against. The ruler is the /// pair's frozen band-reference scale, gathered from the frozen table before re-indexing. The /// weight is the unit's declared mass, aggregated over the unit's instances as the unit law /// declares. The inclusion probability is the draw law's full first-order probability for this @@ -81,6 +97,12 @@ pub(crate) struct TargetUnit { /// The unit's target row. pub target: N, /// The pair's frozen ruler `σ₀(e)`. + /// + /// The zero slope's `f32` reciprocal needs a ruler above `2⁻¹²⁸`. A denominator of the checked + /// [`FrozenRuler`](crate::salt::projector::scale::frozen::FrozenRuler) lies above that bound. + /// The freeze admits `ε` only with `ε² ≥ 2⁻¹⁴⁹`, which puts `ε` above `2⁻⁷⁵`. Every stored + /// `ρ₀ + ε` is at least `ε`, and their geometric mean rounds within the operands' range. A + /// directly constructed unit owes the bound on its own. pub ruler: Positive, /// The unit's weight `w(e)`. pub weight: DNonNegative, @@ -105,11 +127,11 @@ pub(crate) const fn released_weight( /// The released relation draw law, priced per unit. /// /// The released sampler selects relation types uniformly without replacement and then selects -/// distinct edges uniformly without replacement inside each chosen type, so every drawn unit -/// appears exactly once and the deduplicated-set estimator form applies. Under that law a unit's -/// full first-order inclusion probability factors into the group's selection probability times -/// the within-group selection probability, and [`CappedDrawLaw::inclusion`] evaluates exactly -/// that product. +/// distinct edges uniformly without replacement inside each chosen type. Every drawn unit +/// therefore appears exactly once, and the deduplicated-set estimator form applies. Under that +/// law a unit's full first-order inclusion probability factors into the group's selection +/// probability times the within-group selection probability, and [`CappedDrawLaw::inclusion`] +/// evaluates exactly that product. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct CappedDrawLaw { drawn: NonZero, @@ -126,7 +148,7 @@ impl CappedDrawLaw { /// # Panics /// /// This panics when more groups are drawn than exist. The counts come from one sampler call, - /// so a violation is a wiring defect. + /// and a violation is therefore a wiring defect. #[must_use] pub(crate) fn new(drawn: NonZero, total: NonZero, cap: NonZero) -> Self { assert!( @@ -140,7 +162,7 @@ impl CappedDrawLaw { /// Prices one unit's inclusion probability `π(e) = (g/G) · (min(cap, M)/M)`. /// /// `group_size` is the unit's relation type's admitted instance count `M`. A type no larger - /// than the cap contributes all its edges when selected, so its within-group factor is one. + /// than the cap contributes all its edges when selected, and its within-group factor is one. #[expect( clippy::cast_precision_loss, reason = "group counts stay far below f64's exact-integer range for ratio purposes" @@ -151,7 +173,7 @@ impl CappedDrawLaw { let within = self.cap.get().min(group_size.get()) as f64 / group_size.get() as f64; // In domain with no check: each ratio is a positive quotient of a positive numerator - // by a bound at least as large, so both round inside (0, 1] and their product cannot + // by a bound at least as large. Both round inside (0, 1], and their product cannot // cross either endpoint. PositiveUnitFraction::new_unchecked(group * within) } @@ -162,15 +184,16 @@ impl CappedDrawLaw { pub(crate) struct TargetReading { /// The estimand estimate `L̂`, never scaled by the activation. /// - /// The reading stays live at zero activation, so a reference replicate still reads what the + /// The reading stays live at zero activation, where a reference replicate still reads what the /// target term would score. The composite objective's value contribution is the activation - /// times this reading. The domain is signed: under the [`Penalty::Identity`] shape a - /// satisfied unit's negative violation subtracts value. + /// times this reading. The domain is signed: under the [`Penalty::Identity`] shape a satisfied + /// unit's negative violation subtracts value. pub estimand: Finite, /// The accumulated pull on the fitted scale, activation-scaled like the gradient fields. /// - /// Non-negative by construction: the force and the fitted-scale slope both are, so every - /// accumulated term is. [`fan_scale_pull`] carries it into the gauge anchors' coordinates. + /// Non-negative by construction: the force and the fitted-scale slope both are, and every + /// accumulated term is their product. [`fan_scale_pull`] carries it into the gauge anchors' + /// coordinates. pub scale_pull: DNonNegative, } @@ -180,9 +203,19 @@ pub(crate) struct TargetReading { /// the fitted scale and the margin, the penalty is the declared `φ`, the population weight is /// the split-time total `W`, and the activation is the treatment coefficient `λ`. /// -/// The population weight's domain is strictly positive because an empty population resolves into -/// the vacuous-record taxonomy at split time, before any fit exists to evaluate. The division by -/// `W` is therefore total here by construction of the caller's split, not by a runtime guard. +/// The population weight's domain is strictly positive because an empty population makes the run +/// vacuous at split time, before any fit exists to evaluate, and positivity alone does not make +/// the mass quotient total. The divisor is the `f64` product `W·π(e)`, which `f64` +/// multiplication rounds to zero for an exact product at or below `2⁻¹⁰⁷⁵`. The quotient has a +/// nonzero divisor exactly when that product remains a representable positive value, a condition +/// the weight's own domain cannot guarantee: `2⁻¹⁰⁷⁴`, the smallest positive `f64`, is an +/// admitted population weight whose product with an inclusion of one half is zero. A nonzero +/// divisor does not bound the quotient: with a unit weight of one, `W = 1` and an inclusion of +/// `2⁻¹⁰⁷⁴` (admitted by [`PositiveUnitFraction::new`]), the divisor is a representable positive +/// and the quotient reads `+∞`. The mass is finite exactly when the divisor is a representable +/// positive and the quotient rounds inside the finite `f64` range. Against a zero divisor the mass +/// reads infinite for a positive unit weight and NaN for a zero one. An infinite mass makes every +/// fold term it enters non-finite, and the folds' finish refuses the reading in either case. #[derive(Debug, Copy, Clone)] pub(crate) struct TargetEstimator { energy: ContrastEnergy, @@ -212,30 +245,76 @@ impl TargetEstimator { /// /// Adds `activation · w(e)/(W·π(e)) · φ′(v(e))` times each live partial's direction to the /// two fields and returns the reading. The declared penalty evaluates value and slope in one - /// implementation, finite at every finite violation by construction. Its choice is decision - /// 4's, and the ruled subgradient keeps corrective force at a zero violation unless a - /// positive margin already makes distance equality a nonzero violation, which admission - /// enforces against the declared pair. + /// implementation, finite at every finite working-precision violation by construction. Either + /// its slope is nonzero at a zero violation, or a positive margin makes distance equality a + /// nonzero violation. Admission enforces that pairing through [`Penalty::dead_at_equality`]. + /// + /// A coincident endpoint pair on either side has no direction on that side. That side folds a + /// zero contribution while the value still counts, matching the contract of + /// [`ContrastEnergy::evaluate`], and the branch skips the deposit outright, at any force. The + /// same branch skips a distance whose `f32` computation overflowed to `+∞`: + /// `NonNegative::positive` is `Positive::new`, which refuses zero and infinity alike. A + /// canonical coincidence also reads a zero scale slope, whose product with a finite force adds + /// no pull. Against an infinite or NaN force the product is NaN, and the scale fold refuses it + /// at its finish. /// - /// A coincident endpoint pair on either side has no direction on that side, so that side - /// folds a zero contribution while the value still counts, matching the violation core's - /// contract. A canonical coincidence also reads a zero scale slope, so no pull arrives from - /// it either. + /// The deposits themselves pass no check. Each multiplies the force by the side's slope over + /// its distance and by the coordinate difference. The canonical slope is finite at every input. + /// The distances and the differences compute in `f32` from coordinates the fields prove finite + /// one by one: the subtraction overflows for endpoints such as `(f32::MAX, 0)` and + /// `(-f32::MAX, 0)`, and the square-and-sum for a difference of `2⁶⁴` or more. An overflowed + /// distance skips its branch, and a taken branch therefore has a finite positive distance and + /// finite differences. The zero slope is `−∞` for a ruler at or below `2⁻¹²⁸`, the bound + /// [`TargetUnit::ruler`] states. Against that slope, on a taken zero-side branch, a positive + /// finite force deposits infinities along the nonzero components of the zero difference and NaN + /// along the zero ones, and a zero force deposits NaN along both. The finishes read the + /// estimand and the scale pull, neither of which the zero slope enters, and with both in domain + /// the call returns `Ok` while the zero field holds those entries. /// /// # Errors /// /// Returns [`Diverged`] carrying the diverged fold's raw value when an accumulated reading lies - /// outside its domain. The estimand's storage is single width, so a fold whose finished value - /// overflows the narrowing to working precision is refused the same way, carrying the - /// double-width value. The folds are unbounded and data-dependent, an overflowed violation - /// alone injecting +∞, so a diverged reading is an expected numerical refusal that the caller + /// outside its domain, or when the finished estimand overflows the narrowing to its + /// single-width storage (the carried raw value is then the double-width one). The folds are + /// unbounded and data-dependent, and the refusal happens at the folds' finish, after every + /// unit has contributed. The estimand fold checks the accumulated products of mass and + /// penalty value, never the penalty's own output. That output is infinite for a violation + /// overflowed to `+∞` under either variant. For one overflowed to `−∞` it is `−∞` under + /// [`Penalty::Identity`] and the zero branch's `(0, 0)` under [`Penalty::QuadraticHinge`], + /// which adds nothing while the mass is finite. A positive mass carries an infinite value into + /// the sum as that infinity, and a zero mass, which a zero unit weight produces, carries it as + /// NaN, the product `0 · ∞`. Under [`Penalty::Identity`] the opposite infinities of two units + /// also sum to NaN. A divisor `W·π(e)` rounded to zero, or a quotient that overflows `f64` + /// over a representable divisor, reads a non-finite mass, and every term it enters is + /// non-finite, the hinge's zero value included. The finish refuses each of these sums. The + /// force takes the same products: an infinite hinge slope deposits non-finite coordinate + /// gradients, infinite along a nonzero difference component and NaN along a zero one or at a + /// zero activation or a zero mass. The coordinate fields keep every deposit made before a + /// refusal, these included. No finish reads the coordinate fields, and a non-finite deposit + /// alone refuses nothing. With both folds in domain, a ruler at or below `2⁻¹²⁸` returns `Ok` + /// and non-finite zero-side deposits for every unit whose computed zero distance is finite and + /// positive. A zero or overflowed zero distance takes no zero-side branch. An overflowed + /// canonical distance reads an infinite fitted-scale slope and, against a finite zero + /// distance, a `+∞` violation: both folds read non-finite values, and the finish refuses the + /// reading under either penalty. An overflowed zero distance against a finite aligned distance + /// `s·d_c` reads a `−∞` violation and takes no zero-side branch: [`Penalty::Identity`] refuses + /// the value and the hinge folds it as nothing. An infinite aligned distance against an + /// infinite zero distance reads NaN, the difference `∞ − ∞`, whether the canonical distance + /// overflowed or the product `s·d_c` alone did. [`Penalty::Identity`] carries the NaN into the + /// estimand fold, which refuses it. The hinge takes its `(0, 0)` branch. The estimand fold + /// reads zero for a finite mass and the force is zero. The reading then rests on the scale + /// fold, where that force meets the fitted-scale slope. With the canonical distance overflowed + /// the slope is infinite, the product reads NaN, and the scale fold refuses. With the canonical + /// distance finite and the product alone overflowed, the slope is finite, the scale fold adds + /// nothing, the canonical branch deposits the zero force along its finite distance, and the + /// call returns `Ok`. A diverged reading is an expected numerical refusal that the caller /// owns. /// /// # Panics /// /// This panics when the canonical and zero fields cover different row counts, or when a unit /// references a row outside them. Fields, units, and coordinates come from one batch assembly - /// over one forward pass, so a mismatch is a wiring defect. + /// over one forward pass, and a mismatch is therefore a wiring defect. #[expect( clippy::panic_in_result_fn, reason = "a canonical/zero row mismatch is a wiring defect, not a recoverable error" @@ -261,8 +340,8 @@ impl TargetEstimator { let activation = self.activation.widen(); // Accumulated in double precision, products included. The mass and force factors are - // unbounded and data-dependent, so both folds run as derivations and make their one - // claim at the reading's construction. + // unbounded and data-dependent. Both folds run as derivations and make their one claim + // at the reading's construction. let mut estimand = Derivation::::ZERO; let mut scale_pull = Derivation::::ZERO; @@ -270,6 +349,9 @@ impl TargetEstimator { let (source, target) = (unit.source, unit.target); let canonical_difference = canonical[source] - canonical[target]; let zero_difference = zero[source] - zero[target]; + // `length` squares and sums the `f32` differences: an overflow escapes to `+∞` here + // and asserts with debug assertions enabled, ahead of every fold. The energy's + // `div_wide` carries an escaped canonical distance into the fitted-scale slope as `+∞`. let canonical_distance = canonical_difference.length(); let zero_distance = zero_difference.length(); @@ -278,8 +360,10 @@ impl TargetEstimator { .evaluate(unit.ruler, canonical_distance, zero_distance); let (value, slope) = self.penalty.evaluate(f64::from(evaluation.violation)); - // The unit's estimator mass w(e)/(W·π(e)), raw in flight. The inclusion divisor is - // total by type, and the quotient's claim waits for the folds' finish. + // The unit's estimator mass w(e)/(W·π(e)), raw in flight. The divisor is a `DPositive` + // product through the unchecked constructor, which rejects a product rounded to zero + // only under debug assertions. A build without them divides by that zero. The + // quotient's claim waits for the folds' finish. let mass = unit.weight.get() / (denominator * unit.inclusion); estimand = Derivation::::raw(mass) .mul_add(Derivation::::raw(value), estimand); @@ -298,7 +382,10 @@ impl TargetEstimator { if let Some(distance) = zero_distance.positive() { // dv/dy_source = zero_slope · (y_source - y_target)/d₀, the honest reward for - // inflating a zero distance, held by the band projection and never hidden. + // inflating a zero distance, held by the band projection and never hidden. The + // branch runs for a finite positive computed distance alone: `positive()` refuses + // zero and an escaped `+∞`. The slope is `−∞` for a ruler at or below 2⁻¹²⁸ + // (`Positive::recip` divides in `f32`), and `add` deposits the product unchecked. let gradient = DVec2::from(zero_difference) * (force * f64::from(evaluation.zero_slope) / distance.widen()); zero_field.add(source, gradient); @@ -306,7 +393,7 @@ impl TargetEstimator { } } - // The folds are unbounded, so the overflow window is lawful input and the narrow is the + // The folds are unbounded: the overflow window is lawful input, and the narrow is the // checked form rather than the escape. let estimand = estimand.finish()?; @@ -328,8 +415,8 @@ impl TargetEstimator { /// # Panics /// /// This panics when the anchor rows and the fit disagree about the anchor count, or when an -/// anchor row lies outside a field. The rows and the fit come from one gauge, so a mismatch is a -/// wiring defect. +/// anchor row lies outside a field. The rows and the fit come from one gauge, and a mismatch is +/// therefore a wiring defect. pub(crate) fn fan_scale_pull( pull: DNonNegative, fit: &GaugeFit, diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/objective/tests.rs b/libs/@local/graph/atlas/src/salt/projector/loss/objective/tests.rs index d5cce61e460..6f0a5664cd9 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/objective/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/objective/tests.rs @@ -1,8 +1,8 @@ //! Certificates for the target objective's batch term. //! -//! Dyadic fixtures produce exactly representable readings, so estimands, forces, and the -//! scale pull assert exact contracts, and the finite-difference certificates bound their -//! quotients around them. +//! Dyadic fixtures produce exactly representable readings: estimands, forces and the scale pull +//! assert exact contracts. The finite-difference certificates compare the gradient fields against +//! central differences of an `f64` mirror of the estimand on scattered fixtures. #![expect( clippy::float_cmp, @@ -25,6 +25,7 @@ use crate::{ salt::projector::gauge::{DuplicateClassId, GaugeAnchors}, }; +/// A target unit between two rows with the given ruler, weight and inclusion probability. fn unit( source: u64, target: u64, @@ -42,6 +43,9 @@ fn unit( } } +/// Builds a target estimator over a contrast energy of the given scale and margin. +/// +/// The estimator takes `penalty`, the population weight and the activation. fn estimator( scale: f32, margin: f32, @@ -60,12 +64,15 @@ fn estimator( ) } +/// A fresh pair of zeroed gradient fields (canonical, zero) over `rows`. fn fields(rows: usize) -> (GradientField, GradientField) { (GradientField::new(rows), GradientField::new(rows)) } -/// A two-unit fixture whose coordinates, units, and `W = 4`, `s = 2`, `m = 0.25` constants -/// land every reading and every gradient entry on exactly representable values. +/// Builds the two-unit dyadic fixture. +/// +/// Its coordinates, units, and `W = 4`, `s = 2`, `m = 0.25` constants make every reading and every +/// gradient entry exactly representable. fn dyadic_fixture() -> ([Vec2; 4], [Vec2; 4], [TargetUnit; 2]) { let canonical = [ Vec2::new(0.0, 0.0), @@ -84,6 +91,10 @@ fn dyadic_fixture() -> ([Vec2; 4], [Vec2; 4], [TargetUnit; 2]) { (canonical, zero, units) } +/// Matches every dyadic-fixture reading and gradient entry to its hand-computed value. +/// +/// On the dyadic fixture the estimand, scale pull and every entry of both gradient fields equal +/// their hand-computed values exactly. #[test] fn the_reading_and_the_fields_are_exact_on_a_dyadic_batch() { let (canonical, zero, units) = dyadic_fixture(); @@ -118,6 +129,10 @@ fn the_reading_and_the_fields_are_exact_on_a_dyadic_batch() { assert_eq!(zero_entries[NodeRowId::new(3)], DVec2::new(-0.5, 0.0)); } +/// Scales every force and the pull with the activation and leaves the estimand unchanged. +/// +/// Activations of zero and two leave the estimand unchanged, zero activation yields exactly zero +/// force everywhere, and a doubled activation exactly doubles every force and the scale pull. #[test] fn the_activation_scales_every_force_and_never_the_reading() { let (canonical, zero, units) = dyadic_fixture(); @@ -146,7 +161,7 @@ fn the_activation_scales_every_force_and_never_the_reading() { assert_eq!(readings[1].estimand, readings[0].estimand); assert_eq!(readings[2].estimand, readings[0].estimand); - // Zero activation runs the same fold and lands exactly zero force everywhere. + // Zero activation runs the same fold and yields exactly zero force everywhere. assert_eq!(readings[1].scale_pull, 0.0); for row in 0..4 { let row = NodeRowId::new(row); @@ -169,6 +184,11 @@ fn the_activation_scales_every_force_and_never_the_reading() { } } +/// Recovers the declared mean under a capped draw law through the full inclusion, not `G/g`. +/// +/// Under a capped one-type-per-batch draw law, dividing by the full inclusion probability makes the +/// expected reading equal the declared mean, whereas the group factor `G/g` yields the per-type +/// clipped objective instead. #[test] fn the_full_inclusion_divisor_is_unbiased_where_the_group_factor_is_not() { // Relation type A holds {a1, a2} and type B holds {b1}, with one type drawn per batch @@ -252,6 +272,11 @@ fn the_full_inclusion_divisor_is_unbiased_where_the_group_factor_is_not() { assert_ne!(clipped_expectation, f64::from(declared)); } +/// Zeroes the field on a pair's coincident side and the pull only at canonical coincidence. +/// +/// A coincident canonical pair contributes its value with zero canonical field and zero pull, and a +/// coincident zero pair contributes its value with zero field on that side while the pull stays +/// live. #[test] fn a_coincident_side_counts_its_value_and_folds_no_pull() { // Canonical coincidence: the value reads, the canonical field and the pull stay zero. @@ -307,6 +332,10 @@ fn a_coincident_side_counts_its_value_and_folds_no_pull() { ); } +/// Returns a diverged infinite reading for a minimum-positive ruler instead of panicking. +/// +/// A minimum-positive ruler overflows the violation to infinity in working precision and the +/// estimator returns a diverged reading whose raw value is infinite rather than panicking. #[test] fn overflowed_violation_diverges() { // A minimum-positive ruler overflows the violation to +∞ in working precision while every @@ -331,10 +360,10 @@ fn overflowed_violation_diverges() { #[test] fn narrow_overflow_diverges() { - // A tiny ruler lands the violation near 8·10³⁰ in working precision, and the hinge squares - // it to an estimand near 3.2·10⁶¹ at mass 0.5: finite in double width, past the f32 range - // of the reading's storage. The checked narrow refuses with the double-width value instead - // of unwinding or storing ±∞. + // A tiny ruler takes the violation to about 8·10³⁰ in working precision, and the hinge + // squares it to an estimand of about 3.2·10⁶¹ at mass 0.5: finite in double width, past the + // f32 range of the reading's storage. The checked narrow refuses with the double-width value + // instead of unwinding or storing ±∞. let canonical = [Vec2::new(0.0, 0.0), Vec2::new(4.0, 0.0)]; let zero = [Vec2::new(0.0, 0.0), Vec2::new(0.0, 0.0)]; let (mut canonical_field, mut zero_field) = fields(2); @@ -353,6 +382,10 @@ fn narrow_overflow_diverges() { assert!(diverged.raw > f64::from(f32::MAX)); } +/// Reads `CappedDrawLaw::inclusion` as the exact product of its two probabilities. +/// +/// `CappedDrawLaw::inclusion` is the exact product of the type-selection and edge-retention +/// probabilities: `1/16`, `3/4` and `1` on the three fixtures. #[test] fn the_draw_law_prices_inclusion_as_the_exact_product() { let law = CappedDrawLaw::new(nz!(1), nz!(4), nz!(2)); @@ -367,6 +400,11 @@ fn the_draw_law_prices_inclusion_as_the_exact_product() { assert_eq!(law.inclusion(nz!(3)).get(), 1.0); } +/// Reads `released_weight` as `confidence · normalization · strength` with its zero fold. +/// +/// `released_weight` is `confidence · normalization · strength` (`0.375` on the fixture), folds +/// zero confidence to zero force, and applies a unit strength multiplier while the strength head is +/// off. #[test] fn the_released_weight_is_the_retained_factor_census() { let confidence = UnitFraction::new(0.5).expect("one half lies inside [0, 1]"); @@ -430,6 +468,10 @@ fn mirror_estimand( total } +/// Matches both gradient fields to central finite differences at fixed scale. +/// +/// Every component of both gradient fields agrees with a central finite difference of the estimand +/// at fixed scale, for both penalties, on a four-row scattered fixture. #[test] fn field_partials_match_finite_differences_at_a_fixed_scale() { let canonical = [ @@ -563,6 +605,10 @@ fn mirror_scale(source: &[(f64, f64)], target: &[(f64, f64)]) -> f64 { dot.hypot(perp) / variance } +/// Matches the total derivative through the field and live scale channels to finite differences. +/// +/// With the gauge scale refit live over the anchor rows, the total derivative through the field and +/// scale channels matches finite differences on rows that move the fit, bear units, or both. #[test] fn the_scale_channel_completes_the_derivative_through_a_live_refit() { // Rows zero through three anchor the gauge, rows four and five only bear units. diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/penalty.rs b/libs/@local/graph/atlas/src/salt/projector/loss/penalty.rs index 787bf48bdb6..a13758aa856 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/penalty.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/penalty.rs @@ -1,17 +1,18 @@ -//! The sanctioned penalty family over the contrast violation. +//! The penalty family over the contrast violation. //! -//! Decision 4 selects the production penalty. This module carries the family as a closed set of -//! shapes, each computing its value and exact derivative in one implementation, so the slope a -//! gradient deposit consumes is the derivative of the value the estimand records. A caller-supplied -//! callback could pair any value with any claimed slope, and nothing downstream could tell the pair -//! from a derivative. A ruled shape outside the family arrives as a new variant rather than as a -//! callback. +//! This module carries the family as a closed set of shapes, each computing its value and exact +//! derivative in one implementation. The slope a gradient deposit consumes is therefore the +//! derivative of the value the estimand records. A caller-supplied callback could pair any value +//! with any claimed slope, and nothing downstream could tell the pair from a derivative. A new +//! shape arrives as a new variant rather than as a callback. /// The penalty `φ`, mapping a contrast violation to its value and exact derivative. /// -/// Both readings evaluate in double precision, and no variant divides. Every violation widened -/// from the working `f32` precision reads finite, because the widest such value squares inside -/// `f64`'s range. +/// Both readings evaluate in double precision, and no variant divides. Every finite violation +/// widened from the working `f32` precision reads finite under both variants, because the widest +/// such value, `f32::MAX`, squares inside `f64`'s range. The guarantee is that narrow: a finite +/// `f64` violation above `√f64::MAX`, about `1.34·10¹⁵⁴`, overflows the quadratic hinge's square, +/// and an infinite or NaN violation follows the branches [`evaluate`](Self::evaluate) states. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum Penalty { /// `φ(v) = v` with slope `1` everywhere. @@ -29,8 +30,8 @@ pub(crate) enum Penalty { Identity, /// `φ(v) = max(0, v)²` with slope `2·max(0, v)`: smooth at the hinge and dead below it. /// - /// The slope vanishes at a zero violation, so this shape keeps corrective force at distance - /// equality only through a positive margin. Admission enforces that pairing. + /// The slope vanishes at a zero violation. This shape keeps corrective force at distance + /// equality only through a positive margin, and admission enforces that pairing. #[cfg_attr( not(test), expect( @@ -46,11 +47,15 @@ impl Penalty { /// Evaluates `(φ(v), φ′(v))` at the violation. /// /// The violation arrives in double precision, and a caller holding a working-precision - /// reading widens it visibly at the call. The pair is raw like its operand, and a - /// non-finite violation flows through to the consumer's own check. The value is signed - - /// under [`Identity`](Self::Identity) a satisfied pair's negative violation subtracts - /// value - and the slope is non-negative at every violation, because both declared - /// shapes are nondecreasing. + /// reading widens it visibly at the call. The pair is raw like its operand, and a non-finite + /// violation takes the variant's own branch. [`Identity`](Self::Identity) returns the + /// violation itself with slope one: `+∞`, `−∞` and NaN pass through as the value. + /// [`QuadraticHinge`](Self::QuadraticHinge) squares only a violation that compares greater + /// than zero: `+∞` returns `(+∞, +∞)`, while `−∞` and NaN take the zero branch and return + /// `(0, 0)`, indistinguishable from a satisfied pair. The value is signed - under + /// [`Identity`](Self::Identity) a satisfied pair's negative violation subtracts value - and + /// the slope is non-negative at every violation, because both declared shapes are + /// nondecreasing. #[must_use] pub(crate) fn evaluate(self, violation: f64) -> (f64, f64) { match self { @@ -62,9 +67,8 @@ impl Penalty { /// Returns whether the derivative vanishes at a zero violation. /// - /// The ruled shape rule pairs such a penalty with a positive margin, so distance equality still - /// carries corrective force. Admission reads this to enforce the pairing, and the child module - /// locks the answer to the evaluated slope. + /// A penalty that is dead at equality pairs with a positive margin, which keeps corrective + /// force at distance equality. Admission reads this to enforce the pairing. #[must_use] pub(crate) const fn dead_at_equality(self) -> bool { match self { @@ -83,6 +87,7 @@ mod tests { use super::Penalty; + /// Both penalty variants, for tests that must apply to every variant in the family. const FAMILY: [Penalty; 2] = [Penalty::Identity, Penalty::QuadraticHinge]; /// Reads the raw pair for comparison against reference pairs. @@ -118,6 +123,7 @@ mod tests { } } + /// `dead_at_equality` is true exactly when the slope at zero violation is zero. #[test] fn dead_at_equality_agrees_with_the_evaluated_slope() { for penalty in FAMILY { @@ -125,6 +131,7 @@ mod tests { } } + /// Both penalties return finite values and slopes at `±f32::MAX`. #[test] fn the_widest_violation_still_reads_finite() { for penalty in FAMILY { diff --git a/libs/@local/graph/atlas/src/salt/projector/loss/tests.rs b/libs/@local/graph/atlas/src/salt/projector/loss/tests.rs index 108fdc85934..31016a7b8d5 100644 --- a/libs/@local/graph/atlas/src/salt/projector/loss/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/loss/tests.rs @@ -44,8 +44,12 @@ fn proven(points: &[Vec2]) -> &FinitePointField { FinitePointField::new_unchecked(IdSlice::from_raw(points)) } +/// The CPU device the support-term tensors live on, resolved once. static DEVICE: LazyLock = LazyLock::new(|| Device::Cpu.pin(0).resolve()); +/// Builds an affinity curve from literature parameters `a` and `b`. +/// +/// The tests keep both positive and finite. #[expect( clippy::min_ident_chars, reason = "`a` and `b` are the affinity curve's literature parameter names" @@ -54,6 +58,9 @@ fn curve(a: f32, b: f32) -> AffinityCurve { AffinityCurve::new(a, b).expect("test curve parameters are positive and finite") } +/// Builds an affinity energy over `curve(a, b)` with the log guard `epsilon`. +/// +/// The tests keep `b` within the objective's exponent bound. #[expect( clippy::min_ident_chars, reason = "`a` and `b` are the affinity curve's literature parameter names" @@ -66,6 +73,7 @@ fn affinity_energy(a: f32, b: f32, epsilon: f32) -> AffinityEnergy { .expect("the test exponent satisfies the objective bound") } +/// A proximal energy with the given radius and temperature. fn proximal(radius: f32, temperature: f32) -> ProximalEnergy { ProximalEnergy::new( NonNegative::new(radius).expect("the test radius is non-negative"), @@ -73,6 +81,7 @@ fn proximal(radius: f32, temperature: f32) -> ProximalEnergy { ) } +/// A coincident energy with the given radius and Huber threshold. fn coincident(radius: f32, threshold: f32) -> CoincidentEnergy { CoincidentEnergy::new( NonNegative::new(radius).expect("the test radius is non-negative"), @@ -80,6 +89,9 @@ fn coincident(radius: f32, threshold: f32) -> CoincidentEnergy { ) } +/// Builds the relation mixture with scale guard `epsilon`. +/// +/// The classes are `coincident(0.25, 1.0)` and `proximal(1.0, 0.5)`. fn relation_energy(epsilon: f32) -> RelationEnergy { RelationEnergy::new( coincident(0.25, 1.0), @@ -89,6 +101,7 @@ fn relation_energy(epsilon: f32) -> RelationEnergy { .expect("test relation parameters are valid") } +/// A batch-row pair from two literal row numbers. fn pair(one: u32, other: u32) -> NodePair { NodePair::new(BatchRowId::new(one), BatchRowId::new(other)) } @@ -98,6 +111,9 @@ fn gradient(field: &GradientField, row: usize) -> DVec2 { field.as_slice()[BatchRowId::from_usize(row)] } +/// Builds local scales over batch rows from plain `f32` values. +/// +/// The tests keep the values finite and non-negative. fn scales(values: &[f32]) -> LocalScales { let values: Box<[NonNegative]> = values .iter() @@ -133,9 +149,6 @@ fn coordinate_difference( (above - below) / (2.0 * step) } -/// Asserts a derivative against its finite difference. -/// -/// The tolerance scales to the finite difference's own f32 conditioning. #[track_caller] /// Narrows one accumulated field component for a finite-difference comparison. #[expect( @@ -146,6 +159,9 @@ fn component(gradient: DVec2, axis: usize) -> f32 { [gradient.x(), gradient.y()][axis] as f32 } +/// Asserts an analytic derivative agrees with its central finite difference. +/// +/// The tolerance is two percent relative plus `1e-3` absolute, and `context` names the failure. #[track_caller] fn assert_derivative_close(derivative: f32, difference: f32, context: &str) { assert!( @@ -154,6 +170,10 @@ fn assert_derivative_close(derivative: f32, difference: f32, context: &str) { ); } +/// Refuses `b = 0.25` in `AffinityEnergy::new` and admits the bound `b = 0.5`. +/// +/// `AffinityEnergy::new` returns `None` for `b = 0.25`, where the coordinate gradient diverges at +/// coincidence, and `Some` at the bound `b = 0.5`. #[test] fn affinity_energy_rejects_a_shallow_exponent() { // Below `b = 0.5` the coordinate gradient diverges at coincidence. @@ -164,11 +184,14 @@ fn affinity_energy_rejects_a_shallow_exponent() { assert!(AffinityEnergy::new(curve(1.0, 0.5), positive!(0.125)).is_some()); } +/// Reads exactly zero values and `±0.25` derivative masses at `a = b = u = 1`, `ε = 0.5`. +/// +/// With `a = b = 1`, `u = 1` and `ε = 0.5`, both attraction and repulsion values are exactly zero +/// and their derivative masses are exactly `±0.25`. #[test] fn affinity_energies_match_hand_computed_dyadic_values() { - // a = 1, b = 1, u = 1: q = 0.5 exactly. With ε = 0.5 both - // logarithm arguments are exactly 1, so both values are exactly - // zero and the derivative mass is a b q^2 = 0.25 exactly. + // a = 1, b = 1, u = 1: q = 0.5 exactly. With ε = 0.5 both logarithm arguments are exactly + // 1: both values are exactly zero, and the derivative mass a·b·q² is exactly 0.25. let energy = affinity_energy(1.0, 1.0, 0.5); let (value, derivative) = energy.attraction(non_negative!(1.0)); @@ -180,6 +203,7 @@ fn affinity_energies_match_hand_computed_dyadic_values() { assert_eq!(derivative, -0.25); } +/// At `u = 0` both attraction and repulsion have finite values and exactly zero derivative. #[test] fn affinity_energies_have_zero_derivative_at_coincidence() { let energy = affinity_energy(1.577, 0.895, 0.125); @@ -193,10 +217,14 @@ fn affinity_energies_have_zero_derivative_at_coincidence() { assert_eq!(derivative, 0.0); } +/// Matches both affinity derivatives to finite differences for two exponents. +/// +/// For an integer and a fractional exponent, the attraction and repulsion derivatives agree with +/// central finite differences at five probe distances. #[test] fn affinity_derivatives_match_finite_differences() { - // Both an integer and a fractional exponent: the derivative's - // u^(b - 1) factor follows different code paths through powf. + // Both an integer and a fractional exponent: b = 1 makes the derivative's u^(b - 1) factor + // exactly one, and a fractional b exercises the general power. #[expect( clippy::min_ident_chars, reason = "`a` and `b` are the affinity curve's literature parameter names" @@ -234,6 +262,9 @@ fn affinity_derivatives_match_finite_differences() { } } +/// At the radius the proximal value is `temperature · ln 2` and the derivative exactly `0.5`. +/// +/// Far outside the derivative saturates at one, and far inside it vanishes. #[test] fn proximal_energy_matches_hand_computed_values() { // At the radius the argument is exactly zero: the value is @@ -251,6 +282,10 @@ fn proximal_energy_matches_hand_computed_values() { assert!(proximal(8.0, 0.5).evaluate(NonNegative::ZERO).1.get() < 1e-6); } +/// Matches the proximal derivative to finite differences a step above zero. +/// +/// The proximal derivative agrees with central finite differences on a grid that stays a step above +/// zero. #[test] fn proximal_derivative_matches_finite_differences() { let energy = proximal(1.0, 0.5); @@ -299,6 +334,10 @@ fn coincident_energy_matches_hand_computed_regimes() { ); } +/// Matches the coincident derivative to finite differences away from the regime kinks. +/// +/// The coincident derivative agrees with central finite differences on a grid that avoids the +/// regime kinks. #[test] fn coincident_derivative_matches_finite_differences() { let energy = coincident(1.0, 1.0); @@ -340,6 +379,10 @@ fn relation_energy_requires_ordered_radii() { ); } +/// Reads `mixture` as the exact class-weighted sum accumulated in `f64`. +/// +/// `mixture` returns exactly the class-weighted sum of the coincident and proximal values and +/// derivatives, accumulated in `f64`. #[test] fn relation_mixture_is_the_weighted_class_sum() { let energy = relation_energy(0.25); @@ -381,6 +424,10 @@ fn gradient_field_accumulates_and_resets() { assert!(field.as_slice().iter().all(|&entry| entry == DVec2::ZERO)); } +/// Reads exactly zero value and `(∓0.5, 0)` endpoint gradients for one unit-distance pair. +/// +/// One unit-distance pair under the dyadic energy yields exactly zero value and per-endpoint +/// gradients `(-0.5, 0)` and `(0.5, 0)`. #[test] fn attraction_term_matches_hand_computed_gradient() { // One unit-distance pair under the dyadic energy: value exactly @@ -403,6 +450,10 @@ fn attraction_term_matches_hand_computed_gradient() { assert_eq!(gradient(&field, 1), DVec2::from(Vec2::new(0.5, 0.0))); } +/// Points the attraction gradient along the separation and the repulsion gradient against it. +/// +/// The attraction gradient at the first endpoint points along the separation (descent pulls the +/// pair together) and the repulsion gradient points against it. #[test] fn attraction_pulls_and_repulsion_pushes() { let energy = affinity_energy(1.577, 0.895, 0.125); @@ -432,6 +483,10 @@ fn attraction_pulls_and_repulsion_pushes() { assert!(gradient(&field, 0).dot(DVec2::from(toward)).into_raw() < 0.0); } +/// Deposits exactly zero gradient for a coincident repulsion pair with positive value. +/// +/// A coincident repulsion pair has a positive value but deposits exactly zero gradient, having no +/// direction to push along. #[test] fn coincident_pair_contributes_value_but_no_gradient() { let energy = affinity_energy(1.0, 1.0, 0.125); @@ -446,8 +501,8 @@ fn coincident_pair_contributes_value_but_no_gradient() { &mut field, ); - // A coincident negative pair is maximally improbable placement, so - // its value is large - but it has no direction to push along. + // A coincident negative pair is the least probable placement: its value is large, and it + // has no direction to push along. assert!(value > 0.0); assert_eq!(gradient(&field, 0), DVec2::from(Vec2::splat(0.0))); assert_eq!(gradient(&field, 1), DVec2::from(Vec2::splat(0.0))); @@ -463,6 +518,10 @@ fn frame() -> [Vec2; 4] { ] } +/// Matches the accumulated attraction gradient over four weighted pairs to finite differences. +/// +/// The accumulated attraction gradient over four weighted pairs, one duplicated, agrees with +/// central finite differences on every row and axis. #[test] fn attraction_term_gradient_matches_finite_differences() { let energy = affinity_energy(1.577, 0.895, 0.125); @@ -501,6 +560,10 @@ fn attraction_term_gradient_matches_finite_differences() { } } +/// Matches the accumulated repulsion gradient over three weighted pairs to finite differences. +/// +/// The accumulated repulsion gradient over three weighted pairs agrees with central finite +/// differences on every row and axis. #[test] fn repulsion_term_gradient_matches_finite_differences() { let energy = affinity_energy(1.577, 0.895, 0.125); @@ -605,7 +668,7 @@ fn attraction_fixture() -> AttractionIndex { /// Wraps every group of an index with all its edges, as a sampler emitting everything would. /// /// Converts every group into the batch-local shape under the identity row map: the fixture -/// coordinates are corpus-length, so corpus rows and batch positions coincide. +/// coordinates are corpus-length, and corpus rows and batch positions therefore coincide. fn full_batch(index: &AttractionIndex) -> Vec> { let position = |row: NodeRowId| { BatchRowId::new(u32::try_from(row.as_u64()).expect("fixture rows fit the batch encoding")) @@ -630,15 +693,18 @@ fn full_batch(index: &AttractionIndex) -> Vec::new(&[valid], &*DEVICE).is_some()); } +/// Builds the three-row support fixture with anchors at rows 0 and 2. +/// +/// The support options carry a unit threshold and a `0.25` guard. fn support_fixture() -> ( Tensor, SupportTargets, @@ -799,12 +880,16 @@ fn support_fixture() -> ( (coordinates, targets, options) } +/// Matches the autodiff support gradient to the hand-derived chain rule within `1e-5`. +/// +/// The autodiff gradient of the support term agrees with the hand-derived chain rule +/// `scale · weight · min(n, threshold) · (y - t) / (√(d² + ε²) · (r + ε))` within `1e-5` on both +/// anchored rows, and the unanchored row receives none. #[test] fn support_term_gradient_matches_the_analytic_formula() { - // Independent reference: for each anchor, the hand-derived chain - // rule gives dL/dy = scale · weight · min(n, threshold) · (y - t) / - // (√(d^2 + ε^2) · (r + ε)) with n the smoothed normalized - // distance. The autodiff backward pass must agree. + // Independent reference: for each anchor, the hand-derived chain rule gives + // dL/dy = scale · weight · min(n, threshold) · (y - t) / (√(d² + ε²) · (r + ε)) with n the + // smoothed normalized distance. The autodiff backward pass must agree. let (coordinates, targets, options) = support_fixture(); let scale = 1.25; @@ -850,10 +935,14 @@ fn support_term_gradient_matches_the_analytic_formula() { assert_eq!(gradient[3], 0.0); } +/// Reads zero value and an exactly zero gradient for an anchor lying on its target. +/// +/// An anchor whose row lies exactly on its target yields a support value of zero and a defined, +/// exactly zero gradient. #[test] fn support_term_is_finite_at_exact_coincidence() { - // Anchored nodes start exactly on their targets; the smoothed - // distance keeps the gradient defined (and zero) there. + // Anchored nodes start exactly on their targets. The smoothed distance keeps the gradient + // defined (and zero) there. let coordinates: Tensor = Tensor::from_data(TensorData::new(vec![0.5_f32, -0.25], [1, 2]), &*DEVICE).require_grad(); let anchors = [BatchAnchor { diff --git a/libs/@local/graph/atlas/src/salt/projector/miner/mod.rs b/libs/@local/graph/atlas/src/salt/projector/miner/mod.rs index e3e15731f76..bfe2e56f845 100644 --- a/libs/@local/graph/atlas/src/salt/projector/miner/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/miner/mod.rs @@ -6,16 +6,32 @@ //! neighbour, two points close on the map that nothing says belong together. The same bounded //! negative energy that repels ordinary negatives repels this pair, weighted by closeness rank. //! -//! Under a conditioned model the current map is one map per lens value, so a refresh tick mines one -//! [`SpatialField`] per lens extreme and pools the frames with [`MinedFrame::pool`], where a pair -//! mined in both keeps its maximum weight. Saturation makes pooling safe. The bounded negative -//! energy exerts vanishing force on pairs far apart in a frame, so pooled pairs act only where they -//! lie close together. +//! Under a conditioned model the current map is one map per lens value. A refresh tick therefore +//! mines one [`SpatialField`] per lens extreme and pools the frames with [`MinedFrame::pool`], +//! where a pair mined in both keeps its maximum weight. Pooling relies on the repulsion's decay at +//! large distances. The bounded negative energy's force on a pair falls toward zero as the pair's +//! distance grows in a frame and, in the real model, reaches zero at no finite positive distance: +//! the term has no support cutoff. That decay is the large-distance side of a force that is not +//! monotone in the distance. At exactly zero distance the explicit branch of +//! [`AffinityEnergy::repulsion`](crate::salt::projector::loss::AffinityEnergy::repulsion) returns +//! a zero derivative, and for exponents `b > 1/2` the coordinate force also has limit zero as a +//! positive distance shrinks toward coincidence, as the [`loss`](crate::salt::projector::loss) +//! module documents. A pair mined close in one frame and far apart in the other therefore pushes in +//! the second frame with whatever force that distance leaves, and pooling adds that decayed +//! influence rather than nothing. //! -//! The spatial index is exact (a balanced kd-tree), so a query returns the exact nearest points. -//! Mining needs no recall accounting and no retry loop that widens the search when exclusions thin -//! the candidates. A node whose map neighbourhood other evidence fully explains yields an honest -//! short set, because it has no false neighbours to repel. +//! The spatial index is a kd-tree over the frame, [`SpatialField`] over [`KdTree`]. A query returns +//! up to the requested count of other rows, ascending by squared distance with ties resolved by +//! row, under the selection precision that index documents. Mining examines each row's bounded +//! candidate prefix once, `neighbours · search_margin` rows, and never widens the search when +//! exclusions thin it. A short mined set therefore means the examined candidates ran out, not that +//! the frame holds no further admissible partner: with a quota and a margin of one, rows at +//! `x = 0`, `1` and `2`, a semantic edge between the first two and none between the first and the +//! third, row zero examines row one alone, rejects it and mines nothing while row two remains +//! admissible. The single bounded query is the trade the margin buys. Every row costs one readout +//! of `neighbours · search_margin` rows and at most that many exclusion checks, and a wider margin +//! raises that cost for every row. A row whose examined prefix is dense with excluded pairs mines +//! fewer than its quota. #[cfg(test)] mod tests; @@ -37,8 +53,9 @@ use crate::{ /// Validated mining schedule and rank-weight coefficients. /// /// Per row, the miner examines the nearest `neighbours · search_margin` projected points and admits -/// up to `neighbours` of them past the exclusions; the margin is what keeps a row surrounded by its -/// own semantic cluster from starving. An admitted candidate at closeness rank `r` weighs +/// up to `neighbours` of them past the exclusions. The margin is what lets a row surrounded by its +/// own semantic cluster reach candidates past that cluster, within the one examined prefix. An +/// admitted candidate at closeness rank `r` weighs /// `maximum_weight · (1 - r / neighbours)^rank_exponent`: the nearest surviving false neighbour /// carries the full weight and the last admissible rank fades toward zero, satisfying the bounded /// rank-weight contract. @@ -99,7 +116,12 @@ impl MinerOptions { /// Computes the weight of the candidate at closeness `rank`. /// - /// Ranks lie below the quota, and rank zero carries the full bound. + /// Ranks lie below the quota, and rank zero carries the full bound. The real formula + /// `maximum_weight · (1 − rank/neighbours)^rank_exponent` is strictly positive for every rank + /// below the quota. The computed weight converts `rank` and the quota to `f32` first, and that + /// conversion can make the relative rank one: for a quota of `2²⁴ + 1` and rank `2²⁴`, both + /// integers round to `2²⁴` and the weight is zero. A small base raised to a large exponent + /// can also underflow to zero. The represented weight therefore lies in `[0, maximum_weight]`. fn weight(self, rank: usize) -> f32 { #[expect( clippy::cast_precision_loss, @@ -116,11 +138,13 @@ impl MinerOptions { } } -/// The exact 2D neighbour index over one frame's detached coordinates. +/// The 2D neighbour index over one frame's detached coordinates. /// /// Every refresh tick builds one field per lens extreme and drops it with the tick. Queries never -/// mutate the field and are thread-safe. Exactness is part of the contract: consumers account for -/// no recall. +/// mutate the field and are thread-safe. The field wraps [`KdTree`] and inherits its selection +/// precision: a readout equals a full scan where that index's documented conditions hold. The +/// miner keeps no recall accounting of its own, and the index's selection limits pass through to +/// the mined set. pub(crate) struct SpatialField<'frame, N> { tree: KdTree<'frame, N>, } @@ -132,7 +156,12 @@ where /// Indexes one frame of projected coordinates, in row order. /// /// A diverged projection never reaches the index: the frame arrives as a proven-finite - /// field from its readback boundary. + /// field from its readback boundary. Construction is [`KdTree::build`]'s. + /// + /// # Panics + /// + /// This panics when the frame holds more rows than `N` addresses, the condition + /// [`KdTree::build`] states. The field's constructor contracts to prevent it. #[must_use] pub(crate) fn new(coordinates: &'frame FinitePointField) -> Self { Self { @@ -147,11 +176,18 @@ where self.tree.points().len() } - /// Returns the `count` nearest other rows of `row`, ascending by `(squared distance, row)`. + /// Returns up to `count` other rows nearest to `row`, ascending by `(squared distance, row)`. + /// + /// The query is [`KdTree::nearest`]. The index excludes the query row and orders the rows it + /// selects under that key, and equal distances therefore come back in one order regardless of + /// tree traversal. Which rows it selects is the index's contract: [`KdTree`] states where the + /// selection equals a full scan. A `count` beyond the frame's other rows requests them all, + /// under the same selection. Every returned candidate is a potential pair partner. /// - /// The index excludes the query row and selects the exact `count`-set under that order, so - /// equal distances come back in one order regardless of tree traversal and every returned - /// candidate is a potential pair partner. + /// # Panics + /// + /// This panics when `row` is not a frame row, and when the readout's reservation of + /// `count + 1` entries exceeds the vector capacity limit. fn nearest(&self, row: N, count: NonZero) -> Vec> { self.tree.nearest(row, count) } @@ -160,10 +196,10 @@ where /// The exclusion evidence one generation mines against. /// /// The semantic graph vetoes pairs the attraction objective already pulls together (the graph is -/// symmetric, so one row's adjacency decides), and the protection evidence vetoes pairs whose links -/// veto targeted repulsion under the hard channel. Typed-separation control sets and signed-policy -/// conflicts are further exclusions the admission contract names; the initial generation has no -/// signed policies, so both sets are empty here. +/// symmetric, and one row's adjacency decides), and the protection evidence vetoes pairs whose +/// links veto targeted repulsion under the hard channel. The admission contract names two further +/// exclusions, typed-separation control sets and signed-policy conflicts. Both sets are empty +/// here, and this miner applies neither. #[derive(Debug)] pub(crate) struct HardNegativeMiner<'view, N> { semantic: SemanticGraphView<'view, N>, @@ -187,7 +223,7 @@ where /// # Panics /// /// This panics when the two views disagree about the row domain. Both artifacts come from one - /// generation, so a mismatch is a wiring defect. + /// generation, and a mismatch is therefore a wiring defect. #[must_use] pub(crate) fn new( semantic: SemanticGraphView<'view, N>, @@ -210,6 +246,10 @@ where } /// Mines one row's admissible candidates in closeness-rank order. + /// + /// One readout of `neighbours · search_margin` candidates, filtered in order until the quota + /// fills or the readout ends. The result is short when the examined candidates run out, and + /// the search never widens. fn mine_row(&self, field: &SpatialField<'_, N>, row: N) -> Vec<(N, f32)> { let quota = self.options.neighbours().get(); @@ -243,7 +283,11 @@ where /// # Panics /// /// This panics when the frame's row domain disagrees with the exclusion evidence. Both come - /// from one generation, so a mismatch is a wiring defect. + /// from one generation, and a mismatch is therefore a wiring defect. It also panics when a + /// quota exceeds the vector capacity limit: each row reserves `neighbours` entries for its + /// admitted candidates and its readout reserves `neighbours · search_margin + 1`, and a + /// reservation past `isize::MAX` bytes panics on capacity overflow. A refused allocation is + /// not recoverable here. pub(crate) fn mine(&self, field: &SpatialField<'_, N>) -> MinedFrame { assert_eq!( field.rows(), @@ -278,13 +322,17 @@ where /// One frame's mined hard negatives, grouped by anchor row. /// /// Rows keep their candidates in closeness-rank order after a mine and in ascending target order -/// after a pool; the weights ride beside the targets either way, so consumers never reconstruct +/// after a pool. Each target has its weight beside it either way, and consumers never reconstruct /// rank. #[derive(Debug, PartialEq)] pub(crate) struct MinedFrame { /// Mined counterpart rows, grouped into one run per anchor row. targets: Runs, - /// Rank weights in `(0, maximum_weight]`, one beside each target. + /// Rank weights in `[0, maximum_weight]`, one beside each target. + /// + /// The real rank-decay formula is strictly positive below the quota, and the represented `f32` + /// weight can be zero where the rank and the quota round to one `f32` integer or the power + /// underflows, as [`MinerOptions::weight`] states. weights: Box<[f32]>, } @@ -331,7 +379,7 @@ where /// # Panics /// /// This panics when the frames disagree about the row domain. Both come from one refresh tick, - /// so a mismatch is a wiring defect. + /// and a mismatch is therefore a wiring defect. #[must_use] pub(crate) fn pool(&self, other: &Self) -> Self where diff --git a/libs/@local/graph/atlas/src/salt/projector/miner/tests.rs b/libs/@local/graph/atlas/src/salt/projector/miner/tests.rs index 543485d4947..2b91c39ae05 100644 --- a/libs/@local/graph/atlas/src/salt/projector/miner/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/miner/tests.rs @@ -30,10 +30,12 @@ use crate::{ }, }; +/// A `NonZero` from a literal test count. fn nonzero(value: usize) -> NonZero { NonZero::new(value).expect("test counts are nonzero") } +/// Miner options from plain neighbour count, margin, maximum weight and rank exponent. fn options(neighbours: usize, margin: usize, maximum_weight: f32, exponent: f32) -> MinerOptions { MinerOptions::new( nonzero(neighbours), @@ -85,8 +87,9 @@ fn proximal_policy(relation: u64) -> RelationPolicy { } } -/// An instance of `relation` between `source` and `target` with the given link confidence (`None` -/// means unscored, the neutral 1). +/// Builds an instance of `relation` between `source` and `target` with a link confidence. +/// +/// `None` means unscored, the neutral 1. fn instance( edge: u64, relation: u64, @@ -107,6 +110,7 @@ fn instance( } } +/// Builds the relation indexes over `rows` from the given instances under a single proximal policy. fn relation_indexes( rows: usize, instances: Vec>, @@ -295,6 +299,7 @@ fn mined_rows_match_a_brute_force_reference() { ); } +/// With a full quota of four and unit exponent the rank weights are exactly `1, 3/4, 1/2, 1/4`. #[test] fn rank_weights_are_dyadic_at_unit_exponent() { // With five points and no exclusions, row 0 fills its quota of four, and the unit-exponent @@ -346,6 +351,7 @@ fn fully_explained_neighbourhoods_yield_honest_short_sets() { assert!(negatives.row(NodeRowId::new(3)).len() > 0); } +/// Mining the same field twice yields equal frames. #[test] fn mining_is_deterministic() { let coordinates = line_frame(); diff --git a/libs/@local/graph/atlas/src/salt/projector/mod.rs b/libs/@local/graph/atlas/src/salt/projector/mod.rs index 1c0928d23bc..306142579d7 100644 --- a/libs/@local/graph/atlas/src/salt/projector/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/mod.rs @@ -3,26 +3,33 @@ //! The fitting pipeline trains this model and the serving pipeline applies it. A residual MLP reads //! a node's normalized representation and role, modulated by a global condition vector, and //! produces one 2D coordinate per node. Training minimizes a composite of semantic, relational, and -//! support objectives over the published fitting artifacts; inference projects whole corpora +//! support objectives over the published fitting artifacts. Inference projects whole corpora //! batch-wise at a frozen condition. //! //! The map is parametric because it must extend to nodes the fit never saw. A trained checkpoint is -//! a pure function from representation to coordinate, so freshly ingested nodes project into the +//! a pure function from representation to coordinate, and freshly ingested nodes project into the //! existing frame without a refit. Rows project independently - the model reads one row's -//! representation, role, and the global condition, never its neighbours - which is what makes -//! placement idempotent and batch composition irrelevant to the result. Equal inputs therefore -//! place identically: rows sharing an exact representation and role are coincident in every -//! published map, at every lens. The condition input is the relation lens η ∈ [0, 1]: one model -//! covers the whole lens continuum, and the ladder publishes chosen steps of it instead of one -//! model per step. +//! representation, role, and the global condition, never its neighbours - and the map is a +//! function of each row's own input: mathematically, placement is idempotent and batch +//! composition does not enter the result, and rows sharing an exact representation and role share +//! one coordinate at every lens. In finite precision that identity holds up to the batched +//! forward pass, whose backend can choose different kernels for different frame shapes, and the +//! published contracts state their tolerances where it matters. The condition input is the +//! relation lens η ∈ [0, 1]: one model covers the whole lens continuum, and the ladder publishes +//! chosen steps of it instead of one model per step. //! -//! [`model`] defines the architecture and its initialization contracts; [`scale`] measures the -//! detached local radii the relation objective normalizes by; [`sample`] draws the seeded minibatch -//! populations; [`loss`] computes the composite objective and [`budget`] measures its relation -//! forces; [`miner`] finds 2D hard negatives; [`verdict`] reads the supplied human-review input; -//! [`train`] assembles minibatches and evaluates the step objective; [`artifact`] writes and -//! reopens the published checkpoint and the resume state. [`report`] observes the published -//! placement after the fact and participates in none of the above. +//! The modules and what each does: +//! +//! - [`model`] defines the architecture and its initialization contracts. +//! - [`scale`] measures the detached local radii the relation objective normalizes by. +//! - [`sample`] draws the seeded minibatch populations. +//! - [`loss`] computes the composite objective, and [`budget`] measures its relation forces. +//! - [`miner`] finds 2D hard negatives. +//! - [`verdict`] reads the supplied human-review input. +//! - [`train`] assembles minibatches and evaluates the step objective. +//! - [`artifact`] writes and reopens the published checkpoint and the resume state. +//! - [`report`] observes the published placement after the fact and participates in none of the +//! above. pub(crate) mod artifact; pub(crate) mod band; diff --git a/libs/@local/graph/atlas/src/salt/projector/model/mod.rs b/libs/@local/graph/atlas/src/salt/projector/model/mod.rs index 6598565a31a..fb0f248ec4e 100644 --- a/libs/@local/graph/atlas/src/salt/projector/model/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/model/mod.rs @@ -11,18 +11,21 @@ //! ``` //! //! `FiLM` predicts a delta from unit scale and a shift out of the condition vector: `FiLM(v, c) = -//! (1 + Δγ(c)) · v + β(c)`. Modulation sits between normalization and activation, where it gates -//! normalized features directly. Modulation placed before the block's linear and normalization -//! instead loses its scale component to the downstream LN (exactly so for a uniform gamma). -//! Condition columns are opaque to the model. The batch assembler names them, and their count is -//! the [`Architecture`]'s `condition_dimensions`. +//! (1 + Δγ(c)) · v + β(c)`. Modulation applies between normalization and activation, where it +//! scales and shifts normalized features directly. Modulation placed before the block's linear and +//! normalization would instead lose its scale component to the downstream LN: a uniform positive +//! scale cancels exactly in an ideal normalization (zero ε, no bias between the scale and the LN), +//! cancels up to the normalization's ε and the linear's bias in the actual layer, and a negative +//! uniform scale survives as a sign reversal of the centered signal. Condition columns are opaque +//! to the model. The batch assembler names them, and their count is the [`Architecture`]'s +//! `condition_dimensions`. //! //! The unit tests certify both initialization contracts: //! -//! - every residual block is the identity (its second linear and bias initialize to zero), so the +//! - every residual block is the identity (its second linear and bias initialize to zero): the //! initial model is stem plus head; -//! - `FiLM` is the identity for every condition (its linear map and bias initialize to zero), so -//! all conditions share one function before training. +//! - `FiLM` is the identity for every condition (its linear map and bias initialize to zero): all +//! conditions share one function before training. //! //! All other biases initialize to zero and all weights to scaled uniform values, except the role //! embedding, whose per-component scale matches the representation's (a unit-norm vector has @@ -144,7 +147,7 @@ impl Dimension { /// A model's parameters do not describe an architecture. /// -/// The named dimension is the first one that differs; an `actual` of zero on a bias or shift +/// The named dimension is the first one that differs. An `actual` of zero on a bias or shift /// reports the parameter as absent, a shape no present parameter can have. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct ArchitectureMismatch { @@ -189,7 +192,7 @@ impl Error for ArchitectureMismatch {} /// The projection role of a node row. /// -/// Roles distinguish what kind of thing a row is on the map; the model learns one embedding vector +/// Roles distinguish what kind of thing a row is on the map. The model learns one embedding vector /// per role and concatenates it to the representation. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum NodeRole { @@ -197,8 +200,8 @@ pub(crate) enum NodeRole { KnowledgeEntity, /// An ontology type projected as a first-class map citizen. /// - /// The role axis sizes trained models by variant count, so the variant stays in every unit - /// while only the training corpus generators construct it today. + /// The role axis sizes trained models by variant count, and the variant stays in every unit + /// even though only the training corpus generators construct it. #[cfg_attr( not(any(test, feature = "bench")), expect( @@ -210,8 +213,8 @@ pub(crate) enum NodeRole { OntologyType, /// A supported row that is neither of the above. /// - /// The role axis sizes trained models by variant count, so the variant stays in every unit - /// while only the training corpus generators construct it today. + /// The role axis sizes trained models by variant count, and the variant stays in every unit + /// even though only the training corpus generators construct it. #[cfg_attr( not(any(test, feature = "bench")), expect( @@ -227,7 +230,7 @@ impl NodeRole { /// Distinct roles: the role embedding's vocabulary size. pub(crate) const COUNT: usize = core::mem::variant_count::(); - /// This role's embedding index. + /// Returns this role's embedding index. #[inline] #[must_use] pub(crate) const fn index(self) -> u32 { @@ -235,11 +238,16 @@ impl NodeRole { } } +/// The default hidden width. const DEFAULT_WIDTH: NonZero = const { NonZero::new(512).unwrap() }; +/// The default residual block count. const DEFAULT_RESIDUAL_BLOCKS: NonZero = const { NonZero::new(4).unwrap() }; +/// The projector prefix width used by default for the stored representation. const DEFAULT_REPRESENTATION_DIMENSIONS: NonZero = const { NonZero::new(PROJECTOR_DIMENSIONS).unwrap() }; +/// The default role embedding width. const DEFAULT_ROLE_DIMENSIONS: NonZero = const { NonZero::new(16).unwrap() }; +/// The default condition width of one relation-lens column. const DEFAULT_CONDITION_DIMENSIONS: NonZero = const { NonZero::new(1).unwrap() }; /// Output coordinates per row. @@ -286,7 +294,7 @@ pub(crate) struct ProjectorInput { /// Feature-wise linear modulation from a condition vector. /// /// `forward(h, c) = (1 + Δγ(c)) · h + β(c)`, where one linear map produces `[dgamma; beta]`, the -/// delta scale stacked over the shift. The map and its bias initialize to zero, so modulation +/// delta scale stacked over the shift. The map and its bias initialize to zero, and modulation /// starts as the identity for every condition. #[derive(Module, Debug)] struct Film { @@ -294,6 +302,7 @@ struct Film { } impl Film { + /// Builds the zero-initialized modulation map from `condition_dimensions` to `2 · width`. fn new( width: usize, condition_dimensions: usize, @@ -310,6 +319,10 @@ impl Film { } } + /// Modulates `features` of shape `[rows, width]` by `condition` of shape `[rows, condition]`. + /// + /// The map's first `width` outputs are the delta scale and the rest the shift. The result is + /// `(1 + Δγ(c)) · h + β(c)` row by row. fn forward(&self, features: Tensor, condition: Tensor) -> Tensor { let width = features.dims()[1]; let modulation = self.linear.forward(condition); @@ -323,7 +336,7 @@ impl Film { /// One residual block, condition-modulated between LN and activation. /// /// `forward(h, c) = h + W2 SiLU(FiLM(LN(W1 h + b1), c)) + b2`. The second linear and its bias -/// initialize to zero, so the block is the identity before training. +/// initialize to zero, and the block is the identity before training. #[derive(Module, Debug)] struct ResidualBlock { film: Film, @@ -333,6 +346,9 @@ struct ResidualBlock { } impl ResidualBlock { + /// Builds one block of hidden `width`. + /// + /// Scaled input map, layer norm, zero-initialized modulation and output map. fn new( width: usize, condition_dimensions: usize, @@ -347,6 +363,7 @@ impl ResidualBlock { } } + /// Applies the block: `hidden + W2 SiLU(FiLM(LN(W1 hidden + b1), condition)) + b2`. fn forward(&self, hidden: Tensor, condition: Tensor) -> Tensor { let normalized = self .normalization @@ -376,8 +393,8 @@ pub(crate) struct Projector { impl Projector { /// Builds a freshly initialized model. /// - /// This draws every parameter from `rng` in construction order, so equal architectures, stream - /// types, and seeds produce identical models on every backend. + /// This draws every parameter from `rng` in construction order. Equal architectures, stream + /// types, and seeds therefore produce identical models on every backend. #[must_use] pub(crate) fn new(architecture: Architecture, device: &B::Device, mut rng: R) -> Self { let width = architecture.width.get(); @@ -422,7 +439,7 @@ impl Projector { #[must_use] pub(crate) fn forward(&self, input: ProjectorInput) -> Tensor { // The shape checks cost integer compares against host-side dim metadata once per forward - // call, with no device sync, so they are free beside the matmuls they guard. + // call, with no device sync. They are free beside the matmuls they guard. let [rows, representation_dimensions] = input.representation.dims(); assert_eq!( representation_dimensions, self.representation_dimensions, @@ -460,11 +477,11 @@ impl Projector { /// Builds the model a record describes, verified against the architecture. /// - /// A record loaded into a model adopts the record's tensor shapes, so a record decoded against - /// the wrong architecture would produce a structurally wrong model without an error of its own. - /// This constructor reports that as an error: it verifies the block-stack depth before the - /// record loads (a depth mismatch panics inside the module zip) and every parameter shape - /// after. + /// A record loaded into a model adopts the record's tensor shapes. A record decoded against the + /// wrong architecture would therefore produce a structurally wrong model without an error of + /// its own. This constructor reports that as an error: it verifies the block-stack depth before + /// the record loads (a depth mismatch panics inside the module zip) and every parameter + /// shape after. /// /// # Errors /// @@ -480,10 +497,11 @@ impl Projector { record.blocks.len(), )?; - // `load_record` on the next line replaces every parameter this construction draws, so the - // stream's seed is meaningless by design, and a throwaway is the price of reusing the one - // construction path. The checkpoint's rng state is a different object: it resumes the - // *training* draw sequence, and threading it here would launder meaning into dead draws. + // `load_record` on the next line replaces every parameter this construction draws. The + // stream's seed is therefore meaningless by design, and a throwaway is the price of reusing + // the one construction path. The checkpoint's rng state is a different object: it resumes + // the *training* draw sequence, and threading it here would launder meaning into + // dead draws. let throwaway = Xoshiro256PlusPlus::seed_from_u64(0); let model = Self::new(architecture, device, throwaway).load_record(record); model.check_architecture(architecture)?; @@ -574,12 +592,12 @@ struct Site { } impl Site { - /// A layer outside the block stack. + /// Names a layer outside the block stack. const fn model(layer: Layer) -> Self { Self { layer, block: None } } - /// A layer inside residual block `block`. + /// Names a layer inside residual block `block`. const fn block(layer: Layer, block: usize) -> Self { Self { layer, @@ -662,7 +680,7 @@ enum LinearInit { /// Deterministic parameter materialization from one random stream. /// /// Parameters receive sequential identifiers and values drawn from the stream in construction -/// order, replacing whatever the layer configs would have initialized; the backend's global random +/// order, replacing whatever the layer configs would have initialized. The backend's global random /// state is never touched. struct Initialization { rng: R, @@ -718,8 +736,8 @@ impl Initialization { /// Builds an embedding scaled to sit beside a unit-norm vector. /// /// Rows are uniform with per-component variance `1/reference_dimensions`, the per-component - /// variance of a unit-norm `reference_dimensions`-vector, so concatenating a row to such a - /// vector lets neither block dominate a downstream linear by scale alone. + /// variance of a unit-norm `reference_dimensions`-vector. Concatenating a row to such a vector + /// therefore lets neither block dominate a downstream linear by scale alone. fn embedding( &mut self, count: usize, @@ -743,6 +761,11 @@ impl Initialization { embedding } + /// Replaces a parameter's tensor with `values` in `shape`, under the next sequential id. + /// + /// # Panics + /// + /// Panics when the parameter identifier counter overflows `u64`. fn parameter( &mut self, parameter: Param>, @@ -762,6 +785,9 @@ impl Initialization { ) } + /// Draws `count` values uniformly from `[-bound, bound]`, or all zeros for a zero bound. + /// + /// A zero bound does not advance the random stream. fn values(&mut self, count: usize, bound: f32) -> Vec { if bound == 0.0 { return vec![0.0; count]; diff --git a/libs/@local/graph/atlas/src/salt/projector/model/tests.rs b/libs/@local/graph/atlas/src/salt/projector/model/tests.rs index fc9e2f38b01..7e91ad904da 100644 --- a/libs/@local/graph/atlas/src/salt/projector/model/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/model/tests.rs @@ -1,7 +1,7 @@ //! Certificates for the projector model's initialization contracts. //! //! Identity assertions are bit-exact by design: the zero-initialized layers contribute exactly -//! zero, `1 + 0 = 1` exactly, and `h · 1 + 0` reproduces `h` bit for bit, so any drift is a broken +//! zero, `1 + 0 = 1` exactly, and `h · 1 + 0` reproduces `h` bit for bit: any drift is a broken //! contract, not rounding. use std::sync::LazyLock; @@ -19,8 +19,12 @@ use crate::{ math::nz, }; +/// The CPU device every model fixture runs on, resolved once. static DEVICE: LazyLock = LazyLock::new(|| Device::Cpu.pin(0).resolve()); +/// A small architecture with `CONDITION_DIMENSIONS` condition inputs. +/// +/// Width 8, two residual blocks, six representation and four role dimensions. fn tiny() -> Architecture { Architecture { width: nz!(8), @@ -31,15 +35,18 @@ fn tiny() -> Architecture { } } +/// A `rows × columns` inference tensor from row-major `values`. fn matrix(rows: usize, columns: usize, values: Vec) -> Tensor { Tensor::from_data(TensorData::new(values, [rows, columns]), &*DEVICE) } +/// A one-dimensional integer role tensor from `values`. fn roles(values: Vec) -> Tensor { let rows = values.len(); Tensor::from_data(TensorData::new(values, [rows]), &*DEVICE) } +/// A 2-D tensor's contents as row-major `f32` values. fn to_values(tensor: Tensor) -> Vec { tensor .into_data() @@ -62,6 +69,10 @@ fn representation(rows: usize, columns: usize) -> Tensor { matrix(rows, columns, values) } +/// Returns the hidden input unchanged from a fresh `FiLM` layer at every condition. +/// +/// A freshly initialized `FiLM` layer returns its hidden input unchanged for every condition value +/// and width. #[test] fn film_is_identity_at_initialization_for_every_condition() { for condition_dimensions in [1, 3] { @@ -93,7 +104,7 @@ fn film_is_identity_at_initialization_for_every_condition() { /// Pins the modulation arithmetic the identity certificates cannot see. /// /// The `[dgamma; beta]` column order, the `+ 1` on gamma, and per-row conditions. Every value is -/// dyadic, so equality is exact. +/// dyadic: equality is exact. #[test] fn film_modulates_by_hand_computed_values() { let mut rng = Xoshiro256PlusPlus::seed_from_u64(31); @@ -123,6 +134,7 @@ fn film_modulates_by_hand_computed_values() { ); } +/// A freshly initialized residual block returns its input unchanged for every condition width. #[test] fn residual_block_is_identity_at_initialization() { for condition_dimensions in [1, 3] { @@ -154,6 +166,10 @@ fn residual_block_is_identity_at_initialization() { } } +/// Projects the same finite coordinates from a fresh projector whatever the condition. +/// +/// A fresh projector projects the same finite 2-D coordinates whatever the condition value, for +/// one- and three-dimensional conditions. #[test] fn forward_is_condition_invariant_at_initialization() { for (condition_dimensions, architecture) in [(1, tiny::<1>()), (3, tiny::<3>())] { @@ -241,6 +257,7 @@ fn roles_reach_the_output() { ); } +/// Projecting two rows as a batch equals projecting each alone and concatenating. #[test] fn rows_project_independently() { let projector = @@ -271,6 +288,7 @@ fn rows_project_independently() { ); } +/// A condition tensor wider than the architecture panics with the documented message. #[test] #[should_panic(expected = "condition width should match the architecture")] fn forward_rejects_a_mismatched_condition_width() { diff --git a/libs/@local/graph/atlas/src/salt/projector/report/mod.rs b/libs/@local/graph/atlas/src/salt/projector/report/mod.rs index 4a43080cdb3..68af60c5e69 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/mod.rs @@ -1,7 +1,7 @@ //! Reports over the published projector, observers of the placement rather than participants. //! //! Everything under this module reads published generations and measures the projector's -//! behaviour after the fact. Nothing here trains the projector or stages an artifact. +//! behaviour after the fact, as a read-only observer of the placement. //! //! [`replay`] measures how the deployed publish path serves arrivals: nodes a later generation //! fitted that an earlier generation never saw. diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/assemble.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/assemble.rs index d44a76e9fe8..bc943f90c4c 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/assemble.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/assemble.rs @@ -18,11 +18,13 @@ use crate::progress::Progress; /// One estimand's metric pass with its population cells. /// -/// `I` is the estimand universe's position domain, so the entity and class runs cannot consume -/// each other's representatives, and each estimand's row builder below is bound to its own -/// domain. +/// `I` is the estimand universe's position domain. The entity and class runs therefore cannot +/// consume each other's representatives, and each estimand's row builder below is bound to its +/// own domain. struct EstimandRun<'run, I> { + /// The population cells, one per design. cells: Vec, + /// The rank readings over the estimand's universe. pass: MetricPass<'run, I>, } @@ -33,8 +35,8 @@ impl<'run, I: Id> EstimandRun<'run, I> { data: &'run EstimandData, dedup: Option<&'run IdSlice>, ) -> Self { - // The designs were validated against the drawn universe cardinalities, so the data - // arriving here must carry exactly those cardinalities. + // The designs were validated against the drawn universe cardinalities, and the data + // handed over here must therefore carry exactly those cardinalities. debug_assert!( designs .iter() @@ -192,8 +194,8 @@ impl ArrivalReplay { /// /// Projects each distinct sampled row once in bounded batches, recording placed, /// out-of-frame, and non-finite outcomes before computing any conditional metric. A - /// non-finite row is recorded and the surrounding rows retried, so one bad row costs one - /// reading, not the batch. Every rank reading then comes from its estimand's fixed + /// non-finite row is recorded and the surrounding rows retried, and one bad row therefore + /// costs one reading, not the batch. Every rank reading then comes from its estimand's fixed /// comparison universe. Projection failures are outcomes the report records, not errors. /// /// # Panics diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/design.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/design.rs index 2ec49b790a7..77cf49b49b9 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/design.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/design.rs @@ -139,7 +139,7 @@ impl NeighbourhoodDesign { /// /// Each size is checked against the estimand and diagnostic universes. Every universe is /// then proved, at its estimand's actual query and control counts, to fit the rank kernel's - /// integer carriers, so no aggregate the run observes can wrap. + /// integer carriers, and no aggregate the run observes can therefore wrap. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/draw.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/draw.rs index de0198348fe..6e8bac2a583 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/draw.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/draw.rs @@ -69,8 +69,8 @@ pub(super) struct DrawnSamples { impl DrawnSamples { /// Derives the replay's generator from the sampling seed. /// - /// The pinned name keeps the derivation disjoint from every other seeded consumer's, so a - /// replay and a fit sharing a seed value still draw independently. + /// The pinned name keeps the derivation disjoint from every other seeded consumer's, and a + /// replay and a fit sharing a seed value therefore still draw independently. #[expect( clippy::little_endian_bytes, reason = "the derivation preimage pins the canonical little-endian bytes" @@ -85,12 +85,12 @@ impl DrawnSamples { /// Draws every sample of both estimands under the seed. /// - /// The draw order is pinned, so equal seeds replay the whole design. Entity queries draw - /// first and the joint entity universe-and-control draw follows. The class draws repeat that - /// order. Each estimand's universe and controls come from one joint draw whose first + /// The draw order is pinned, and equal seeds therefore replay the whole design. Entity queries + /// draw first and the joint entity universe-and-control draw follows. The class draws repeat + /// that order. Each estimand's universe and controls come from one joint draw whose first /// `comparisons` indices are the universe and whose rest are the controls, which is what /// makes the two disjoint. Classes enter their draw with equal weight, one index per class, - /// so member multiplicity buys a class no extra chance. + /// and member multiplicity therefore buys a class no extra chance. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/error.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/error.rs index f39198b2a3b..80e414c14ab 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/error.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/error.rs @@ -17,8 +17,8 @@ use crate::{ pub(crate) enum ReplayError { /// A generation's coordinates were not placed by the trained projector. /// - /// A landmark-baseline generation publishes coordinates the projector never produced, so a - /// replay over it would attribute the baseline's behaviour to the projector. + /// A landmark-baseline generation publishes coordinates the projector never produced, and a + /// replay over it would therefore attribute the baseline's behaviour to the projector. NotProjectorPlaced { /// The generation whose placement disqualifies it. generation: GenerationId, @@ -27,8 +27,8 @@ pub(crate) enum ReplayError { }, /// The generations record different embedding contracts. /// - /// Representations produced under different contracts are not one input space, so a - /// cross-generation distance would compare incommensurable coordinates. + /// Representations produced under different contracts are not one input space, and a + /// cross-generation distance would therefore compare incommensurable coordinates. EmbedderMismatch { /// The generation named as earlier. earlier: GenerationId, @@ -65,7 +65,7 @@ pub(crate) enum ReplayError { /// The read failure. source: io::Error, }, - /// A generation records no temporal axes, so no transaction-time order can hold. + /// A generation records no temporal axes, and no transaction-time order can hold over it. UnrecordedTemporalAxes { /// The generation without recorded axes. generation: GenerationId, @@ -193,7 +193,7 @@ pub(crate) enum ReplayError { /// A design's observation load exceeds the rank kernels' integer carriers. /// /// The worst-case rank penalties over this many observations would overflow the metric - /// aggregate's accumulation or readback arithmetic, so construction refuses the design + /// aggregate's accumulation or readback arithmetic. Construction therefore refuses the design /// before anything accumulates. AggregateCapacityExceeded { /// The comparison universe whose worst-case penalties overflow. diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/extract.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/extract.rs index 60fffe9edf8..d0195fb441a 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/extract.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/extract.rs @@ -34,7 +34,7 @@ pub(super) struct GenerationColumns<'run> { } impl<'run> GenerationColumns<'run> { - /// Admits the columns of one corpus, so a value cannot hold mismatched columns. + /// Admits the columns of one corpus. A value therefore cannot hold mismatched columns. /// /// # Errors /// @@ -94,14 +94,23 @@ impl<'run> GenerationColumns<'run> { /// One generation's opened artifact files, alive while the columns borrow from them. pub(super) struct GenerationArtifacts { + /// The validated node identity table. identities: IdentityTableArchive, + /// The mapped representation matrix. representations: ArrayFile, + /// The mapped row-position column. positions: ArrayFile, + /// The mapped wire-coordinate column. wire: ArrayFile, } impl GenerationArtifacts { /// Opens one generation's identity, representation, row-position, and wire artifacts. + /// + /// # Errors + /// + /// Returns the [`ReplayError`] naming the first artifact that fails to open, or the identity + /// table that opens but does not validate. pub(super) fn open(generation: &Generation) -> Result { let id = generation.id(); let files = &generation.repository().files; @@ -147,7 +156,11 @@ impl GenerationArtifacts { /// Gathers the published wire coordinate of every node row. /// /// The wire column lives in base delivery order. The row-position column maps each node row - /// to its base position, so the gather leaves one wire point per node row. + /// to its base position, and the gather therefore leaves one wire point per node row. + /// + /// # Errors + /// + /// Returns the errors [`WireArtifacts::gathered`] documents. pub(super) fn wire_of_row( &self, generation: &Generation, @@ -160,6 +173,12 @@ impl GenerationArtifacts { } /// Borrows the columns the partition consumes. + /// + /// # Errors + /// + /// Returns [`ReplayError::InvalidRepresentations`] when the representation matrix does not + /// read as rows of the projector width, and [`ReplayError::Rows`] when the columns disagree + /// on their row count. pub(super) fn columns<'files>( &'files self, generation: &Generation, @@ -206,8 +225,8 @@ impl WireArtifacts<'_> { /// # Panics /// /// This panics when the position column names a slot beyond the wire column. Both columns - /// belong to one rehashed generation, so such a slot is a publisher defect, never a lawful - /// input. + /// belong to one rehashed generation, and such a slot is therefore a publisher defect, never + /// a lawful input. pub(super) fn gathered( self, generation: GenerationId, @@ -227,8 +246,8 @@ impl WireArtifacts<'_> { /// The later generation's opened edge-endpoint artifact. /// -/// The open and the decode are two steps because the decoded pairs borrow the mapped file, so -/// the type owns the file and the borrow happens through [`pairs`](Self::pairs). +/// The open and the decode are two steps because the decoded pairs borrow the mapped file. The +/// type therefore owns the file, and the borrow happens through [`pairs`](Self::pairs). pub(super) struct EndpointArtifact { /// The mapped edge-endpoint column. file: ArrayFile, @@ -251,8 +270,8 @@ impl EndpointArtifact { /// Views the staged endpoint column as the artifact stores it. /// - /// The node row id is little-endian by construction, so the typed column is the stored form - /// and the view is exact on every architecture. + /// The node row id is little-endian by construction. The typed column is therefore the stored + /// form, and the view is exact on every architecture. /// /// # Errors /// @@ -287,6 +306,7 @@ mod tests { identity::{BasePosition, NodeRowId}, }; + /// A generation id whose 64 hex digits spell `ordinal`. fn generation(ordinal: u8) -> GenerationId { format!("{ordinal:064x}") .parse() @@ -308,8 +328,8 @@ mod tests { /// Stages a row-position column through the pipeline's own column writer. /// - /// [`SizedColumn`] stamps the variant from the element type, so the staged file carries - /// the exact tag every published `position-of-row.arr` carries. + /// [`SizedColumn`] stamps the variant from the element type, and the staged file therefore + /// carries the exact tag every published `position-of-row.arr` carries. fn staged_positions(directory: &Utf8PathBuf, positions: &[u32]) -> ArrayFile { let column: Vec = positions .iter() @@ -333,8 +353,8 @@ mod tests { /// Stages an endpoint column with the given variant tag. /// - /// [`ArrayVariant::U64Le`] mirrors the ingest's own writer call; the native variant stages - /// the refusal fixture. + /// [`ArrayVariant::U64Le`] mirrors the ingest's own writer call, and the native variant + /// stages the refusal fixture. fn staged_endpoints( directory: &Utf8PathBuf, pairs: &[[u64; 2]], @@ -351,6 +371,10 @@ mod tests { ArrayFile::open(&path).expect("the staged column opens") } + /// `GenerationColumns::new` fails with `Rows` on columns of unequal length. + /// + /// `GenerationColumns::new` fails with `Rows` when the id, representation and wire columns + /// disagree in length. #[test] fn columns_refuse_mismatched_rows() { let wire = [Vec2::new(0.0, 0.5)]; @@ -373,6 +397,10 @@ mod tests { )); } + /// Gathers each row's wire coordinate through a staged position permutation. + /// + /// Gathering the wire column through a staged position permutation returns each row's wire + /// coordinate. #[test] fn wire_gathers_the_staged_artifacts() { let directory = scratch("wire-gathers"); @@ -403,9 +431,13 @@ mod tests { ); } + /// A native-endian position column fails with `InvalidPositions` naming the generation. + /// + /// A position column tagged native-endian rather than little-endian fails with + /// `InvalidPositions` naming the generation. #[test] fn wire_refuses_a_native_position_column() { - // The pipeline persists base positions little-endian; a native-tagged column is not + // The pipeline persists base positions little-endian. A native-tagged column is not // the published form and refuses rather than reads. let directory = scratch("wire-refuses-native"); let path = directory.join("position-of-row.arr"); @@ -433,11 +465,15 @@ mod tests { )); } + /// Panics with the standard out-of-bounds message on a position beyond the wire column. + /// + /// A position naming a slot beyond the wire column panics with the standard out-of-bounds + /// message. #[test] #[should_panic(expected = "index out of bounds")] fn wire_position_beyond_column_panics() { - // Both columns belong to one rehashed generation, so a position naming a slot beyond - // the wire column is a publisher defect and dies loudly instead of misreading. + // Both columns belong to one rehashed generation. A position naming a slot beyond + // the wire column is therefore a publisher defect and dies loudly instead of misreading. let directory = scratch("wire-beyond-panics"); let positions = staged_positions(&directory, &[0, 5]); let wire = staged_wire(&directory, &[Vec2::new(0.0, 0.5), Vec2::new(1.0, 1.5)]); @@ -451,6 +487,7 @@ mod tests { ); } + /// A little-endian staged endpoint column reads back as its source-target pairs. #[test] fn endpoints_read_the_staged_column() { let directory = scratch("endpoints-read"); @@ -468,6 +505,7 @@ mod tests { assert_eq!(read, [[0, 1], [7, 7]]); } + /// A native-endian endpoint column fails with `InvalidEndpoints` naming the generation. #[test] fn endpoints_refuse_the_native_variant() { let directory = scratch("endpoints-refuse-native"); diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/metric.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/metric.rs index 2af5cf51f2f..b3479fca4e6 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/metric.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/metric.rs @@ -28,8 +28,8 @@ use crate::{ /// /// The returned row is raw `u32` because it feeds [`RankScratch`] and /// [`NeighbourhoodAggregate::observe`], whose rank vocabulary is raw, and it never leaves this -/// module. Construction refuses a comparison universe beyond the `u32` rank domain, so the cast -/// is total over admitted universes. +/// module. Construction refuses a comparison universe beyond the `u32` rank domain, and the cast +/// is therefore total over admitted universes. #[expect( clippy::cast_possible_truncation, reason = "construction refuses a comparison universe beyond the u32 rank domain" @@ -59,15 +59,22 @@ fn one_query_reading( /// The running deployed-minus-refit sums over placed queries. pub(super) struct PairedAccumulator { + /// Placed queries accumulated. queries: usize, + /// The summed recall differences. recall: DFinite, + /// The summed trustworthiness differences. trustworthiness: DFinite, + /// The summed continuity differences. continuity: DFinite, + /// The summed intrusion-rate differences. intrusion_rate: DFinite, + /// The summed extrusion-rate differences. extrusion_rate: DFinite, } impl PairedAccumulator { + /// Starts every sum at zero with no placed query. const fn new() -> Self { Self { queries: 0, @@ -79,6 +86,7 @@ impl PairedAccumulator { } } + /// Adds one placed query's paired differences to the running sums. const fn accumulate(&mut self, difference: &DifferenceRow) { self.queries += 1; self.recall += difference.recall; @@ -93,8 +101,8 @@ impl PairedAccumulator { let queries = NonZero::new(self.queries)?; let count = DPositive::from_usize(queries); - // The divisor is a query count, so count ≥ 1 never magnifies: every mean stays within - // its finite numerator's bound. + // The divisor is a query count of at least one and never magnifies: every mean stays + // within its finite numerator's bound. let mean = |sum: DFinite| (sum / count).finish_unchecked(); Some(PairedSummary { @@ -112,25 +120,37 @@ impl PairedAccumulator { /// One novelty split's population aggregates at one neighbourhood size. struct NoveltyCells { + /// The deployed readings of the split's placed queries. deployed: NeighbourhoodAggregate, + /// The refit readings of the split's queries. refit: NeighbourhoodAggregate, } /// The deduplication diagnostic's aggregates at one neighbourhood size. struct DedupCells { + /// The deployed readings over the restricted universe. deployed: NeighbourhoodAggregate, + /// The refit readings over the restricted universe. refit: NeighbourhoodAggregate, + /// The control readings over the restricted universe. controls: NeighbourhoodAggregate, } /// One estimand's population aggregates at one neighbourhood size. pub(super) struct PopulationCells { + /// The deployed readings of every placed query. deployed: NeighbourhoodAggregate, + /// The refit readings of every query. refit: NeighbourhoodAggregate, + /// The fitted controls' readings. controls: NeighbourhoodAggregate, + /// The readings of queries whose bytes occur in `G0`. seen: NoveltyCells, + /// The readings of queries whose bytes occur nowhere in `G0`. novel: NoveltyCells, + /// The running deployed-minus-refit sums. paired: PairedAccumulator, + /// The deduplication diagnostic's cells, where the pass carries the lens. dedup: Option, } @@ -200,9 +220,12 @@ impl PopulationCells { /// The deduplication lens over one universe: representatives and their own scratch. /// /// The representative list maps each deduplication position to the universe position it -/// restricts to, so the lens can only consume distances keyed by its own universe's domain. +/// restricts to, and the lens can therefore only consume distances keyed by its own universe's +/// domain. struct DedupLens<'run, I> { + /// The universe position each deduplication position restricts to, ascending. representatives: &'run IdSlice, + /// The rank scratch sized to the restricted universe. scratch: RankScratch, } @@ -222,22 +245,27 @@ impl DedupLens<'_, I> { /// The rank-metric pass over one fixed comparison universe. /// /// Borrows the universe's embeddings, both wire framings, and the validated designs. The pass -/// owns the reusable rank scratch, so it allocates two `u32` rows per universe regardless of -/// query count. `I` is the universe's position domain, so one estimand's pass cannot consume -/// another estimand's positions or representatives. +/// owns the reusable rank scratch and therefore allocates two `u32` rows per universe regardless +/// of query count. `I` is the universe's position domain, and one estimand's pass therefore +/// cannot consume another estimand's positions or representatives. pub(super) struct MetricPass<'run, I> { + /// The validated designs, one per neighbourhood size. designs: &'run [NeighbourhoodDesign], + /// The universe members' embeddings, in position order. universe: &'run IdSlice>, + /// The universe members' wire coordinates, one column per generation. wire: &'run Pair>, + /// The rank scratch sized to the universe. scratch: RankScratch, + /// The deduplication lens, where the pass carries one. dedup: Option>, } impl<'run, I: Id> MetricPass<'run, I> { - /// A pass over one universe, with the deduplication lens where one is handed over. + /// Opens a pass over one universe, with the deduplication lens where one is handed over. /// - /// The embedding rows arrive in universe draw order, which is what binds them to the - /// position domain here. + /// The embedding rows are handed over in universe draw order, which is what binds them to + /// the position domain here. pub(super) fn new( designs: &'run [NeighbourhoodDesign], universe: &'run [AlignedVecN], @@ -256,7 +284,7 @@ impl<'run, I: Id> MetricPass<'run, I> { } } - /// One sampled query's readings, merged into the population cells. + /// Reads one sampled query and merges its readings into the population cells. /// /// The refit reading exists for every query. The deployed reading and the paired difference /// exist exactly when the outcome placed the query. @@ -371,8 +399,8 @@ impl<'run, I: Id> MetricPass<'run, I> { /// One fitted control's readings, merged into the population cells. /// - /// The control reads under the earlier generation's own wire frame, so its readings share the - /// deployed readings' normalizer. + /// The control reads under the earlier generation's own wire frame, and its readings therefore + /// share the deployed readings' normalizer. pub(super) fn control( &mut self, embedding: &AlignedVecN, diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/mod.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/mod.rs index 4820037f78b..11e85500866 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/mod.rs @@ -6,10 +6,9 @@ //! generation never saw, projected online through the fitted generation's own published //! projector. This report replays that deployment retrospectively over two published generations. //! The arrivals of the later generation `G1` project through `G0`'s certified projector, and each -//! projected -//! neighbourhood is read against the representation-space truth, beside the counterfactual -//! reading the arrival's own `G1` fit produced and beside fitted controls that share every -//! normalizer. +//! projected neighbourhood is read against the representation-space truth, beside the +//! counterfactual reading the arrival's own `G1` fit produced and beside fitted controls that +//! share every normalizer. //! //! # Populations and estimands //! @@ -17,19 +16,19 @@ //! (present in `G1`, absent from `G0`), stable comparison rows (present in both with byte-equal //! projector representations), and revised fitted rows (present in both with differing bytes). //! Revised rows are counted and excluded, because serving never projects revised bytes. A -//! post-fit edition keeps its fitted coordinate until a refit, so an arrival label on such a row -//! would read the wrong mechanism. Arrivals split further by whether their representation bytes -//! already occur anywhere in `G0` - only a novel representation tests generalization beyond an -//! input the model has already seen. An empty arrival population is a refusal, never a perfect -//! result. +//! post-fit edition keeps its fitted coordinate until a refit, and an arrival label on such a row +//! would therefore read the wrong mechanism. Arrivals split further by whether their +//! representation bytes already occur anywhere in `G0` - only a novel representation tests +//! generalization beyond an input the model has already seen. An empty arrival population is a +//! refusal, never a perfect result. //! -//! An entity estimand and a class estimand ride every reading. The entity estimand samples rows, so -//! a duplicated representation weighs by its multiplicity: the consumer-facing view, where a -//! heavily duplicated entity really does dominate what serving shows. The class estimand forms -//! byte-exact representation classes over the full eligible populations first and samples -//! classes with equal weight, each read at its deterministic representative (the class's lowest -//! later row), so multiplicity buys a class neither inclusion odds nor query weight. A -//! deduplication diagnostic beside them restricts the entity draw's own universe to one member +//! Every reading carries an entity estimand and a class estimand. The entity estimand samples +//! rows, and a duplicated representation therefore weighs by its multiplicity: the consumer-facing +//! view, where a heavily duplicated entity really does dominate what serving shows. The class +//! estimand forms byte-exact representation classes over the full eligible populations first and +//! samples classes with equal weight, each read at its deterministic representative (the class's +//! lowest later row). Multiplicity therefore buys a class neither inclusion odds nor query weight. +//! A deduplication diagnostic beside them restricts the entity draw's own universe to one member //! per class, isolating how much duplication inside that draw moved the entity readings. //! //! # Orderings and readings @@ -54,9 +53,9 @@ //! able to carry the experiment. Everything below the artifact extraction - every data refusal, //! the partition, the class formation, the sampling, the incident scan - runs identically on //! extracted and on fabricated columns. [`ArrivalReplay::report`] then drives one -//! [`PublishedProjector`](projection::PublishedProjector), -//! the trait standing where `G0`'s reopened projector stands in production. The production -//! adapter binds the real projector and its construction-time certificate. +//! [`PublishedProjector`](projection::PublishedProjector), the trait standing where `G0`'s +//! reopened projector stands in production. The production adapter binds the real projector and +//! its construction-time certificate. //! //! [`NeighbourhoodAggregate`]: crate::salt::quality::metric::NeighbourhoodAggregate @@ -98,8 +97,9 @@ mod tests; /// One value per generation of the replayed pair, named by its temporal side. /// -/// Every earlier/later duo travels through this carrier, so two same-typed values cannot swap -/// silently at a call boundary. +/// Every earlier/later duo passes through this carrier, and each side is assigned and read by +/// name rather than by position. The names make a swap of two same-typed values visible at the +/// construction site, though not a type error. pub(super) struct Pair { /// The value on the earlier, deployed side `G0`. pub earlier: T, @@ -239,16 +239,20 @@ impl PopulationCounts { /// One replay, extracted and validated, ready to drive a published projector. /// /// Construction copies everything the run reads, from sampled embeddings and wire coordinates to -/// populations and designs, so the source generations' mappings are released before the run -/// starts and the run itself touches no artifact. +/// populations and designs. The source generations' mappings are therefore released before the +/// run starts, and the run itself touches no artifact. pub(crate) struct ArrivalReplay { /// The pair's identities. generation: Pair, /// The pair's recorded snapshot axes. axes: Pair, + /// The seed every sample drew under. seed: u64, + /// The requested sizes, echoed into the report. requested: RequestedDesign, + /// The population, class, and sample counts. populations: PopulationCounts, + /// The validated neighbourhood designs, one per requested size. designs: Vec, /// The distinct rows both estimands project, each once. plan: ProjectionPlan, @@ -275,7 +279,7 @@ impl ArrivalReplay { /// generation's identity, representation, row-position, and wire-coordinate artifacts plus /// the later generation's edge endpoints. The joined rows are partitioned and the /// byte-exact classes formed over the full populations. Every sample draws under the seed, - /// and construction copies what the run reads, so the returned value holds no artifact + /// and construction copies what the run reads. The returned value therefore holds no artifact /// mapping. /// /// # Errors @@ -319,6 +323,12 @@ impl ArrivalReplay { /// /// Validation, partition, class formation, sampling, and copying all run here, identically /// on extracted and on fabricated columns. + /// + /// # Errors + /// + /// Returns the [`ReplayError`] naming the first refusal: the pair's snapshot order, no + /// requested neighbourhood, a universe beyond the rank domain, an empty arrival population, a + /// draw the populations cannot carry, or a design the samples cannot carry. fn from_columns( columns: &Pair>, edges: &IdSlice, diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/plan.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/plan.rs index bf497a3a8be..1b27685b079 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/plan.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/plan.rs @@ -79,8 +79,8 @@ impl EstimandData { /// This fires when a sequence declares an exact length whose embedding matrix layout cannot /// fit `isize`, since the matrix is allocated from that declared length before any of its /// pairs is consumed. It also fires when a pair sequence yields fewer items than its - /// declared exact length, so a short sequence cannot leave silently zeroed embedding rows - /// behind. The pair indexes must stay inside their columns as well: every pair's + /// declared exact length, and a short sequence therefore cannot leave silently zeroed + /// embedding rows behind. The pair indexes must stay inside their columns as well: every pair's /// `earlier_row` lies inside the earlier wire column while its `later_row` lies inside the /// representation column, and only a universe pair's `later_row` reaches the later wire /// column. diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/population.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/population.rs index 18cb0ee347c..65e0e255dc9 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/population.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/population.rs @@ -16,7 +16,7 @@ use crate::{ #[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] #[serde(rename_all = "kebab-case")] pub(crate) enum Novelty { - /// The exact bytes occur in `G0`, so the model has seen this input. + /// The exact bytes occur in `G0`: the model has seen this input. Seen, /// The bytes occur nowhere in `G0`: the reading that tests generalization. Novel, @@ -90,8 +90,8 @@ pub(super) struct StableClass { /// One byte-exact representation class of the arrival population. /// /// The representative is the class's lowest-later-row member: a deterministic rule, independent -/// of any draw. Byte-equal arrivals share their novelty by construction, so the class carries it -/// whole. +/// of any draw. Byte-equal arrivals share their novelty by construction, and the class therefore +/// carries it whole. #[derive(Copy, Clone)] pub(super) struct ArrivalClass { /// The representative member's later-generation row. @@ -268,7 +268,7 @@ impl IncidentStats { .collect(); for &[source, target] in edges { - // A self-referential edge is incident once, so it enters once. + // A self-referential edge is incident once and therefore enters once. let both = [(source, target), (target, source)]; let ends = if source == target { &both[..1] diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/preflight.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/preflight.rs index d1b4b77b9a3..f91498e6ea3 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/preflight.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/preflight.rs @@ -1,8 +1,8 @@ //! Generation compatibility and artifact integrity, checked before any extraction. //! -//! A replay attributes the whole gap between two generations' readings to later arrivals, so -//! the pair must differ in nothing else it can refuse on. Both placements come from the trained -//! projector. Both fits ran under one embedding contract and one configuration, the seed +//! A replay attributes the whole gap between two generations' readings to later arrivals. The +//! pair must therefore differ in nothing else it can refuse on. Both placements come from the +//! trained projector. Both fits ran under one embedding contract and one configuration, the seed //! included. The artifact bytes must also be the bytes the metadata documents bound, because //! the report's evidence identity is the generation pair's identity. @@ -41,10 +41,12 @@ impl<'doc> GenerationContract<'doc> { /// A generation pair admitted for replay. /// /// Construction is the admission: the contracts agree and every bound artifact hashes to its -/// metadata record. Everything downstream reaches the generations through this value, so an -/// unadmitted pair cannot be extracted. +/// metadata record. Everything downstream reaches the generations through this value, and an +/// unadmitted pair therefore cannot be extracted. pub(super) struct VerifiedPair<'run> { + /// The earlier generation `G0`. earlier: &'run Generation, + /// The later generation `G1`. later: &'run Generation, } @@ -187,6 +189,7 @@ mod tests { }, }; + /// A generation id whose 64 hex digits spell `ordinal`. fn generation(ordinal: u8) -> GenerationId { format!("{ordinal:064x}") .parse() @@ -207,12 +210,16 @@ mod tests { } } + /// The SHA-256 digest of `seed`, standing in for a recorded file hash or embedder fingerprint. fn digest(seed: &str) -> Sha256Digest { let mut hasher = Sha256::new(); hasher.update(seed.as_bytes()); hasher.finalize() } + /// Builds a reproducibility record with the config at `config_seed`. + /// + /// The embedder fingerprint derives from `embedder`, and the record names no prior. fn reproducibility(config_seed: u64, embedder: &str) -> Reproducibility { Reproducibility { config: config(config_seed), @@ -221,6 +228,10 @@ mod tests { } } + /// A baseline-placed earlier generation fails with `NotProjectorPlaced` naming it. + /// + /// A pair whose earlier generation was baseline-placed fails with `NotProjectorPlaced` naming + /// it. #[test] fn contract_placement() { let shared = reproducibility(7, "embedder"); @@ -247,6 +258,7 @@ mod tests { )); } + /// A pair whose embedder fingerprints differ fails with `EmbedderMismatch`. #[test] fn contract_embedder() { let earlier = reproducibility(7, "embedder"); @@ -268,10 +280,11 @@ mod tests { assert!(matches!(result, Err(ReplayError::EmbedderMismatch { .. }))); } + /// A pair differing only in the config seed fails with `ConfigMismatch`. #[test] fn contract_config() { - // The seed is part of the complete configuration echo, so a pair - // differing only there refuses. + // The seed is part of the complete configuration echo, and a pair + // differing only there therefore refuses. let earlier = reproducibility(7, "embedder"); let later = reproducibility(8, "embedder"); let result = Pair { @@ -291,9 +304,13 @@ mod tests { assert!(matches!(result, Err(ReplayError::ConfigMismatch { .. }))); } + /// Accepts a pair differing only in the prior lineage, which the contract excludes. + /// + /// A pair differing only in the prior lineage agrees, since the prior is outside the compared + /// contract. #[test] fn contract_pass() { - // The prior lineage lawfully differs, so it is deliberately outside + // The prior lineage lawfully differs and is deliberately outside // the compared contract. let earlier = reproducibility(7, "embedder"); let mut later = reproducibility(7, "embedder"); @@ -328,10 +345,14 @@ mod tests { dir } + /// Builds a [`FileName`] from a plain literal. fn file_name(name: &str) -> FileName { FileName::new(name.to_owned()).expect("the fixture name is a plain file name") } + /// Writes `bytes` to `name` under `directory` and records its digest. + /// + /// The return value is the repository entry recording the digest. fn bound_file(directory: &Utf8PathBuf, name: &str, bytes: &[u8]) -> RepositoryFile { fs::write(directory.join(name), bytes).expect("the fixture file is writable"); let mut hasher = Sha256::new(); @@ -342,6 +363,7 @@ mod tests { } } + /// Artifacts whose bytes match their recorded digests verify. #[test] fn integrity_verified() { let directory = scratch("integrity-verified"); @@ -354,6 +376,10 @@ mod tests { .expect("intact bytes match the recorded digests"); } + /// A rewritten artifact fails with `ArtifactIntegrity` naming the role and observed digest. + /// + /// An artifact rewritten after recording fails with `ArtifactIntegrity` naming the role and + /// carrying the observed digest. #[test] fn integrity_tampered() { let directory = scratch("integrity-tampered"); @@ -381,6 +407,7 @@ mod tests { )); } + /// A recorded artifact missing from the directory fails with `ReadArtifact` naming the role. #[test] fn integrity_unreadable() { let directory = scratch("integrity-unreadable"); diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/projection.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/projection.rs index 3b63f9bbd35..54043c84f4d 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/projection.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/projection.rs @@ -9,8 +9,8 @@ use crate::{ /// Rows handed to one [`PublishedProjector::project`] call. /// -/// The bound keeps each call's staging copy small and gives the progress observation its cadence; -/// the projector itself batches however it likes. +/// The bound keeps each call's staging copy small and gives the progress observation its cadence. +/// The projector itself batches however it likes. const PROJECTION_BATCH_ROWS: usize = 256; /// One arrival's projection outcome through the published projector. @@ -41,7 +41,7 @@ pub(crate) struct NonFinitePlacement { /// /// In production the implementor wraps the serving placer bound to `G0`, whose construction /// certifies the reopened checkpoint against the generation's own published coordinates. The -/// trait mirrors that placer's projection contract; `&mut self` additionally admits stateful +/// trait mirrors that placer's projection contract, and `&mut self` additionally admits stateful /// implementations. pub(crate) trait PublishedProjector { /// Projects one batch of arrival representations. @@ -84,8 +84,8 @@ pub(crate) trait PublishedProjector { /// Projects one row range, splitting around each non-finite row. /// /// A failing call reports its first non-finite row and drops the outcomes of the rows before - /// it, so those rows re-project in a narrower call. The projector may be stateful, so no - /// call's outcome is assumed from another's. + /// it, and those rows therefore re-project in a narrower call. The projector may be stateful, + /// and no call's outcome is assumed from another's. /// /// # Panics /// @@ -135,8 +135,11 @@ pub(crate) trait PublishedProjector { /// One row's projection outcome, held with its wire coordinate where one exists. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum ProjectedOutcome { + /// The row projected inside the fitted frame, at this wire coordinate. Placed(Vec2), + /// The row projected outside the fitted frame. OutOfFrame, + /// The row's projection produced a non-finite coordinate. NonFinite, } diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/report.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/report.rs index 9cff292a84a..2b83daca0ab 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/report.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/report.rs @@ -241,10 +241,10 @@ pub(crate) struct NeighbourhoodBlock { /// One neighbourhood size's diagnostic readings over the entity draw, deduplicated in place. /// /// The diagnostic restricts the entity estimand's own sampled universe to one member per -/// byte-exact representation class, so it isolates how much duplication inside that very draw -/// moved the entity readings. It estimates no population. The deduplicated universe is smaller -/// than the entity universe, so readings at one `k` sit on a different normalizer than the -/// entity rows beside them. Its membership follows the entity draw. +/// byte-exact representation class, and it therefore isolates how much duplication inside that +/// very draw moved the entity readings. It estimates no population. The deduplicated universe is +/// smaller than the entity universe, and readings at one `k` therefore use a different normalizer +/// than the entity rows beside them. Its membership follows the entity draw. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct DedupBlock { /// The neighbourhood size the block reads at. @@ -338,11 +338,12 @@ pub(crate) struct IncidentEdgeSummary { /// One replay's complete evidence record. /// /// The report identifies its data and design, from both generations and their temporal axes to -/// the seed and every requested size, so a serialized report re-reads without the configuration -/// that produced it. The entity estimand samples rows, and a duplicated representation weighs -/// by its multiplicity. The class estimand samples byte-exact representation classes with equal -/// weight, read at deterministic representatives. Each estimand answers its own question, and -/// readings at one `k` compare within one estimand before across estimands. +/// the seed and every requested size. A serialized report therefore re-reads without the +/// configuration that produced it. The entity estimand samples rows, and a duplicated +/// representation weighs by its multiplicity. The class estimand samples byte-exact +/// representation classes with equal weight, read at deterministic representatives. Each +/// estimand answers its own question, and readings at one `k` compare within one estimand before +/// across estimands. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct ReplayReport { /// The earlier generation `G0`, whose published projector placed the queries. diff --git a/libs/@local/graph/atlas/src/salt/projector/report/replay/tests.rs b/libs/@local/graph/atlas/src/salt/projector/report/replay/tests.rs index 4c1584ebadd..19cf3abe317 100644 --- a/libs/@local/graph/atlas/src/salt/projector/report/replay/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/report/replay/tests.rs @@ -31,6 +31,7 @@ fn entity(ordinal: u128) -> crate::postgres::id::ArchivedEntityId { } } +/// A generation id whose 64 hex digits spell `ordinal`. fn generation(ordinal: u8) -> GenerationId { format!("{ordinal:064x}") .parse() @@ -42,6 +43,7 @@ const fn row(value: u64) -> NodeRowId { NodeRowId::new(value) } +/// Temporal axes with both transaction and decision time at `seconds` past the Unix epoch. fn axes(seconds: i64) -> TemporalAxes { TemporalAxes { transaction_time: hash_graph_temporal_versioning::Timestamp::from_unix_timestamp(seconds), @@ -51,10 +53,10 @@ fn axes(seconds: i64) -> TemporalAxes { /// One fabricated corpus of rows on the unit circle of the representation's leading 2-plane. /// -/// Each row's representation points at its angle and its wire coordinate is the same point, -/// so representation-space and wire-space orderings agree exactly for angles within one -/// half-turn of each other - the geometry every faithful-projector certificate leans on. Equal -/// angles produce byte-equal representations. +/// Each row's representation points at its angle and its wire coordinate is the same point: +/// representation-space and wire-space orderings agree exactly for angles within one half-turn +/// of each other, the geometry every faithful-projector certificate leans on. Equal angles +/// produce byte-equal representations. struct Corpus { ids: Vec, representations: MatrixN, @@ -62,6 +64,10 @@ struct Corpus { } impl Corpus { + /// Builds a corpus whose wire coordinates are unit-circle points. + /// + /// Each `(id, angle)` row places at the unit-circle point at that angle, coupled to the + /// representation. fn new(rows: &[(u128, f32)]) -> Self { let mut corpus = Self::decoupled( rows, @@ -88,6 +94,7 @@ impl Corpus { } } + /// The corpus as generation columns under `id`, recorded at `at` (or unrecorded when `None`). fn columns(&self, id: GenerationId, at: Option) -> GenerationColumns<'_> { GenerationColumns::new( id, @@ -102,10 +109,9 @@ impl Corpus { /// The standing pair holds five stable rows, one revised row, one departure, and two arrivals. /// -/// The later generation's arrivals sit at rows 6 (entity 8, novel) and 7 (entity 9, whose +/// The later generation's arrivals are at rows 6 (entity 8, novel) and 7 (entity 9, whose /// bytes equal the departed entity 7's, hence seen). Entity 6 revises its bytes between the -/// generations. Every stable row is byte-distinct, so stable classes and stable rows -/// coincide. +/// generations. Every stable row is byte-distinct: stable classes and stable rows coincide. fn standing_pair() -> (Corpus, Corpus) { let earlier = Corpus::new(&[ (1, 0.1), @@ -144,6 +150,9 @@ const STANDING_EDGES: &IdSlice = /// An edgeless later generation. const NO_EDGES: &IdSlice = IdSlice::from_raw(&[]); +/// Replay sizes for the standing pair. +/// +/// The replay runs four queries and comparisons with one control and one neighbourhood of size one. fn one_neighbourhood() -> ReplaySizes { ReplaySizes { queries: NonZero::new(4).expect("the fixture query cap is nonzero"), @@ -156,6 +165,9 @@ fn one_neighbourhood() -> ReplaySizes { } } +/// Runs the arrival replay over the standing pair at `seed`. +/// +/// The edges draw at `seed` under the one-neighbourhood sizes. fn standing_replay(seed: u64) -> Result { let (earlier, later) = standing_pair(); ArrivalReplay::from_columns( @@ -172,7 +184,7 @@ fn standing_replay(seed: u64) -> Result { /// The faithful projector, projecting each row to the wire point its own leading components name. /// /// On the aligned fixture geometry this reproduces every universe member's published wire -/// coordinate exactly, so the deployed ordering equals the reference ordering. +/// coordinate exactly: the deployed ordering equals the reference ordering. struct PlanarProjector; impl PublishedProjector for PlanarProjector { @@ -207,6 +219,10 @@ impl PublishedProjector for ScriptedProjector { } } +/// Partitions the standing pair into five stable rows, one revised row and two arrivals. +/// +/// Partitioning the standing pair finds five stable rows on matching positions, one revised row, +/// and two arrivals at later rows 6 and 7, one of them seen before. #[test] fn partition_standing_pair() { let (earlier, later) = standing_pair(); @@ -238,6 +254,10 @@ fn partition_standing_pair() { assert_eq!(populations.arrivals_seen, 1); } +/// Groups byte-equal representations into three stable classes and two arrival classes. +/// +/// Byte-equal representations group into classes: three stable classes with member counts `2, 2, 1` +/// at their lowest rows, and two arrival classes, a duplicated novel one and a seen singleton. #[test] fn class_formation() { // Stable rows pair up at angles 0.2 and 0.9 beside a singleton at @@ -280,6 +300,10 @@ fn class_formation() { assert_eq!(arrivals[arrival_class(1)].novelty, Novelty::Seen); } +/// An earlier generation without temporal axes fails with `UnrecordedTemporalAxes`. +/// +/// A pair whose earlier generation has no recorded temporal axes fails with +/// `UnrecordedTemporalAxes` naming that generation. #[test] fn axes_unrecorded() { let (earlier, later) = standing_pair(); @@ -320,6 +344,7 @@ fn pair_unordered() { } } +/// A later generation with no new identities fails with `EmptyArrivals`. #[test] fn arrivals_empty() { let (earlier, _) = standing_pair(); @@ -406,6 +431,7 @@ fn stable_classes_insufficient() { )); } +/// A neighbourhood size of three over a universe of four fails with `NeighbourhoodDesign`. #[test] fn neighbourhood_oversized() { let (earlier, later) = standing_pair(); @@ -430,6 +456,11 @@ fn neighbourhood_oversized() { )); } +/// Reads every designed count, optimum and identity under the faithful planar projector. +/// +/// Under the faithful planar projector every population count is as designed, both arrivals place, +/// every refit and deployed reading lies at its optimum, every paired difference is exactly zero, +/// and the per-query, class and control rows carry the expected identities and member counts. #[test] #[expect( clippy::float_cmp, @@ -464,8 +495,8 @@ fn faithful_path_optimum() { assert_eq!(report.outcomes, placed_pair); assert_eq!(report.class_outcomes, placed_pair); - // The fixture geometry makes every ordering agree, so each reading - // sits at its optimum and every paired difference is exactly zero. + // The fixture geometry makes every ordering agree: each reading lies at its optimum, and + // every paired difference is exactly zero. let block = &report.neighbourhoods[0]; let class_block = &report.class_neighbourhoods[0]; for (name, row) in [ @@ -498,10 +529,9 @@ fn faithful_path_optimum() { assert_eq!(class_paired.queries, 2); assert_eq!(class_paired.mean.recall, 0.0); - // Per-query rows ride ascending by later row. The novel arrival - // comes first and the seen one second, each placed with zero - // difference. The arrival classes are singletons on the same rows, - // so the class rows mirror them with member count one. + // Per-query rows are ordered ascending by later row. The novel arrival comes first and the + // seen one second, each placed with zero difference. The arrival classes are singletons on + // the same rows: the class rows mirror them with member count one. assert_eq!(report.queries.len(), 2); assert_eq!(report.queries[0].novelty, Novelty::Novel); assert_eq!(report.queries[0].entity, entity(8).into()); @@ -559,12 +589,17 @@ fn incident_edges_once() { ); } +/// Records out-of-frame outcomes with refit readings kept and no deployed or paired ones. +/// +/// A projector placing every arrival out of frame yields out-of-frame outcomes with refit readings +/// kept and no deployed or paired readings. #[test] #[expect( clippy::float_cmp, reason = "the fixture geometry makes the refit readings exactly one" )] fn out_of_frame_keeps_refit() { + /// A projector that places every embedding out of frame at world `(9, 9)`. struct Outside; impl PublishedProjector for Outside { fn project( @@ -603,14 +638,17 @@ fn out_of_frame_keeps_refit() { } } +/// Asks the projector again for the remaining row after a non-finite placement. +/// +/// A non-finite placement in a two-row batch makes the replay ask the projector again for the +/// remaining row alone, recording one non-finite and one placed outcome in both estimands. #[test] fn non_finite_retry() { let replay = standing_replay(7).expect("the standing pair carries the design"); - // Both estimands sample the same two arrival rows, so the plan - // projects two distinct rows in one batch whose first row fails - // non-finitely, and the projector is asked again for the remainder - // alone. + // Both estimands sample the same two arrival rows: the plan projects two distinct rows in + // one batch whose first row fails non-finitely, and the projector is asked again for the + // remainder alone. let placed = ArrivalPlacement::Placed { wire: Vec2::new(0.5, 0.5), }; @@ -638,8 +676,8 @@ fn non_finite_retry() { 1, ); - // The class estimand's representatives are the same rows, so each - // class row reads the one projection its row received. + // The class estimand's representatives are the same rows: each class row reads the one + // projection its row received. assert_eq!(report.class_outcomes, split); assert_eq!(report.class_queries[0].outcome, PlacementOutcome::NonFinite); assert_eq!(report.class_queries[1].outcome, PlacementOutcome::Placed); @@ -647,8 +685,8 @@ fn non_finite_retry() { #[test] fn non_finite_mid_batch_split() { - // A third arrival widens the plan to three rows so the failure can - // sit strictly inside the batch. + // A third arrival widens the plan to three rows, which lets the failure fall strictly + // inside the batch. let earlier = Corpus::new(&[(1, 0.1), (2, 0.5), (3, 0.9), (4, 1.3), (5, 1.7)]); let later = Corpus::new(&[ (1, 0.1), @@ -700,10 +738,9 @@ fn non_finite_mid_batch_split() { #[test] fn duplicate_rows_dedup() { - // The stable population spreads eight rows over six byte-classes - // (two duplicate pairs, four singletons), so the sampled entity - // universe of four rows deduplicates to between two and four - // representatives while both estimands' joint draws still fit. + // The stable population spreads eight rows over six byte-classes (two duplicate pairs, four + // singletons): the sampled entity universe of four rows deduplicates to between two and + // four representatives while both estimands' joint draws still fit. let earlier = Corpus::new(&[ (1, 0.2), (2, 0.2), @@ -762,6 +799,7 @@ fn seed_replay() { assert_eq!(one, two); } +/// A replay report round-trips through JSON to an equal report. #[test] fn report_roundtrip() { let report = standing_replay(7) @@ -809,6 +847,10 @@ fn joint_sample_overflow() { )); } +/// Admits a neighbourhood of size one over a universe of two and refuses it over one. +/// +/// `NeighbourhoodDesign::new` accepts size one over a universe of two and refuses it over a +/// universe of one with `NeighbourhoodDesign`. #[test] fn horizon_design_refusal() { let size = NonZero::new(1).expect("the neighbourhood size is nonzero"); @@ -821,11 +863,12 @@ fn horizon_design_refusal() { )); } +/// A comparison count of `2³²` fails with `UniverseBeyondRankDomain` before any sampling refusal. #[cfg(target_pointer_width = "64")] #[test] fn universe_beyond_rank_domain_refusal() { - // The standing pair could never host this draw, so reaching the - // sampling refusals instead would prove the domain check ran late. + // The standing pair could never host this draw: reaching the sampling refusals instead + // would prove the domain check ran late. let (earlier, later) = standing_pair(); let result = ArrivalReplay::from_columns( @@ -847,7 +890,7 @@ fn universe_beyond_rank_domain_refusal() { )); } -/// Both derivation fixtures pin the metric wiring numerically, so their sizes ride together. +/// Both derivation fixtures pin the metric wiring numerically and share these sizes. fn derivation_sizes() -> ReplaySizes { ReplaySizes { queries: NonZero::new(1).expect("the fixture query cap is nonzero"), @@ -984,14 +1027,11 @@ fn weighting_pair() -> (Corpus, Corpus) { tie is exact in f32" )] fn metric_orientation() { - // The stable rows are entities 1..=6 with rows 1 and 2 byte-equal, - // leaving five stable classes, and one novel arrival sits at row - // 6. Wire columns are - // decoupled from the representations to make trustworthiness and - // continuity read different values: a swapped by_reference/by_map - // argument pair anywhere in the wiring exchanges them and fails - // here. Every expected value is hand-derived from the rank-kernel - // definitions at k = 1, horizon 2. + // The stable rows are entities 1..=6 with rows 1 and 2 byte-equal, leaving five stable + // classes, and one novel arrival is at row 6. Wire columns are decoupled from the + // representations to make trustworthiness and continuity read different values: a swapped + // by_reference/by_map argument pair anywhere in the wiring exchanges them and fails here. + // Every expected value is hand-derived from the rank-kernel definitions at k = 1, horizon 2. // // Seed 12345 realizes these draws (pinned by the seeded sampler): // entity universe = rows {1, 2, 3, 4} (positions p0..p3), @@ -1008,28 +1048,24 @@ fn metric_orientation() { // // Refit wires put the reference-farthest nearest: entity refit // ordering [p0, p1, p2, p3], class refit [cp1, cp2, cp0, cp3]. - // Entity refit at k=1, m=4 (worst = 3): the refit-nearest p0 sits - // at reference rank 2, and that penalty of 2 passes horizon 2, so - // trust = 1 - 2/3 with intrusion = 1. The reference-nearest p3 has - // refit rank 3 (penalty 3), giving continuity = 0 with extrusion - // = 1. - // Trust and continuity read different values, which is the - // orientation witness. Dedup refit at m' = 3 (worst = 2): the same - // penalties normalize to trust 1 - 2/2 = 0, the normalizer split - // made numeric. Class refit at m = 4: nearest cp1 has reference - // rank 2 (trust 1/3, intrusion), reference-nearest cp3 sits last + // Entity refit at k = 1, m = 4 (worst = 3): the refit-nearest p0 has reference rank 2, and + // that penalty of 2 passes horizon 2: trust = 1 - 2/3 with intrusion = 1. The + // reference-nearest p3 has refit rank 3 (penalty 3), giving continuity = 0 with + // extrusion = 1. Trust and continuity read different values, which is the orientation + // witness. Dedup refit at m' = 3 (worst = 2): the same penalties normalize to + // trust 1 - 2/2 = 0, the normalizer split made numeric. Class refit at m = 4: nearest cp1 + // has reference rank 2 (trust 1/3, intrusion), and the reference-nearest cp3 is last // (continuity 0, extrusion). // - // The scripted projector places the arrival at the earlier-frame origin, - // where the earlier wire column mirrors the reference order, so - // every deployed reading is optimal and the paired differences are - // recall +1, trust +2/3, continuity +1, intrusion -1, extrusion -1. + // The scripted projector places the arrival at the earlier-frame origin, where the earlier + // wire column mirrors the reference order: every deployed reading is optimal, and the + // paired differences are recall +1, trust +2/3, continuity +1, intrusion -1, extrusion -1. // // The entity control (row 0) reads from earlier wire (2.5, 4) // against members at x = 3, 4, 2, 1 on the axis: two designed exact // ties (16.25 against p0/p2 and 18.25 against p1/p3) break by // ascending position, and the byte-equal pair ties exactly in - // reference space, so the control reads optimal through three real + // reference space: the control reads optimal through three real // ties. The class control (row 4) reference-ranks row 3 // nearest (0.3 against row 5's 0.4) while its map ordering leads // with cp3, penalty 1 on each side: trust = continuity = 1 - 1/3. @@ -1087,8 +1123,8 @@ fn metric_orientation() { assert_eq!(class_refit.intrusion_rate.get(), 1.0); assert_eq!(class_refit.extrusion_rate.get(), 1.0); - // The deployed placement mirrors the reference order in every - // family, so each deployed row is optimal. + // The deployed placement mirrors the reference order in every family: each deployed row is + // optimal. for (name, deployed) in [ ("entity", &report.neighbourhoods[0].deployed), ("class", &report.class_neighbourhoods[0].deployed), @@ -1113,8 +1149,8 @@ fn metric_orientation() { assert_eq!(paired.mean.intrusion_rate, -1.0); assert_eq!(paired.mean.extrusion_rate, -1.0); - // The one query is novel, so the novel split repeats the whole - // reading and the seen split is empty. + // The one query is novel: the novel split repeats the whole reading, and the seen split is + // empty. assert!(report.neighbourhoods[0].refit_seen.is_none()); assert_eq!( report.neighbourhoods[0] @@ -1156,7 +1192,7 @@ fn metric_orientation() { fn class_weighting() { // The stable population spreads seven rows over five classes: rows // 0..=2 share one representation (class A) and rows 3..=6 are - // singletons. One novel arrival sits at row 7. Each reading family + // singletons. One novel arrival is at row 7. Each reading family // weighs the duplicated class its own way, and the fixture // separates every family numerically at k = 1, horizon 2. // @@ -1167,10 +1203,9 @@ fn class_weighting() { // class universe = representative rows {0, 3, 4, 6} (A, B, C, E), // class control = class D (row 5). // - // Entity refit, m = 4 (worst = 3): refit order [p0, p1, p2, p3] - // leads with an A copy whose reference rank is 2 (the pair ties in - // reference space and breaks by position), so trust = 1 - 2/3; - // reference-nearest D sits last, continuity = 0. The duplicate pair + // Entity refit, m = 4 (worst = 3): refit order [p0, p1, p2, p3] leads with an A copy whose + // reference rank is 2 (the pair ties in reference space and breaks by position): + // trust = 1 - 2/3, and the reference-nearest D is last, continuity = 0. The duplicate pair // holds two of four universe slots: entity weighting. // // Dedup diagnostic, m' = 3 (worst = 2): one A survives and the @@ -1178,11 +1213,10 @@ fn class_weighting() { // entity reading. Its membership still follows the entity draw. // // Class estimand, m = 4 classes at their lowest-row representatives - // (worst = 3): refit order [cp2, cp0, cp1, cp3] leads with C whose - // reference rank is 1 (penalty 1, inside the horizon), so trust = - // 1 - 1/3 and intrusion = 0; reference-nearest E sits last, - // continuity = 0, extrusion = 1. All three trust values differ: - // entity 1/3, dedup 0, class 2/3. + // (worst = 3): refit order [cp2, cp0, cp1, cp3] leads with C whose reference rank is 1 + // (penalty 1, inside the horizon): trust = 1 - 1/3 and intrusion = 0, and the + // reference-nearest E is last, continuity = 0, extrusion = 1. All three trust values + // differ: entity 1/3, dedup 0, class 2/3. let report = derivation_report( weighting_pair(), ArrivalPlacement::OutOfFrame { diff --git a/libs/@local/graph/atlas/src/salt/projector/sample/mod.rs b/libs/@local/graph/atlas/src/salt/projector/sample/mod.rs index c282ce158a5..8f558166771 100644 --- a/libs/@local/graph/atlas/src/salt/projector/sample/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/sample/mod.rs @@ -3,16 +3,14 @@ //! Each training step draws three edge populations: //! //! - [`SemanticEdgeSampler`] draws positive pairs from the semantic graph proportional to their -//! fuzzy weight, with replacement, so gradient work concentrates where the attraction evidence -//! is; +//! fuzzy weight, with replacement: gradient work concentrates where the attraction evidence is; //! - [`RelationEdgeSampler`] draws typed attraction instances, choosing relation types uniformly -//! and capping edges per type, so raw edge frequency cannot hand the layout to high-volume +//! and capping edges per type: raw edge frequency cannot hand the layout to high-volume //! relations; //! - [`OrdinaryNegativeSampler`] draws uniform distinct node pairs and admits them only past every //! veto - self pairs, semantic-positive pairs, and pairs the protection evidence bars from -//! ordinary repulsion. Typed-separation control sets and signed-policy conflicts are further -//! vetoes the admission contract names; the initial generation has no signed policies, so both -//! sets are empty here. +//! ordinary repulsion. The admission contract names two further vetoes, typed-separation control +//! sets and signed-policy conflicts. Both sets are empty here, and this sampler applies neither. //! //! Every sampler draws from a caller-supplied random stream and nothing else: equal artifacts, //! stream types, and seeds reproduce a batch exactly. @@ -39,9 +37,11 @@ use crate::{ /// Weight-proportional semantic-positive edge sampler. /// -/// Draws are independent (with replacement): a duplicate edge in one batch is a legitimate sample, -/// and the estimator needs no without-replacement correction. The drawn weight itself stays out of -/// the emitted pair - proportional sampling already accounts for it. +/// Draws are with replacement, each from the same weight-proportional distribution, and the +/// estimator treats them as independent: a duplicate edge in one batch is a legitimate sample, and +/// no without-replacement correction applies. Independence and exact proportionality are +/// properties of the ideal draw the seeded pseudorandom stream stands in for. The drawn weight +/// itself stays out of the emitted pair - proportional sampling already accounts for it. #[derive(Debug)] pub(crate) struct SemanticEdgeSampler<'graph, N> { graph: SemanticGraphView<'graph, N>, @@ -133,9 +133,9 @@ where } }; - // The last cumulative entry therefore exceeds every target, so the partition point - // lands in `1..=rows`; rows without weight repeat their predecessor's total and are - // never selected. + // The last cumulative entry therefore exceeds every target, and the partition + // point lies in `1..=rows`. Rows without weight repeat their predecessor's total + // and are never selected. let row = self .cumulative .partition_point(|&sum| sum <= target) @@ -152,8 +152,8 @@ where } } - // The walk rebuilds the constructor's partial sums (same values, same order), so it - // reaches the row's total and the target lies strictly below it. + // The walk rebuilds the constructor's partial sums (same values, same order). It + // reaches the row's total, and the target lies strictly below it. let id = chosen.expect("the row's rebuilt weight sums cover every drawn target"); NodePair::new(row, id) }) @@ -178,9 +178,9 @@ pub(crate) struct SampledRelationEdges<'index, N, E> { /// The sampler draws relation types uniformly without replacement, and each selected type /// contributes at most the per-type cap of distinct edges: the cap is the relation objective's own /// semantic anti-domination factor - a high-volume type must not own the geometry by edge count - -/// not a performance knob. Uniform type selection is the strongest anti-skew choice; a -/// square-root-of-edge-count weighting is the sanctioned alternative if quality evidence shows the -/// cap alone starves high-volume relations. +/// not a performance knob. Uniform type selection is the strongest anti-skew choice. A +/// square-root-of-edge-count weighting is the alternative if quality evidence shows the cap alone +/// starves high-volume relations. #[derive(Debug)] pub(crate) struct RelationEdgeSampler<'index, N, E> { groups: &'index [AttractionGroup], @@ -202,7 +202,7 @@ where /// Draws up to `types` relation types, allocating the draw list in `alloc`. /// - /// Fewer types than requested means every type participates; a group smaller than the cap + /// Fewer types than requested means every type participates. A group smaller than the cap /// contributes all its edges. /// /// The per-group edge vectors stay on the global allocator: they belong to @@ -262,7 +262,7 @@ where /// # Panics /// /// This panics when the two views disagree about the row domain. Both artifacts come from one - /// generation, so a mismatch is a wiring defect. + /// generation, and a mismatch is therefore a wiring defect. #[must_use] pub(crate) fn new( semantic: SemanticGraphView<'view, N>, @@ -296,7 +296,7 @@ where alloc: A, ) -> Vec, A> { let rows = u64::try_from(self.semantic.rows()).expect("graph rows fit the row-id encoding"); - // Pairs need two distinct rows, so the empty and singleton corpora sample nothing. + // Pairs need two distinct rows: the empty and singleton corpora sample nothing. let Some(bound) = NonZero::new(rows).filter(|bound| bound.get() >= 2) else { return Vec::new_in(alloc); }; @@ -319,7 +319,7 @@ where let pair = NodePair::new(left, right); - // A vetoed pair stays vetoed; remembering it before the veto checks skips their cost on + // A vetoed pair stays vetoed. Remembering it before the veto checks skips their cost on // repeats. if !seen.insert(pair) { continue; @@ -340,7 +340,7 @@ where /// Returns whether the pair is a semantic-positive edge. /// - /// The graph is symmetric, so one row's adjacency decides. + /// The graph is symmetric, and one row's adjacency decides. fn is_semantic_positive(&self, pair: NodePair) -> bool { self.semantic .row(pair.lhs()) diff --git a/libs/@local/graph/atlas/src/salt/projector/sample/tests.rs b/libs/@local/graph/atlas/src/salt/projector/sample/tests.rs index 2449ed8978b..fbb35c38b84 100644 --- a/libs/@local/graph/atlas/src/salt/projector/sample/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/sample/tests.rs @@ -24,14 +24,17 @@ use crate::{ }, }; +/// Seeds a [`Xoshiro256PlusPlus`] generator from `seed`. fn rng(seed: u64) -> Xoshiro256PlusPlus { Xoshiro256PlusPlus::seed_from_u64(seed) } +/// A node pair from two literal row numbers. fn pair(one: u64, other: u64) -> NodePair { NodePair::new(NodeRowId::new(one), NodeRowId::new(other)) } +/// The `(lhs, rhs)` row numbers of `pairs`, for sorting and comparison. fn keys(pairs: &[NodePair]) -> Vec<(u64, u64)> { pairs .iter() @@ -98,6 +101,9 @@ fn instance( } } +/// Builds the relation indexes over `rows` from certified `policies` and instances. +/// +/// The attraction options are the defaults. fn relation_indexes( rows: usize, policies: &[RelationPolicy], @@ -112,6 +118,7 @@ fn relation_indexes( .expect("the fixture instances satisfy the input contract") } +/// Sixty-four semantic draws from a three-edge graph are all graph edges. #[test] fn semantic_draws_are_graph_edges() { let graph = semantic_graph(4, &[(0, 1, 0.5), (1, 2, 0.25), (2, 3, 1.0)]); @@ -129,6 +136,10 @@ fn semantic_draws_are_graph_edges() { } } +/// Reads the semantic sampler's total weight as exactly `3.5` on the dyadic fixture. +/// +/// The semantic sampler's total weight is the stored sum over both directions of each edge, exactly +/// `3.5` for the dyadic fixture. #[test] fn semantic_total_weight_sums_both_edge_directions() { // The symmetric graph stores each undirected edge twice, and the @@ -148,8 +159,7 @@ fn semantic_total_weight_sums_both_edge_directions() { #[test] fn semantic_draws_follow_the_weights() { - // The second edge's weight is vanishing: one draw landing on it in - // a 128-draw batch would be a 1-in-1e28 event for the fixed seed. + // the weights contrast 1 with 1e-30 to check concentration on the unit-weight edge. let graph = semantic_graph(4, &[(0, 1, 1.0), (2, 3, 1.0e-30)]); let sampler = SemanticEdgeSampler::new(graph.view()).expect("the graph has weight"); @@ -161,6 +171,7 @@ fn semantic_draws_follow_the_weights() { ); } +/// `SemanticEdgeSampler::new` returns `None` for a graph with no edge weight. #[test] fn semantic_sampler_rejects_an_edgeless_graph() { let graph = semantic_graph(3, &[]); @@ -170,6 +181,7 @@ fn semantic_sampler_rejects_an_edgeless_graph() { ); } +/// Equal seeds reproduce a semantic batch and different seeds draw different batches. #[test] fn semantic_sampling_is_seeded() { let graph = semantic_graph(5, &[(0, 1, 0.5), (1, 2, 0.5), (2, 3, 0.5), (3, 4, 0.5)]); @@ -187,6 +199,10 @@ fn semantic_sampling_is_seeded() { ); } +/// Draws both relation types under a per-type cap with distinct edges per group. +/// +/// With six instances of one relation and two of another, both types participate under a per-type +/// cap, each group draws at most the cap, and the drawn edges within a group are distinct. #[test] fn relation_caps_bind_per_type_under_skew() { let policies = [proximal_policy(3), proximal_policy(9)]; @@ -231,6 +247,7 @@ fn relation_caps_bind_per_type_under_skew() { } } +/// Requesting more relation types than the index holds returns every group once, in group order. #[test] fn relation_type_requests_beyond_the_index_return_every_group() { let policies = [proximal_policy(3), proximal_policy(9)]; @@ -250,6 +267,7 @@ fn relation_type_requests_beyond_the_index_return_every_group() { assert_eq!(relations, [3, 9], "all groups participate, in group order"); } +/// Equal seeds reproduce a relation batch and different seeds draw different edges. #[test] fn relation_sampling_is_seeded() { let policies = [proximal_policy(3)]; @@ -293,6 +311,10 @@ fn negative_fixture() -> ( (graph, indexes) } +/// Draws exactly the admissible negative pairs of the four-row fixture exhaustively. +/// +/// An exhaustive negative draw from the four-row fixture yields exactly the pairs that are neither +/// self pairs, semantic edges nor protected. #[test] fn negatives_pass_every_veto() { let (graph, indexes) = negative_fixture(); @@ -302,7 +324,7 @@ fn negatives_pass_every_veto() { ProtectionConfig::default(), ); - // A request for sixteen from an admissible pool of four is pool-limited and exhaustive, so the + // A request for sixteen from an admissible pool of four is pool-limited and exhaustive: the // assertion pins the whole admissible set. The vetoes remove the self pairs, the semantic edge // (0, 1), and the protected pair (2, 3). let mut draws = keys(&sampler.sample_in(16, rng(23), Global)); @@ -310,6 +332,7 @@ fn negatives_pass_every_veto() { assert_eq!(draws, [(0, 2), (0, 3), (1, 2), (1, 3)]); } +/// With the ordinary protection channel off, the protected linked pair joins the admissible pool. #[test] fn disabling_ordinary_protection_admits_linked_pairs() { let (graph, indexes) = negative_fixture(); @@ -326,6 +349,7 @@ fn disabling_ordinary_protection_admits_linked_pairs() { ); } +/// Equal seeds reproduce a negative batch and different seeds draw different batches. #[test] fn negative_sampling_is_seeded() { let graph = semantic_graph(12, &[(0, 1, 0.5)]); @@ -349,6 +373,10 @@ fn negative_sampling_is_seeded() { ); } +/// Yields an empty batch from a two-row graph whose only pair is a semantic edge. +/// +/// A two-row graph whose only pair is a semantic edge has an empty admissible pool and yields an +/// empty batch. #[test] fn tiny_domains_return_shorter_batches() { // With two rows and one semantic edge, the admissible pool is empty. diff --git a/libs/@local/graph/atlas/src/salt/projector/scale/frozen/mod.rs b/libs/@local/graph/atlas/src/salt/projector/scale/frozen/mod.rs index ce8a0db4e96..a59a99ae280 100644 --- a/libs/@local/graph/atlas/src/salt/projector/scale/frozen/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/scale/frozen/mod.rs @@ -11,30 +11,33 @@ //! //! Each `ρ₀` is the median 2D distance from a row to its nearest semantic neighbours - the same //! reading as the live [`LocalScales`](super::LocalScales), taken once, over a neighbour index -//! set that freezes with the value. The set matters as much as the number: the per-row band -//! bounds every row's displacement from the boundary field, a median is 1-Lipschitz in the -//! uniform norm of its inputs, so the frozen `ρ₀` mis-states the live local scale by at most -//! twice the band - entry by entry, over the same index set. Re-selecting neighbours at -//! comparison time would compare medians over different sets and void the bound, so -//! [`FrozenRuler::live_scales`] reads the live field over the frozen sets and consumes no -//! neighbour table. +//! set that freezes with the value. The set matters as much as the number. The per-row band +//! bounds every row's displacement from the boundary field, and a median is 1-Lipschitz in the +//! uniform norm of its inputs. Therefore the frozen `ρ₀` mis-states the live local scale by at +//! most twice the band - entry by entry, over the same index set. Re-selecting neighbours at +//! comparison time would compare medians over different sets and void the bound. +//! [`FrozenRuler::live_scales`] therefore reads the live field over the frozen sets and consumes +//! no neighbour table. //! //! `ε = ε_rel · s_ref` shifts coincident rows (`ρ₀ = 0`) off zero. The factored form is unit -//! covariance: `σ₀` must be homogeneous of degree one in world units, so the declared number -//! `ε_rel` is dimensionless and the units come from `s_ref`, the RMS spread of the boundary field -//! about its centroid - strictly positive on any publishable map, and indifferent to the +//! covariance: `σ₀` must be homogeneous of degree one in world units. The declared number `ε_rel` +//! is dimensionless, and `ε` takes its world units from `s_ref`, the RMS spread of the boundary +//! field about its centroid - strictly positive on any publishable map, and indifferent to the //! duplicate stratum that zeroes the median of `ρ₀`. The declared `ε_rel` must sit inside a -//! two-sided dimensionless window: at least `κ_ε · β_proj` when the replicate band exists, so the -//! duplicate stratum's response to band-legal movement stays bounded by `2/κ_ε` instead of -//! growing as `1/ε`; at most the declared quantile of the positive `ρ₀` over `s_ref`, so the -//! regularizer stays small against the corpus's own local-scale distribution. An empty window -//! says replicate noise is not small against that distribution, and no ruler regularization is -//! honest there - the freeze refuses rather than squeezes. +//! two-sided dimensionless window. The lower bound, `κ_ε · β_proj`, binds when the replicate band +//! exists: it keeps the duplicate stratum's response to band-legal movement bounded by `2/κ_ε` +//! instead of growing as `1/ε`. The upper bound is the declared quantile of the positive `ρ₀` +//! over `s_ref`: it keeps the regularizer small against the corpus's own local-scale +//! distribution. An empty window says replicate noise is not small against that distribution, +//! and no ruler regularization is honest there - the freeze refuses rather than squeezes. //! //! Every freeze-time failure is one refusal class, [`InvalidRuler`]: a reference that cannot //! exist, an undeclared or out-of-window `ε_rel`, a degenerate spread, and a value-domain -//! violation all mean the estimand's denominator does not exist, so no training starts. None of -//! them changes behaviour - there is no degraded mode. +//! violation all mean the estimand's denominator does not exist. The reference and declaration +//! checks are coordinate-free and refuse at session admission, before the opening segment. The +//! measured checks run at the phase boundary `K`, after the opening segment has trained, and +//! their refusal ends the run before the target phase starts. None of them changes behaviour - +//! there is no degraded mode. #[cfg(test)] mod tests; @@ -64,24 +67,29 @@ use crate::{ /// half and the representation checks alone. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct RulerFloor { - /// `κ_ε`: the dimensionless sensitivity constant. The duplicate stratum's response to - /// band-legal zero-field movement is bounded by `2/κ_ε`, so this constant prices how much - /// coincident-stratum sensitivity the objective tolerates. Its value is an open owner - /// decision. Its role in the lower test is not. + /// `κ_ε`: the dimensionless sensitivity constant. + /// + /// The duplicate stratum's response to band-legal zero-field movement is bounded by `2/κ_ε`. + /// This constant prices how much coincident-stratum sensitivity the objective tolerates. Its + /// value remains an open choice. Its role in the lower test is fixed. pub kappa_epsilon: Positive, - /// `β_proj`: the dimensionless per-row projection radius, the band constraint's size in - /// units of `s_ref`. + /// `β_proj`: the dimensionless per-row projection radius. + /// + /// The band constraint's size in units of `s_ref`. pub projection_band: Positive, } /// The declared constants a ruler freeze validates. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct RulerParameters { - /// `ε_rel`: the dimensionless regularizer. Its value is an open owner decision inside the - /// window. The window itself is fixed. + /// `ε_rel`: the dimensionless regularizer. + /// + /// Its value is an open choice inside the window. The window itself is fixed. pub epsilon_rel: Positive, - /// The declared quantile defining the window's upper bound: `q⁺(ρ₀)` is the smallest - /// positive local scale with at least this share of the positive scales at or below it. + /// The declared quantile defining the window's upper bound. + /// + /// `q⁺(ρ₀)` is the smallest positive local scale with at least this share of the positive + /// scales at or below it. pub scale_quantile: PositiveUnitFraction, /// The window's lower half, present when the band artifact exists. pub floor: Option, @@ -92,13 +100,16 @@ pub(crate) struct RulerParameters { pub(crate) struct FrozenRuler { /// `ρ₀` per row: the median 2D distance to the frozen neighbour set on the boundary field. scales: Box>, - /// `ρ₀ + ε` per row, precomputed at the freeze so the per-pair read is one product and one - /// root. In the typed domain by the two representation checks: at least `ε`, and finite - /// under the largest scale. The add happens once here instead of once per pair, with the - /// same f32 arithmetic, so the precomputation is value-identical. + /// `ρ₀ + ε` per row, precomputed at the freeze. + /// + /// The per-pair read is then one product and one root. In the typed domain by the two + /// representation checks: at least `ε`, and finite under the largest scale. The add happens + /// once here instead of once per pair, with the same f32 arithmetic, and the precomputation + /// is value-identical. shifted: Box>, - /// The frozen neighbour index sets hold one row of slots per node row, each set in - /// ascending stored-distance order with ties in row order. + /// The frozen neighbour index sets, one row of slots per node row. + /// + /// Each set is in ascending stored-distance order with ties in row order. neighbours: IdMatrix, /// `s_ref`: the boundary field's RMS spread about its centroid. reference_spread: Positive, @@ -117,16 +128,18 @@ where /// The checks run in declaration order: local scales and their index sets over the /// zero-condition field, the reference spread, the window's upper bound from the positive /// scales, window emptiness and membership, then the two representation checks on the - /// absolute epsilon. The first failed check is the refusal. Rows measure in parallel, and - /// every reduction is bit-deterministic under any thread schedule: the results are declared - /// constants persisted with the generation, so a replay must reproduce them exactly. + /// absolute `ε`. The first failed check is the refusal. Rows measure in parallel, and + /// every reduction is bit-deterministic under any thread schedule. The results are declared + /// constants persisted with the generation, and a replay must reproduce them exactly. /// /// # Errors /// /// Returns [`InvalidRuler`] carrying the first failed check: a non-finite scale reading, a /// spread outside the positive f32 domain, no positive scale to read the window's upper - /// bound from, an empty window, an out-of-window `ε_rel`, or an epsilon whose coincident or - /// densest pair product leaves the value domain. + /// bound from, an empty window, an out-of-window `ε_rel`, or an `ε` failing a representation + /// check (the product `ε_rel · s_ref` underflowing the working precision, the rounded `ε`'s + /// exact square below the domain's minimum, or the largest ε-shifted scale's widened square at + /// or above the `f32` maximum). /// /// The field covers the table's rows and the table stores at least one neighbour per row - /// wiring contracts checked in debug builds, since both artifacts come from one generation. @@ -171,7 +184,7 @@ where // The spread reduction is bit-deterministic under any thread schedule (the field's // contract): the narrowed value becomes a declared constant persisted with the - // generation, so a replay must reproduce it exactly. Finite with no scan: the boundary + // generation, and a replay must reproduce it exactly. Finite with no scan: the boundary // forward refuses a diverged frame before any freeze sees it. let spread = zero_field.rms_spread(); let reference_spread = @@ -202,12 +215,14 @@ where }); } - // The representation checks keep every stored reading representable. The narrowed ε - // must itself be an f32 value: an ε_rel small against s_ref underflows it to zero, and - // the refusal then carries the exact double product, the reading no working precision - // holds. The floor comparison requires the coincident pair's exact product ε² at or - // above the domain's minimum, so every pair denominator stays inside the validated - // window rather than at its subnormal edge. + // The representation checks declare the window every stored reading lies in. The product + // ε_rel · s_ref first rounds into the working f32 value ε: an ε_rel small against s_ref + // underflows it to zero, and the refusal then carries the exact double product, the + // reading no working precision holds. The floor comparison then squares the rounded ε + // exactly in double and requires that square, the coincident pair's product, at or above + // the domain's minimum positive value. It is an admission bound on that exact square, + // stricter than f32 rounding, which rounds some refused squares up to the smallest + // subnormal rather than to zero. let Some(epsilon) = parameters.epsilon_rel.checked_mul(reference_spread) else { return Err(InvalidRuler::RepresentationFloor { epsilon_abs: parameters.epsilon_rel.mul_wide(reference_spread), @@ -225,12 +240,14 @@ where .map(|&(scale, _)| scale) .max() .expect("the table validation guarantees at least two rows"); - // The exact double sum of two working-precision values stays finite, at most 2¹²⁹, so - // the check compares its square against the f32 maximum rather than the sum itself. A - // sum whose square clears that bound sits near 2⁶⁴, far under where f32 overflows, so - // the narrowed runtime sum cannot overflow once the check passes. + // The widened double sum of two working-precision values stays finite, at most 2¹²⁹, and + // need not be exact (1 + 2⁻⁷⁴ rounds to 1). The check compares its square against the f32 + // maximum and refuses at equality. A sum whose square passes sits below 2⁶⁴, far under + // where f32 overflows, and the narrowed runtime sum cannot overflow once the check + // passes. The bound declares the window and does not say that a refused pair's geometric + // mean would overflow. let shifted_exact = DNonNegative::from(largest) + DPositive::from(epsilon); - // Total: the exact double sum of two working-precision values is at most 2¹²⁹. Its + // Total: the widened double sum of two working-precision values is at most 2¹²⁹. Its // square is at most 2²⁵⁸, far inside the `f64` range. if DPositive::new_unchecked(shifted_exact.get() * shifted_exact.get()) >= DPositive::from(Positive::MAX) @@ -247,8 +264,8 @@ where neighbours.extend_from_slice(&set[..set_len]); } - // Every scale is at most the checked largest, and rounding is monotone, so the typed - // add cannot leave the domain. + // Every scale is at most the checked largest, and rounding is monotone. The typed add + // therefore cannot leave the domain. let shifted: IdVec<_, _> = scales.iter().map(|&scale| scale + epsilon).collect(); Ok(Self { @@ -263,10 +280,11 @@ where /// Returns the pair's denominator `σ₀ = √((ρ₀(source)+ε)(ρ₀(target)+ε))`. /// - /// The reads hit the precomputed ε-shifted scales, so one call is two loads, a product, and - /// a root, and the geometric mean is total on its own: the widened product is exact, and - /// the mean of two representable positives is representable. The freeze-time window checks - /// bound where inside the domain the reading can land. + /// One call is two loads of the precomputed ε-shifted scales, a product, and a root. The + /// geometric mean is total on its own: the widened product of two `f32` values is exact in + /// `f64`, and the mean of two finite positives rounds within their range (the mean of `1` and + /// `2` is `√2`, rounded). The freeze-time window checks bound where inside the domain the + /// reading can lie. /// /// # Panics /// @@ -279,15 +297,17 @@ where /// Measures the live field's local scales over the frozen neighbour sets. /// - /// This is the staleness comparison's reading: the band bounds every row's displacement from + /// This is the staleness comparison's reading. The band bounds every row's displacement from /// the boundary field, each frozen set is fixed, and a median is 1-Lipschitz in the uniform - /// norm of its inputs, so `|live − frozen| ≤ 2·band` holds row by row - over the frozen sets - /// and only there. Rows are independent and computed in parallel. + /// norm of its inputs. Therefore `|live − frozen| ≤ 2·band` holds row by row - over the frozen + /// sets and only there. Rows are independent and computed in parallel. /// /// # Errors /// - /// Returns [`NonFiniteScale`] naming the smallest affected row when a live distance - /// overflows the finite range (pre-divergence coordinates). + /// Returns [`NonFiniteScale`] naming the smallest row whose selected median is non-finite: a + /// live distance's `f32` square or sum overflowed to `+∞` between finite coordinates, and the + /// escaped distances reached the median. Overflowed distances sorted past the median leave a + /// row's scale finite. /// /// The coordinates cover the frozen row count - a wiring contract checked in debug builds, /// since the field and the ruler come from one run. @@ -470,7 +490,7 @@ where /// /// The reading is the smallest positive scale with at least a `scale_quantile` share of the /// positive scales at or below it: rank `⌈q·m⌉` of the ascending positive scales. The sort runs -/// in parallel. Equal scales are interchangeable, so instability changes nothing. +/// in parallel. Equal scales are interchangeable, and sort instability changes nothing. fn positive_quantile( scales: impl Iterator, parameters: RulerParameters, diff --git a/libs/@local/graph/atlas/src/salt/projector/scale/frozen/refusal.rs b/libs/@local/graph/atlas/src/salt/projector/scale/frozen/refusal.rs index 66d827f1848..46c72bcec3b 100644 --- a/libs/@local/graph/atlas/src/salt/projector/scale/frozen/refusal.rs +++ b/libs/@local/graph/atlas/src/salt/projector/scale/frozen/refusal.rs @@ -4,38 +4,47 @@ use core::{error::Error, fmt}; use crate::math::{DNonNegative, DPositive, Positive}; -/// An invalid ruler refuses before training, and every failed check is this one class. +/// The ruler's one refusal class, carrying the failed check's reading. /// /// A missing reference, a missing `ε_rel`, an out-of-window `ε_rel`, and a representation failure -/// are the same refusal - the estimand's denominator does not exist, so no fit starts. The -/// variants carry the failed check's reading and nothing branches on them: there is no degraded -/// mode. The missing-reference and missing-epsilon variants are the trainer's to construct, -/// where schedule and configuration are validated. The rest are this module's. +/// are the same refusal - the estimand's denominator does not exist, and the target phase does +/// not start. The variants carry the failed check's reading and nothing branches on them: there +/// is no degraded mode. The missing-reference and missing-epsilon variants are the trainer's to +/// construct at session admission, before the opening segment, where schedule and configuration +/// are validated. The rest are this module's, measured at the phase boundary after the opening +/// segment has trained. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum InvalidRuler { - /// The schedule names no relation boundary, so no zero-condition field exists to measure on: - /// the zero field must exist as a trained control before it can serve as one. + /// The schedule names no relation boundary, and no zero-condition field exists to measure on. + /// + /// The zero field must exist as a trained control before it can serve as the ruler's + /// reference. MissingReference, - /// No `ε_rel` is declared, so the regularizer does not exist. + /// No `ε_rel` is declared, and the regularizer therefore does not exist. MissingEpsilon, - /// A local scale overflowed the finite range at the freeze, naming the smallest affected - /// row: the boundary field carries pre-divergence coordinates. + /// A frozen local scale, a row's selected median distance, overflowed the finite range. + /// + /// The variant names the smallest affected row. A 2D distance's `f32` square or sum + /// overflowed to `+∞` between finite boundary coordinates, and the escaped distances reached + /// the row's median. Overflowed distances sorted past the median leave a scale finite. NonFiniteScale { /// The smallest affected node row. row: N, }, - /// The boundary field's RMS spread is not a strictly positive f32, so no degree-one unit - /// carrier exists. Zero spread means every row coincides. An unrepresentable spread means - /// the field is already past the working precision. + /// The boundary field's RMS spread is not a strictly positive `f32`. + /// + /// No degree-one unit carrier therefore exists. Zero spread means every row coincides. An + /// unrepresentable spread means the field is already past the working precision. SpreadOutOfDomain { /// The spread as measured, in double precision. spread: f64, }, - /// Every local scale is zero, so the window's upper bound has no positive-scale - /// distribution to read. + /// Every local scale is zero, and the window's upper bound has no positive scales to read. NoPositiveScale, - /// The window is empty: the lower test's floor exceeds the upper bound, so replicate noise - /// is not small against the corpus's local-scale distribution and no `ε_rel` is honest. + /// The window is empty: the lower test's floor exceeds the upper bound. + /// + /// Replicate noise is then not small against the corpus's local-scale distribution, and no + /// `ε_rel` is honest. EmptyWindow { /// `κ_ε · β_proj`. floor: DPositive, @@ -51,17 +60,28 @@ pub(crate) enum InvalidRuler { /// `q⁺(ρ₀) / s_ref`. ceiling: DNonNegative, }, - /// `ε² = (ε_rel · s_ref)²` falls below the value domain's minimum positive value, so a - /// coincident pair's product would round to zero. + /// The absolute `ε` fails the floor of the value domain. + /// + /// The product `ε_rel · s_ref` first rounds into the working `f32` value `ε`, and the check + /// then squares that rounded `ε` exactly in double precision. The variant arises when the + /// product underflows the working precision to zero, or when the rounded `ε`'s exact square + /// falls below the domain's minimum positive value. The bound is an admission window on that + /// exact square: it also refuses an `ε` whose square `f32` arithmetic would round up to the + /// smallest subnormal rather than to zero. RepresentationFloor { - /// `ε_rel · s_ref` in double precision: the narrowed working value widened exactly, - /// or the exact product where narrowing underflows to zero. + /// The absolute `ε` the floor check refused, in double precision. + /// + /// The rounded working value `ε` widened exactly when its exact square fell below the + /// minimum, or the exact double product `ε_rel · s_ref` when its narrowing into the + /// working precision underflowed to zero and no working value exists. epsilon_abs: DPositive, }, - /// The largest ε-shifted scale's square leaves the finite range, so a pair of the densest - /// rows would overflow. + /// The largest ε-shifted scale's widened square reaches or exceeds the `f32` maximum. + /// + /// The check refuses at equality. It declares the representation window and does not by + /// itself say that the pair of largest local scales' geometric mean would overflow. RepresentationCeiling { - /// `max ρ₀ + ε`, the exact double sum. + /// `max ρ₀ + ε`, the widened double sum. shifted_scale: DPositive, }, } diff --git a/libs/@local/graph/atlas/src/salt/projector/scale/frozen/tests.rs b/libs/@local/graph/atlas/src/salt/projector/scale/frozen/tests.rs index ccb24ebbe31..fb269e75d9a 100644 --- a/libs/@local/graph/atlas/src/salt/projector/scale/frozen/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/scale/frozen/tests.rs @@ -1,8 +1,8 @@ //! Certificates for the frozen ruler. //! -//! Exact-arithmetic fixtures pin the freeze's readings. Medians, quantiles, and the reference -//! spread are exactly representable, so the asserted constants are exact contracts rather than -//! tolerances. +//! Exact-arithmetic fixtures pin the freeze's readings. Every asserted median, quantile, spread +//! and denominator is exactly representable, and the asserted constants are therefore exact +//! contracts rather than tolerances. #![expect( clippy::float_cmp, @@ -29,8 +29,9 @@ fn frame(points: &[Vec2]) -> &FinitePointField { FinitePointField::new_unchecked(IdSlice::from_raw(points)) } -/// A neighbour table from per-row `(column, stored distance)` lists in ascending column -/// order. Every row lists the same count, and stored distances live in the cosine range. +/// A neighbour table from per-row `(column, stored distance)` lists in ascending column order. +/// +/// Every row lists the same count, and stored distances lie in the cosine range. fn table(rows: usize, entries: impl Fn(usize) -> Vec<(usize, f32)>) -> Knn { let mut indptr = vec![0_u64]; let mut columns = Vec::new(); @@ -62,6 +63,12 @@ fn twin_table(rows: usize) -> Knn { }) } +/// Builds ruler parameters without a floor. +/// +/// # Panics +/// +/// Panics when `epsilon_rel` is not finite and positive, or `quantile` is not finite and in `(0, +/// 1]`. fn params(epsilon_rel: f32, quantile: f64) -> RulerParameters { RulerParameters { epsilon_rel: Positive::new(epsilon_rel).expect("test epsilon is positive"), @@ -71,6 +78,11 @@ fn params(epsilon_rel: f32, quantile: f64) -> RulerParameters { } } +/// Adds a ruler floor to `parameters`. +/// +/// # Panics +/// +/// Panics when `kappa_epsilon` or `projection_band` is not finite and positive. fn with_floor( mut parameters: RulerParameters, kappa_epsilon: f32, @@ -83,6 +95,8 @@ fn with_floor( parameters } +/// Freezes exact declared constants for rows at distance two. +/// /// Rows at distance two read `ρ₀ = 2` each, `s_ref = 1`, and a window ceiling of `2` - every /// declared constant is exact. #[test] @@ -115,10 +129,11 @@ fn freezes_scales_sets_and_constants_from_the_boundary_field() { ); } -/// Seventeen rows make row 0's table sixteen entries wide, one more than the set uses. Row 1 -/// carries the largest stored distance, so the frozen set is rows 2..=16 and the frozen -/// median over 2D distances `{2..16}` is 9. The live reading then follows the frozen set: -/// moving row 5 far away re-reads over the same indices and shifts the median to 10. +/// Seventeen rows make row 0's table sixteen entries wide, one more than the set uses. +/// +/// Row 1 carries the largest stored distance. The frozen set is therefore rows 2..=16, and the +/// frozen median over their 2D distances `{2..=16}` is 9. The live reading then follows the frozen +/// set: moving row 5 far away re-reads over the same indices and shifts the median to 10. #[test] fn live_scales_read_the_frozen_sets_not_a_reselection() { let rows = 17; @@ -155,8 +170,10 @@ fn live_scales_read_the_frozen_sets_not_a_reselection() { assert_eq!(scales[NodeRowId::new(0)].get(), 10.0); } -/// Per-row displacements of at most one band move every frozen-set median by at most twice -/// the band: the staleness bound the frozen index sets exist to keep. +/// Bounds every frozen-set median's move by twice the band under one-band displacements. +/// +/// Per-row displacements of at most one band move every frozen-set median by at most twice the +/// band: the staleness bound the frozen index sets exist to keep. #[test] fn staleness_stays_within_twice_the_band() { let rows = 17; @@ -282,8 +299,10 @@ fn an_empty_window_refuses_before_membership() { ); } -/// Coincident twin pairs at two distinct locations give positive spread with every scale -/// zero, so no upper bound exists to read. +/// Refuses coincident twin pairs with positive spread and every scale zero. +/// +/// Coincident twin pairs at two distinct locations give positive spread with every scale zero: no +/// upper bound exists to read. #[test] fn an_all_coincident_corpus_has_no_upper_bound_to_read() { let coordinates = [ @@ -319,8 +338,10 @@ fn zero_spread_refuses_before_the_window() { ); } -/// A window-legal epsilon can still square below the value domain: `ε = 2⁻⁸¹` passes the -/// window (ceiling 2) and refuses at the representation floor. +/// Refuses a window-legal `ε = 2⁻⁸¹` at the representation floor. +/// +/// A window-legal epsilon can still square below the value domain: `ε = 2⁻⁸¹` passes the window +/// (ceiling 2) and refuses at the representation floor. #[test] fn a_subrepresentable_epsilon_refuses() { let coordinates = [Vec2::new(0.0, 0.0), Vec2::new(2.0_f32.powi(-60), 0.0)]; @@ -339,9 +360,11 @@ fn a_subrepresentable_epsilon_refuses() { ); } -/// The densest pair's shifted product must stay finite: every reading here is finite - -/// `ρ₀ = 3·2⁶²`, `ε = 3·2⁶¹` - and the shifted square `(9·2⁶¹)² = 81·2¹²²` reaches past -/// `f32`, so the freeze refuses before any pair could overflow at runtime. +/// Refuses a largest shifted scale whose square reaches past `f32` while every reading is finite. +/// +/// The largest shifted scale must square inside `f32`: every reading here is finite (`ρ₀ = 3·2⁶²`, +/// `ε = 3·2⁶¹`), and the shifted square `(9·2⁶¹)² = 81·2¹²²` reaches past `f32`. The freeze +/// therefore refuses before any pair denominator could overflow at runtime. #[test] fn an_overflowing_shifted_scale_refuses() { let coordinates = [Vec2::new(0.0, 0.0), Vec2::new(3.0 * 2.0_f32.powi(62), 0.0)]; @@ -356,9 +379,11 @@ fn an_overflowing_shifted_scale_refuses() { ); } -/// The window's ceiling moves with the declared order statistic: positives `{2, 2, 4, 4}` -/// read 2 at the median and 4 at the upper quartile, so one `ε_rel` sits outside the first -/// window and inside the second. +/// Moves the window's ceiling with the declared order statistic. +/// +/// The window's ceiling moves with the declared order statistic: positives `{2, 2, 4, 4}` read 2 at +/// the median and 4 at the upper quartile. One `ε_rel` therefore lies outside the first window and +/// inside the second. #[test] fn the_ceiling_follows_the_declared_quantile() { let coordinates = [ @@ -379,8 +404,10 @@ fn the_ceiling_follows_the_declared_quantile() { .expect("the upper quartile raises the ceiling past the declared epsilon"); } -/// Zero scales stay out of the quantile's distribution. A fixture of coincident rows beside -/// one separated pair leaves `q⁺ = 2` rather than zero, which admits a modest `ε_rel`. +/// Zero scales stay out of the quantile's distribution. +/// +/// A fixture of coincident rows beside one separated pair leaves `q⁺ = 2` rather than zero, which +/// admits a modest `ε_rel`. #[test] fn the_quantile_reads_positive_scales_alone() { let coordinates = [ diff --git a/libs/@local/graph/atlas/src/salt/projector/scale/mod.rs b/libs/@local/graph/atlas/src/salt/projector/scale/mod.rs index 3e9f26f185b..3442be4214b 100644 --- a/libs/@local/graph/atlas/src/salt/projector/scale/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/scale/mod.rs @@ -7,7 +7,7 @@ //! them at a configured cadence. //! //! The neighbour set is the [`LOCAL_SCALE_NEIGHBOURS`] nearest rows by stored high-dimensional -//! distance. The neighbour table stores each row's entries in ascending row order, so this module +//! distance. The neighbour table stores each row's entries in ascending row order. This module //! selects the nearest subset by distance and breaks ties by row id. #[cfg(test)] @@ -31,12 +31,15 @@ use crate::{ /// neighbours the median. Tables storing fewer neighbours contribute them all. pub(crate) const LOCAL_SCALE_NEIGHBOURS: usize = 15; -/// A node row's local scale overflowed the finite range. +/// A node row's local scale, its selected median distance, overflowed the finite range. /// /// `row` is the smallest node row whose scale came out non-finite. The coordinates are finite at -/// entry, so the only non-finite reading this computation can produce is a distance that -/// overflows to `+∞`, from pre-divergence coordinates large enough that their difference leaves -/// the finite range. +/// entry, and the only non-finite reading this computation can produce is a 2D distance whose +/// `f32` arithmetic overflows to `+∞`: the coordinate differences square and sum in `f32`, and a +/// finite coordinate difference of `2⁶⁴` or more already overflows its square. The `+∞` sorts +/// last among the row's distances, and the scale is non-finite when the median selection reaches +/// an escaped distance and finite otherwise: one escaped distance is the median of a +/// one-neighbour row, and escaped distances sorted past the median leave the scale finite. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct NonFiniteScale { /// The smallest affected node row. @@ -57,8 +60,8 @@ impl Error for NonFiniteScale where N: fmt::Debug + fmt::Display {} /// Validated per-node local radii in node-row order. /// -/// Every value is a [`NonNegative`]: finite and at least zero, so dividing by a scale plus a -/// positive ε is total. +/// Every value is a [`NonNegative`], finite and at least zero. A scale plus a positive ε is a +/// positive divisor, and it is finite whenever the `f32` sum is below `f32::MAX`. #[derive(Debug, PartialEq)] pub(crate) struct LocalScales(Box>); @@ -79,13 +82,16 @@ where /// /// # Errors /// - /// Returns [`NonFiniteScale`] naming the smallest affected row when a distance overflows the - /// finite range (pre-divergence coordinates). + /// Returns [`NonFiniteScale`] naming the smallest row whose selected median is non-finite: a + /// 2D distance's `f32` square or sum overflowed to `+∞` between finite coordinates, and the + /// escaped distances reached the median. A row whose overflowed distances sort past the median + /// keeps a finite scale. /// /// # Panics /// /// This panics when the coordinate count differs from the table's row count or the table stores - /// no neighbours. Both artifacts come from one generation, so a mismatch is a wiring defect. + /// no neighbours. Both artifacts come from one generation, and a mismatch is therefore a wiring + /// defect. #[expect( clippy::panic_in_result_fn, reason = "row-domain agreement is a wiring contract asserted at entry; the error channel \ @@ -136,9 +142,10 @@ where /// The value is `√((scale(source) + ε) · (scale(target) + ε))`: the geometric mean of the /// pair's ε-shifted local scales. Dividing a pair's distance by it yields the locally /// normalized distance `z`, comparable between dense and sparse map regions. `epsilon` - /// shifts a zero scale off zero, and the geometric mean is total - the widened product is - /// exact and the mean of two representable positives is representable. Every configured - /// `ε` reads a finite normalization. + /// shifts a zero scale off zero. Each shift is an `f32` addition, finite whenever the scale + /// and `ε` sum below `f32::MAX`, and the geometric mean of two finite positives rounds within + /// their range (the mean of `1` and `2` is `√2`, rounded). A finite normalization does not by + /// itself bound the caller's `f32` quotient. /// /// # Panics /// @@ -160,8 +167,8 @@ where /// A placed frame beside local scales covering the same rows. /// /// The pairing claims one row domain and nothing more. Scales are detached measurements that a -/// consumer may read against a re-forwarded frame from a later step, so which frame measured them -/// stays the call site's contract rather than this type's. +/// consumer may read against a re-forwarded frame from a later step. Which frame measured them is +/// the call site's contract rather than this type's. #[derive(Debug, Copy, Clone)] pub(crate) struct ScaledFrame<'frame, N> { /// The placed coordinates. @@ -179,7 +186,7 @@ where /// # Panics /// /// This panics when the scales do not cover the coordinate rows: the pair describes one - /// corpus, so a mismatch is a wiring defect. + /// corpus, and a mismatch is therefore a wiring defect. #[must_use] pub(crate) fn new( coordinates: &'frame FinitePointField, @@ -254,10 +261,12 @@ pub(crate) const fn sorted_median(distances: &[NonNegative]) -> NonNegative { /// Computes one row's median 2D distance to its nearest neighbours. /// -/// A distance between pre-divergence coordinates can overflow, and the escaped `+∞` sorts last -/// under the bit order, so it reaches the median only when overflow dominates the row. The -/// median returns unclaimed, and the table constructor's finish detects divergence at the -/// corpus level rather than per distance. +/// The distance squares and sums the coordinate differences in `f32`, and a coordinate difference +/// of `2⁶⁴` or more overflows its square to `+∞` between finite coordinates. The escaped `+∞` +/// sorts last under the bit order, and the median is non-finite when the selection reaches an +/// escaped distance, which one distance does in a one-neighbour row. The median returns +/// unclaimed, and the table constructor's finish detects divergence at the corpus level rather +/// than per distance. fn row_scale( coordinates: &IdSlice, knn: &KnnView<'_, N>, @@ -276,6 +285,9 @@ where let mut distances = [NonNegative::ZERO; LOCAL_SCALE_NEIGHBOURS]; for (distance, &(_, neighbour)) in distances.iter_mut().zip(&nearest[..count]) { + // `Vec2::distance` squares and sums in `f32`, and an overflow escapes to `+∞` here. With + // debug assertions enabled the scalar square and sum assert at this operation, ahead of + // the finish's refusal. *distance = coordinates[row].distance(coordinates[neighbour]); } distances[..count].sort_unstable(); diff --git a/libs/@local/graph/atlas/src/salt/projector/scale/tests.rs b/libs/@local/graph/atlas/src/salt/projector/scale/tests.rs index 5f92831aaa4..5ca565b9014 100644 --- a/libs/@local/graph/atlas/src/salt/projector/scale/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/scale/tests.rs @@ -1,7 +1,7 @@ //! Certificates for local-scale measurement. //! -//! Fixture distances and coordinates are hand-picked exactly representable values, so the asserted -//! medians are exact contracts. +//! The asserted medians are exact contracts: fixture distances and coordinates are hand-picked +//! exactly representable values. #![expect( clippy::float_cmp, @@ -50,10 +50,10 @@ fn complete_table(rows: usize, distances: impl Fn(usize) -> Vec) -> /// Selection follows stored distance, not storage order. /// -/// Seventeen rows make row 0's table sixteen entries wide, one more than the scale uses. Row 1 - -/// first in storage order - carries the largest stored distance, so the nearest fifteen are rows -/// 2..=16. With row `j` placed at `(j, 0)`, the correct median over 2D distances `{2..=16}` is 9; -/// selecting the first fifteen by storage order would include row 1 and yield 8. +/// Seventeen rows make row 0's table sixteen entries wide, one more than the scale uses. Row 1, +/// first in storage order, carries the largest stored distance. The nearest fifteen are therefore +/// rows 2..=16, and with row `j` placed at `(j, 0)` the median over their 2D distances `{2..=16}` +/// is 9. Selecting the first fifteen by storage order would include row 1 and yield 8. #[test] fn selects_neighbours_by_stored_distance_not_storage_order() { let rows = 17; @@ -100,4 +100,5 @@ fn even_neighbour_counts_use_the_midpoint() { // No test drives a non-finite coordinate through `compute`. The one production caller // (`refresh::forward`) rejects non-finite readback points before the frame exists. The one // reachable non-finite reading, a distance overflowing to +∞ from pre-divergence coordinates, -// is unconstructible in debug builds by design. +// is unconstructible with debug assertions enabled: the assertions in `NonNegative`'s square +// and sum refuse the escaped +∞ before the median sees it. diff --git a/libs/@local/graph/atlas/src/salt/projector/train/batch/draw.rs b/libs/@local/graph/atlas/src/salt/projector/train/batch/draw.rs index bd3a3035e88..df6c4868c14 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/batch/draw.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/batch/draw.rs @@ -51,18 +51,15 @@ pub(crate) struct SupportAnchor { /// Computes one landmark's median layout distance to its nearest skeleton neighbours. /// /// The neighbour count and median convention are the corpus local-scale kernel's -/// ([`insert_nearest`] and [`sorted_median`]); the skeleton is capacity-bounded, so the nearest set -/// comes from a plain pass over the layout. -// PERF: this runs once per landmark and is an all-nearest-neighbours -// scan. The cost is O(S^2) distance evaluations over the -// capacity-bounded skeleton and tens of milliseconds once per fit. If +/// ([`insert_nearest`] and [`sorted_median`]). The skeleton is capacity-bounded, and the nearest +/// set comes from a plain pass over the layout. +// PERF: this runs once per landmark and is an all-nearest-neighbours scan. The cost is O(S²) +// distance evaluations over the capacity-bounded skeleton and tens of milliseconds once per fit. If // skeleton capacity ever rises enough to matter, the fix is algorithmic -// before it is SIMD. Build one kd-tree over the layout (kiddo is -// already in-tree for serving) and take the fifteen nearest per -// landmark in O(S log S) total. The median consumes distances only, so -// tied neighbour choices cannot change the result. An exact index -// reproduces the brute-force output bit for bit. Measure at a raised -// capacity before acting. +// before it is SIMD. Build one kd-tree over the layout (the crate's `math::KdTree` already wraps +// kiddo) and take the fifteen nearest per landmark in O(S log S) total. The median consumes +// distances only, so tied neighbour choices cannot change the result. An exact index reproduces the +// brute-force output bit for bit. Measure at a raised capacity before acting. fn skeleton_scale(coordinates: &IdSlice, ordinal: N) -> NonNegative where N: Id, @@ -87,7 +84,7 @@ impl SupportAnchor { /// Anchors every skeleton landmark at its laid-out coordinate. /// /// With the skeleton's own local ruler as its radius, and each anchor's row translated - /// through `class_of`, the door from the skeleton's corpus rows into the trainer's own row + /// through `class_of`, the map from the skeleton's corpus rows into the trainer's own row /// domain. /// /// The radius is the median layout distance to the landmark's nearest skeleton neighbours. @@ -118,9 +115,9 @@ impl SupportAnchor { /// One step's drawn populations, in corpus row space. /// /// Each family carries its estimator scale, the factor that makes the family's batch sum an -/// unbiased estimate of the family objective that [`super`] documents. For the relation family that -/// objective is the capped-sampling one, a per-type clipped total. An empty family carries a zero -/// scale, and its term contributes nothing. +/// unbiased estimate of the family objective that [`super`] documents, under the sampler's stated +/// draw distribution. For the relation family that objective is the capped-sampling one, a +/// per-type clipped total. An empty family carries a zero scale, and its term contributes nothing. /// /// The population vectors live in the draw's allocator. The relation draws' nested edge vectors /// stay global (see the module documentation). @@ -155,22 +152,22 @@ pub(crate) struct Populations<'index, N, E, A: Allocator = Global> { /// The target objective's unit draws, per-type capped like the relation family. /// /// Drawn exactly when the caller says the target estimand exists - every step from the - /// boundary on a target-configured run - independent of the step's step and of the - /// activation, so a zero-activation reference replicate consumes the identical stream. The - /// batch assembly never touches this family: the target term forwards its own row set at - /// the estimand's two steps instead of riding the batch frame. + /// boundary on a target-configured run - independent of the training step's lens step and of + /// the activation, and a zero-activation reference replicate consumes the identical stream. + /// The batch assembly never touches this family: the target term forwards its own row set at + /// the estimand's two steps instead of reusing the batch frame. pub target: Vec, A>, - /// The step's relation-lens step. + /// The training step's lens step. pub eta: NonNegative, } /// The per-step inputs the sampler combines with its frozen plan. /// -/// The plan's counts are frozen for the run, while these facts change step by step, so a draw +/// The plan's counts are frozen for the run, while these facts change step by step, and a draw /// call names them once as one value. #[derive(Debug, Clone, Copy)] pub(crate) struct DrawContext<'frame, N> { - /// `η`: the step step's relation activation. + /// `η`: the training step's lens step, the relation activation. pub eta: NonNegative, /// The pooled hard-negative frame, absent before the first refresh tick. pub mined: Option<&'frame MinedFrame>, @@ -180,7 +177,7 @@ pub(crate) struct DrawContext<'frame, N> { pub anchors: &'frame [SupportAnchor], /// Whether the target estimand exists at this step. /// - /// A fact of the schedule and the run configuration, never of the activation value, so the + /// A fact of the schedule and the run configuration, never of the activation value: the /// target family's stream consumption is identical between a zero-activation reference /// replicate and a live target run. pub target: bool, @@ -210,7 +207,8 @@ where /// # Panics /// /// This panics when the semantic graph and the protection evidence disagree about the row - /// domain. Both artifacts come from one generation, so a mismatch is a wiring defect. + /// domain. Both artifacts come from one generation, and a mismatch is therefore a wiring + /// defect. #[must_use] pub(crate) fn new( semantic: SemanticGraphView<'view, N>, @@ -243,7 +241,7 @@ where /// # Panics /// /// This panics when the mined frame's row domain disagrees with the artifacts'. Both come from - /// one training run, so a mismatch is a wiring defect. + /// one training run, and a mismatch is therefore a wiring defect. pub(crate) fn draw( &self, context: DrawContext<'_, N>, @@ -313,8 +311,8 @@ where alloc.clone(), ); - // The target family draws last, so a run without it consumes the exact stream the - // released trainer consumes today. + // The target family draws last, and a run without it consumes the exact stream the released + // trainer consumes. let target = if context.target { self.relation.sample_in( self.plan.relation_types, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/batch/mod.rs b/libs/@local/graph/atlas/src/salt/projector/train/batch/mod.rs index a52e51be3a0..aafdd50b8b6 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/batch/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/batch/mod.rs @@ -2,21 +2,22 @@ //! //! [`BatchSampler::draw`] pulls one step's populations from the built artifacts in corpus row //! space, together with each family's estimator scale. [`Batch::assemble`] re-indexes the -//! populations into the batch-local row domain the loss terms speak - corpus keys convert to -//! `BatchRowId` positions here and nowhere else, so the type system keeps the two domains apart - -//! and [`Batch::input`] materializes the model input tensors for the participating rows, padded to +//! populations into the batch-local row domain the loss terms index. Corpus keys convert to +//! [`BatchRowId`] positions here and nowhere else, and the type system keeps the two domains apart. +//! [`Batch::input`] materializes the model input tensors for the participating rows, padded to //! [`ROW_ALIGNMENT`] so the tensor shapes stay inside every GPU kernel's launch constraints. //! //! Draws consume the caller's random stream in a fixed family order (semantic, ordinary, hard, //! relation, landmark, anchor, target), and a skipped family consumes nothing. Equal artifacts, //! plans, stream types, and seeds therefore reproduce a batch exactly. //! -//! The drawing and assembly paths allocate per step, so both expose `_in` variants in the standard +//! The drawing and assembly paths allocate per step, and both expose `_in` variants in the standard //! library's allocator pattern: [`BatchSampler::draw_in`] and [`Batch::assemble_in`] place every //! population and batch vector in the caller's allocator, and the plain methods are defaulting -//! wrappers over the global one. The allocator covers the batch spine; the structures nested inside -//! draws (the relation draws' and [`RelationEdges`]' edge vectors, the gathered [`LocalScales`]) -//! and the tensor buffers of [`Batch::input`] - consumed by the backend - stay global. +//! wrappers over the global one. The allocator covers the batch's own vectors. The structures +//! nested inside draws (the relation draws' and [`RelationEdges`]' edge vectors, the gathered +//! [`LocalScales`]) and the tensor buffers of [`Batch::input`] - consumed by the backend - stay +//! global. use core::{alloc::Allocator, num::NonZero, ops::Range}; use std::alloc::Global; @@ -46,11 +47,11 @@ pub(crate) use self::draw::{BatchSampler, DrawContext, Populations, SupportAncho /// The batch's gathered-row count varies per step - draws and deduplication decide it - and it /// becomes the reduction dimension of the backward matmuls. Some GPU matmul kernels elected by /// shape-bucketed autotune constrain that dimension to a plane-size multiple and abort on shapes -/// that violate it, so every materialized frame pads its row count to this alignment. A generous -/// power of two covers every plausible plane size and collapses the per-step shape variety the -/// election is sensitive to. +/// that violate it. Every materialized frame therefore pads its row count to this alignment. A +/// generous power of two covers every plausible plane size and collapses the per-step shape variety +/// the election is sensitive to. /// -/// Padded rows replicate the last participating row and no population references them, so they +/// Padded rows replicate the last participating row, and no population references them. They /// receive exactly zero force and contribute exactly zero parameter gradient. pub(crate) const ROW_ALIGNMENT: NonZero = NonZero::new(256).expect("the row alignment is non-zero"); @@ -69,11 +70,11 @@ pub(crate) struct NodeColumns<'corpus, N> { /// One assembled minibatch, re-indexed to the batch-local row domain. /// -/// `rows` lists the participating corpus rows in ascending order; a population's [`BatchRowId`] -/// position `i` refers to `rows[i]`. The corpus-to-local map is monotone, so canonical pair +/// `rows` lists the participating corpus rows in ascending order. A population's [`BatchRowId`] +/// position `i` refers to `rows[i]`. The corpus-to-local map is monotone, and canonical pair /// ordering survives re-indexing. /// -/// The batch vectors live in the assembly's allocator; the relation entries' nested edge vectors +/// The batch vectors live in the assembly's allocator. The relation entries' nested edge vectors /// and the gathered scales stay global (see the module documentation). #[derive(Debug)] pub(crate) struct Batch { @@ -107,7 +108,7 @@ pub(crate) struct Batch { /// /// Present exactly when relation edges are. pub scales: Option>, - /// The step's relation-lens step. + /// The training step's lens step. pub eta: NonNegative, } @@ -117,15 +118,15 @@ where { /// Re-indexes drawn populations into the batch-local row domain. /// - /// `scales` is the corpus-wide local-scale table of the step's step; the batch gathers the - /// participating rows' entries. The opening semantic-only segment has no scale tables and - /// passes [`None`] - its draws carry no relation edges. + /// `scales` is the corpus-wide local-scale table of the training step's lens step. The batch + /// gathers the participating rows' entries. The opening semantic-only segment has no scale + /// tables and passes [`None`] - its draws carry no relation edges. /// /// # Panics /// /// This panics when relation edges are present without a scale table, or when a drawn row lies - /// outside the table. The draws and the tables come from one training run, so a mismatch is a - /// wiring defect. + /// outside the table. The draws and the tables come from one training run, and a mismatch is + /// therefore a wiring defect. #[must_use] pub(crate) fn assemble( populations: Populations<'_, N, E>, @@ -272,14 +273,14 @@ where /// /// With the row dimension padded to [`ROW_ALIGNMENT`]. /// - /// The condition vector is the relation lens, and every row carries the batch's step as its - /// single column. The model is parametric in the condition width. + /// The condition vector is the relation lens, and every row carries the batch's lens step as + /// its single column. The model is parametric in the condition width. /// /// # Panics /// /// This panics when the representation and role columns disagree in length or a batch row lies - /// outside them. The columns and the draws come from one generation, so a mismatch is a wiring - /// defect. + /// outside them. The columns and the draws come from one generation, and a mismatch is + /// therefore a wiring defect. #[must_use] pub(crate) fn input( &self, @@ -292,15 +293,15 @@ where /// Materializes the batch's model input at an explicit row alignment. /// /// The row count pads up to the next `alignment` multiple. Padded rows replicate the last - /// participating row and carry the batch's step. No population references them, so they project - /// dead coordinates that receive exactly zero force. Production goes through [`Batch::input`], - /// and certificates pass `1` to obtain the unpadded frame. + /// participating row and carry the batch's lens step. No population references them, and they + /// project dead coordinates that receive exactly zero force. Production goes through + /// [`Batch::input`], and certificates pass `1` to obtain the unpadded frame. /// /// # Panics /// /// This panics when the representation and role columns disagree in length or a batch row lies - /// outside them. The columns and the draws come from one generation, so a mismatch is a wiring - /// defect. + /// outside them. The columns and the draws come from one generation, and a mismatch is + /// therefore a wiring defect. #[must_use] pub(crate) fn input_aligned( &self, @@ -354,8 +355,8 @@ where /// # Panics /// /// This panics when the representation and role columns disagree in length or a row lies - /// outside them. The columns and the draws come from one generation, so a mismatch is a wiring - /// defect. + /// outside them. The columns and the draws come from one generation, and a mismatch is + /// therefore a wiring defect. #[must_use] pub(super) fn input_gather( self, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/error.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/error.rs index 9b900678cf0..1be09bc07c9 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/error.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/error.rs @@ -21,20 +21,24 @@ use crate::{ /// [`TargetRefusal`] states the one consequence they share. #[derive(Debug, Clone, PartialEq)] pub(crate) enum TargetRefusalCause { - /// The ruler could not freeze over the boundary frame, so the estimand's denominator does - /// not exist. + /// The ruler could not freeze over the boundary frame. + /// + /// The estimand's denominator does not exist. Ruler(InvalidRuler), - /// The band constraint refused to freeze: the declared radius does not exist over the - /// stored coordinates. + /// The band constraint refused to freeze. + /// + /// The declared radius does not exist over the stored coordinates. Band(BandRefusal), /// The gauge refused, at the boundary freeze or inside a step's live fit. Gauge(GaugeRefusal), - /// A step's accumulated target reading diverged: an unbounded data-dependent fold left - /// double precision, or the finished estimand overflowed its narrowing to working - /// precision, so the step publishes no reading. + /// A step's accumulated target reading diverged. + /// + /// An unbounded data-dependent fold left double precision, or the finished estimand overflowed + /// its narrowing to working precision. The step publishes no reading. Reading(Diverged), - /// A per-evaluation evidence reading refused, and an evaluation that cannot state its - /// declared evidence publishes nothing. + /// A per-evaluation evidence reading refused. + /// + /// An evaluation that cannot state its declared evidence publishes nothing. Evidence(EvidenceRefusal), } @@ -59,12 +63,13 @@ where /// The failed reading beside everything a refused run measured before it. /// -/// A refusal ends the run with no activation candidate and no target claim, so the prior active -/// generation stays the serving one. The refusal is an -/// outcome rather than an error, so the run record cannot die with an unwinding call. What the -/// record holds at each refusal stage is fixed. The boundary record is present exactly when -/// the radius freeze completed, and the target record exactly when the phase froze. The -/// preserved interval ends at the last completed reading before the failed one. +/// A refusal ends the run with no activation candidate and no target claim, and the prior active +/// generation stays the serving one. The refusal is an outcome rather than an error: the run +/// returns it as a value, and the record reaches the caller on that return path instead of being +/// dropped by an error conversion. What the record holds at each refusal stage is fixed. The +/// boundary record is present exactly when the radius freeze completed, and the target record +/// exactly when the phase froze. The preserved interval ends at the last completed reading before +/// the failed one. #[derive(Debug)] pub(crate) struct TargetRefusal { /// The training step the refusal fired at. @@ -75,7 +80,7 @@ pub(crate) struct TargetRefusal { /// /// The run record as accumulated at the refusal, with the boundary record and the target /// record sealed into it. Rows in the record name the trainer's own row domain: the record - /// is a reading of the run that refused, so no later boundary re-labels it. + /// is a reading of the run that refused, and no later boundary re-labels it. #[cfg_attr( not(test), expect( @@ -135,11 +140,13 @@ pub(crate) enum TrainError { opening: TrainingSchedule, resumed: TrainingSchedule, }, - /// The target configuration's ruler refused at admission: the schedule leaves the ruler no - /// reference to freeze. + /// The target configuration's ruler refused at admission. + /// + /// The schedule leaves the ruler no reference to freeze. Ruler(InvalidRuler), - /// The target configuration's gauge refused at admission: the declared draw cannot support - /// the run's evidence obligation. + /// The target configuration's gauge refused at admission. + /// + /// The declared draw cannot support the run's evidence obligation. Gauge(GaugeRefusal), /// A step's accumulated target reading diverged, and the target pass owns the refusal. TargetReading(Diverged), @@ -147,16 +154,15 @@ pub(crate) enum TrainError { CanonicalStepOutOfSchedule { step: usize }, /// The target estimand's declared unit population carries no mass. /// - /// A forceless attraction index and an index whose every instance weighs zero resolve the - /// same way: the run belongs to the vacuous-record taxonomy, decided at split time, before - /// any fit exists to evaluate. + /// A forceless attraction index and an index whose every instance weighs zero both resolve to a + /// vacuous run, which is decided at split time, before any fit exists to evaluate. EmptyTargetPopulation, /// The target objective needs relation-type draws, and the plan draws none. TargetWithoutUnitDraws, /// The declared penalty's slope dies at a zero violation while the margin is zero. /// - /// Distance equality would then carry no corrective force, which the ruled subgradient - /// constraint forbids: such a penalty pairs only with a positive margin. + /// Distance equality would then carry no corrective force, which the objective forbids: such + /// a penalty pairs only with a positive margin. PenaltyWithoutForceAtEquality, /// A row belongs to more than one declared split population. /// diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/evidence.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/evidence.rs index 64f16cca33d..84342ff4c44 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/evidence.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/evidence.rs @@ -1,8 +1,8 @@ //! The record a training run keeps about itself. //! -//! Training measures itself as it runs, and the measurements return with the trained model, so -//! a reader judges a run from its published record alone. Step-indexed readings append in step -//! order, and the boundary's record carries the full measurement it froze from. +//! Training measures itself as it runs, and the measurements return with the trained model. A +//! reader therefore judges a run from its published record alone. Step-indexed readings append in +//! step order, and the boundary's record carries the full measurement it froze from. use super::{objective::TargetEvidence, options::RelationLens}; use crate::{ @@ -34,8 +34,9 @@ impl FrozenRadius { /// # Panics /// /// This panics when a measured radius fails the lens's radius ordering. The trainer composed - /// this exact energy at the boundary before freezing the radius, so the failure is a defect - /// of the freeze rather than a data condition. + /// this exact energy at the boundary before freezing the radius, and under the lens the radius + /// froze under the failure is a defect of the freeze rather than a data condition. A different + /// lens can fail the ordering as a data condition. #[must_use] pub(crate) fn energy(self, lens: &RelationLens) -> Option { match self { @@ -70,8 +71,8 @@ pub(crate) struct RefreshFraction { /// The weighted fraction of reviewed-Proximal mass at or below the frozen radius. /// /// Measured over the tick's low-step frame and its low-step scale table, the same - /// step/frame-scale pair the freeze measured on, so the series reads calibration drift and - /// never answers a movement question. The boundary tick contributes the first entry, and + /// step/frame-scale pair the freeze measured on. The series therefore reads calibration drift + /// and never answers a movement question. The boundary tick contributes the first entry, and /// later entries drift against it. pub fraction: DNonNegative, } @@ -102,7 +103,7 @@ pub(crate) struct TrainingEvidence { /// Per-tick boundary-drift readings, in step order. /// /// Empty until the boundary froze a measured radius: the fraction is defined against the - /// frozen radius, so pre-boundary and vacuous ticks have nothing to read. + /// frozen radius, and pre-boundary and vacuous ticks have nothing to read. pub fractions: Vec, /// The target objective's run evidence. /// diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/fixture.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/fixture.rs index f6e856dd27d..889d71fe01b 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/fixture.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/fixture.rs @@ -2,7 +2,7 @@ //! //! The corpus, its relation evidence, the training options, and the target objective's declared //! inputs live here once, and construction is deterministic: equal calls read back bit-identical -//! fixtures, so runs meant to share an input share it exactly. +//! fixtures, and runs meant to share an input therefore share it exactly. use core::num::NonZero; @@ -49,13 +49,16 @@ use crate::{ /// Rows per semantic cluster. pub(super) const HALF: usize = 4; +/// Rows in the corpus: two clusters. pub(super) const ROWS: usize = 2 * HALF; +/// Components in the corpus's representation storage. pub(super) const CAPACITY: usize = ROWS * PROJECTOR_DIMENSIONS; /// The reviewed relation type of the boundary fixtures. pub(super) const RELATION: u64 = 11; +/// The fixture generator for `seed`. pub(super) fn rng(seed: u64) -> Xoshiro256PlusPlus { Xoshiro256PlusPlus::seed_from_u64(seed) } @@ -149,6 +152,7 @@ pub(super) fn instance( } } +/// The reviewed Proximal verdict over [`RELATION`]. pub(super) const fn proximal_verdict() -> ResolvedVerdict { ResolvedVerdict { relation: OntologyRowId::new(RELATION), @@ -158,16 +162,24 @@ pub(super) const fn proximal_verdict() -> ResolvedVerdict { /// One training corpus's owned artifacts. pub(super) struct Corpus { + /// The two-clique semantic graph. pub graph: SemanticGraph, + /// The attraction and protection indexes over the relation evidence. pub indexes: RelationIndexes, + /// The complete-graph neighbour table. pub knn: Knn, + /// The aligned representation storage, one row per corpus row. pub storage: BoxedVecN, + /// The node roles, one per corpus row. pub roles: Vec, + /// The landmark anchors, one per cluster. pub landmarks: Vec>, + /// The resolved reviewed verdicts. pub verdicts: Vec, } impl Corpus { + /// Borrows the corpus as trainer inputs with no target objective declared. pub(super) fn inputs(&self) -> TrainerInputs<'_, NodeRowId, EdgeRowId> { TrainerInputs { semantic: self.graph.view(), @@ -217,9 +229,8 @@ pub(super) fn corpus_with( ) .expect("the fixture instances satisfy the input contract"); - // Cluster-patterned representations: a shared sign block plus one - // row-distinct component, so cluster members map to similar inputs - // while every row stays distinguishable. + // Cluster-patterned representations: a shared sign block plus one row-distinct component. + // Cluster members therefore map to similar inputs while every row stays distinguishable. let mut storage = BoxedVecN::zero(); let array = storage.as_array_mut(); for row in 0..ROWS { @@ -270,6 +281,9 @@ pub(super) const fn schedule( .expect("the fixture schedule is valid") } +/// Builds the fixture's training options under `schedule`. +/// +/// The batch plan, energies, and coefficients every fixture run shares. pub(super) fn options(schedule: TrainingSchedule) -> TrainOptions { TrainOptions { schedule, @@ -309,8 +323,9 @@ pub(super) fn options(schedule: TrainingSchedule) -> TrainOptions { } } -/// A target corpus carrying one Proximal relation with two instances, so rows {2, 3, 6, 7} -/// stay force-free for the gauge draw. +/// Builds a target corpus carrying one Proximal relation with two instances. +/// +/// Rows {2, 3, 6, 7} stay force-free for the gauge draw. pub(super) fn target_corpus() -> Corpus { corpus_with( &[proximal_policy(RELATION)], @@ -322,10 +337,15 @@ pub(super) fn target_corpus() -> Corpus { /// The target objective's draw-side fixtures, owned so the trainer inputs can borrow them. pub(super) struct TargetDraws { + /// The force-free rows the gauge draws. pub gauge_rows: Vec, + /// The gauge rows' duplicate classes. pub gauge_classes: Vec, + /// Each row's covariate stratum. pub strata: Vec, + /// The held-out reference population, empty in the fixture. pub held_out: Vec, + /// The matched-control reference population, empty in the fixture. pub matched_controls: Vec, } @@ -334,6 +354,10 @@ pub(super) fn split_digest() -> Sha256Digest { Sha256Digest::of(b"fixture split rule") } +/// Builds target-objective draws over the fixture rows. +/// +/// The draws include gauge rows and classes, one stratum per cluster, and empty reference +/// populations. pub(super) fn target_draws() -> TargetDraws { TargetDraws { gauge_rows: [2, 3, 6, 7].map(NodeRowId::new).to_vec(), @@ -342,13 +366,14 @@ pub(super) fn target_draws() -> TargetDraws { strata: (0..ROWS) .map(|row| StratumId::new(u32::from(!first_cluster(row)))) .collect(), - // Force and gauge claim every fixture row, so the fixture split declares empty - // reference populations. + // Force and gauge claim every fixture row, and the fixture split therefore declares + // empty reference populations. held_out: Vec::new(), matched_controls: Vec::new(), } } +/// The corpus's trainer inputs with the target objective declared over `draws`. pub(super) fn target_inputs<'run>( corpus: &'run Corpus, draws: &'run TargetDraws, @@ -369,6 +394,11 @@ pub(super) fn target_inputs<'run>( } } +/// Builds the fixture's target options at `activation`. +/// +/// # Panics +/// +/// Panics when `activation` is negative or not finite. pub(super) const fn target_options(activation: f32) -> TargetOptions { TargetOptions { canonical_step: nz!(2), diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/inputs.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/inputs.rs index 22381084b1b..cd56fcd7480 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/inputs.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/inputs.rs @@ -43,7 +43,7 @@ pub(crate) struct TrainerInputs<'run, N, E> { pub verdicts: &'run [ResolvedVerdict], /// The target objective's whole configuration, absent on a released-configuration run. /// - /// The declared constants and the borrowed draws travel as one value, so a lone half is + /// The declared constants and the borrowed draws travel as one value, and a lone half is /// unrepresentable. The reference replicate is a present configuration at zero activation, /// never an absent one. pub target: Option>, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/mod.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/mod.rs index f5a00334758..30b4104a17a 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/mod.rs @@ -5,17 +5,17 @@ //! force-mass-weighted 25th percentile of the locally normalized distance `z` over the //! reviewed-Proximal attraction pairs, measured against the boundary's own coordinates - composes //! the relation energy, and opens the step ladder, round-robining the steps across the lens steps -//! with the relation term scaled by each step's step. Refresh ticks at a configured cadence +//! with the relation term scaled by each step's lens step. Refresh ticks at a configured cadence //! re-measure everything defined over current coordinates: per-step local scales, hard negatives //! mined at both lens extremes, and the displacement telemetry. //! //! Optimization is Adam under a cosine learning-rate schedule, with one backward pass per step -//! through the budget surrogate. A seed fixes every batch draw, so draws are deterministic. The +//! through the budget surrogate. A seed fixes every batch draw, and draws are deterministic. The //! backend's gradient accumulation need not be deterministic. //! //! The boundary measures the frozen radius from reviewed evidence, and the full measurement - //! per-type quantiles, mass shares, leave-one-type-out radii, and the evaluated stability -//! certificate - persists in generation evidence, so a reader judges the freeze against data +//! certificate - persists in generation evidence, and a reader judges the freeze against data //! from the published artifact alone. Each scale-bearing tick also re-measures the weighted //! fraction of reviewed mass inside the frozen radius, the drift series beside the freeze. //! A corpus whose attraction index carries no force at all trains vacuously: the relation term @@ -26,7 +26,7 @@ //! measures on, every ladder step enforces the band constraint and folds the batch estimator at //! the estimand's two steps, and every post-boundary tick reads the per-evaluation evidence. //! [`mod@objective`] owns that machinery, and its whole configuration is optional: a released run -//! passes none of it and trains exactly as before. +//! passes none of it and trains exactly as a run without a target objective. mod error; mod evidence; @@ -114,10 +114,10 @@ pub(crate) struct ResumePoint> { /// The resume checkpoint's record of the training state at entry of the boundary step. /// -/// The schedule rides in full so a resumed run can verify it trains under the schedule the opening -/// segment ran under. The scheduler position is redundant with the boundary by construction, and -/// the open path rejects a record where the two disagree. The generator rides as the generator's -/// own 32 state bytes, which pins the pipeline's generator algorithm. +/// The schedule is recorded in full so a resumed run can verify it trains under the schedule the +/// opening segment ran under. The scheduler position is redundant with the boundary by +/// construction, and the open path rejects a record where the two disagree. The generator is +/// recorded as its own 32 state bytes, which pins the pipeline's generator algorithm. #[cfg_attr( not(test), expect(dead_code, reason = "no fit caller resumes from a checkpoint yet") @@ -138,14 +138,18 @@ struct ResumeRecord> { /// The training state at entry of the boundary step. /// /// This is the fork point of a run. The opening segment produces the state and the ladder consumes -/// it. [`Self::write_checkpoint`] serializes it with the caller's generator position, so a resumed +/// it. [`Self::write_checkpoint`] serializes it with the caller's generator position, and a resumed /// ladder starts from the same boundary. The state is opaque and exists only as the output of -/// [`fit_to_boundary`] or of [`Self::open_checkpoint`], so no ladder ever starts from a state no -/// opening segment produced. +/// [`fit_to_boundary`] or of [`Self::open_checkpoint`]. The open path validates the record's +/// structure, schedule and scheduler position, which is what a ladder can check about the state it +/// starts from. /// /// The state excludes the boundary work itself, the radius freeze and the opening refresh. That -/// work happens at entry of [`fit_from_boundary`] and derives from the model alone, so every ladder -/// resumed from one boundary state freezes the bit-equal radius on a deterministic backend. +/// work happens at entry of [`fit_from_boundary`] over the state's model together with the run's +/// inputs (the supplied columns, the neighbour table, the attraction index, the verdicts) and lens +/// settings, which [`fit_from_boundary`] lets a fork vary. Ladders resumed from one boundary state +/// freeze the bit-equal radius when those inputs and settings are equal and the backend executes +/// deterministically. // No Debug: the optimizer adaptor does not implement it. pub(crate) struct BoundaryState> { training: Training, @@ -161,7 +165,7 @@ impl> BoundaryState { /// ladder records its own from the boundary on. /// /// The written bytes are not canonical. The optimizer record is a map whose serialization - /// order may differ between processes, so two writes of one training state need not be + /// order may differ between processes, and two writes of one training state need not be /// byte-equal. Identity lives in the decoded state and round-trips exactly. /// /// # Errors @@ -209,9 +213,10 @@ impl> BoundaryState { /// /// The open path verifies the parameters against `architecture`, the schedule against its /// own validity domain, and the scheduler position against the boundary before it returns - /// the state. The record type fixes the generator state's length. The state round-trip is - /// exact: a generator captured from a live stream is never the all-zero state the - /// generator's seeding remaps. + /// the state. Those checks validate the record's structure and cannot tell whether an + /// opening segment produced it. The record type fixes the generator state's length. The + /// state round-trip is exact: a generator captured from a live stream is never the all-zero + /// state the generator's seeding remaps. /// /// The reopened state's evidence starts fresh and covers the segment it runs, including the /// boundary measurement. The opening segment's evidence belongs to the run that produced the @@ -247,9 +252,10 @@ impl> BoundaryState { }) .ok_or(CheckpointError::InvalidSchedule)?; - // The scheduler advances once per step and reads its position before use, so after the - // opening segment's `boundary` steps it sits at `boundary - 1`. A boundary of zero - // leaves the pre-first-step sentinel, which is what the wrapping subtraction produces. + // The scheduler advances once per step and reads its position before use. After the + // opening segment's `boundary` steps it is therefore at `boundary - 1`. A boundary of + // zero leaves the pre-first-step sentinel, which is what the wrapping subtraction + // produces. if record.scheduler != schedule.boundary().wrapping_sub(1) { return Err(CheckpointError::SchedulerPosition { position: record.scheduler, @@ -287,10 +293,12 @@ impl> BoundaryState { /// `rng`. Equal models, inputs, options, and seeds draw equal batches. Coordinate-level /// reproducibility additionally depends on the backend's own determinism. /// -/// Every step reports its loss to `progress` on evaluation. The run behaves identically under any -/// observer. +/// Every step reports its loss to `progress` on evaluation. The observer's snapshot appetite +/// selects reported rows without consuming training randomness, and the batch draws are the same +/// under every observer. What the observer's callbacks do, and whether the backend executes +/// identically, lie outside that guarantee. /// -/// The run is the composition of [`fit_to_boundary`] and [`fit_from_boundary`]; call the phases +/// The run is the composition of [`fit_to_boundary`] and [`fit_from_boundary`]. Call the phases /// directly to checkpoint or fork at the boundary. /// /// A target objective's refusal is not an error: it returns as [`FitOutcome::TargetRefused`] @@ -307,7 +315,7 @@ impl> BoundaryState { /// # Panics /// /// This panics when the inputs disagree about the corpus row domain or an anchor references a row -/// outside it. All inputs come from one generation, so a mismatch is a wiring defect. +/// outside it. All inputs come from one generation, and a mismatch is therefore a wiring defect. pub(crate) fn fit< N: Id, E: Id, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/evidence.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/evidence.rs index 1ac79e336de..c27b2cfbcd5 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/evidence.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/evidence.rs @@ -40,7 +40,7 @@ where /// # Panics /// /// This panics when the two tables disagree about the row count. Both leave one frozen - /// ruler, so a mismatch is a wiring defect. + /// ruler, and a mismatch is therefore a wiring defect. pub(crate) fn new( scales: Box>, neighbours: IdMatrix, @@ -85,7 +85,7 @@ where &self.neighbours } - /// Neighbour entries per row. + /// Returns the neighbour entries per row. #[cfg_attr( not(test), expect( @@ -103,9 +103,9 @@ where /// The target objective's run evidence, one record per target-configured run segment. /// /// The identity carries the freeze's scalar constants, and the typed artifacts - the boundary -/// field and the ruler's tables - ride beside it in their own containers, so the writer that -/// persists a generation receives the exact values every reading was measured on and owns -/// their file identity. The estimand trajectory and the per-evaluation readings hold the +/// field and the ruler's tables - accompany it in their own containers. The writer that +/// persists a generation therefore receives the exact values every reading was measured on and +/// owns their file identity. The estimand trajectory and the per-evaluation readings hold the /// run's measurements, and the enforcement record's final state closes the record. #[derive(Debug, PartialEq)] pub(crate) struct TargetEvidence { @@ -133,7 +133,7 @@ pub(crate) struct TargetEvidence { /// The final per-row enforcement maxima `u(n) = max ‖z_pre − z_K‖/s_ref`, node-row order. /// /// Already dimensionless: the band record divides every displacement by the reference - /// spread as it accumulates, so the calibration consumes these readings unscaled. + /// spread as it accumulates, and the calibration consumes these readings unscaled. pub row_maxima: Box>, /// The enforcement record's final cumulative readings. pub enforcement: EnforcementSummary, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/inputs.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/inputs.rs index 43a93469d4f..7f85cf9d548 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/inputs.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/inputs.rs @@ -3,8 +3,8 @@ //! Every value here is decided before optimization. [`TargetOptions`] carries the declared //! constants, [`TargetSplit`] the versioned split identity with its reference populations, //! [`GaugeDraw`] the stratified anchor draw, and [`TargetInputs`] binds them beside the -//! covariate strata into the one value the trainer admits. Nothing in this module is -//! measured during the run. +//! covariate strata into the one value the trainer admits. Every value here is fixed before +//! optimization and stays fixed through the run. use core::{fmt, num::NonZero}; @@ -24,8 +24,8 @@ use crate::{ /// /// Every field is a declared value of the run configuration, from the treatment activation and /// the stage radius to the ruler's regularizer window and the gauge rules whose numbers bind -/// only when declared. The penalty rides as a declared member of the sanctioned family, so the -/// wiring fixes no variant choice. +/// only when declared. The penalty is a declared member of the closed family, and the wiring +/// fixes no variant choice. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct TargetOptions { /// The canonical condition's index into [`STEPS`](super::super::super::STEPS). @@ -53,7 +53,7 @@ pub(crate) struct TargetOptions { pub minimum_effective_count: Option, /// The gauge fit's maximum normalized residual. The bar binds only when declared. pub residual_bar: Option, - /// The penalty `φ`, drawn from the sanctioned family. + /// The penalty `φ`, drawn from the closed family. /// /// The family evaluates value and exact slope in one implementation, finite at every finite /// violation by construction. The variant is the caller's declared choice. The declared @@ -90,11 +90,11 @@ impl fmt::Display for SplitPopulation { /// The validated split identity. /// -/// The rule digest rides beside the reference populations one versioned split fixed before +/// The rule digest accompanies the reference populations that one versioned split fixed before /// optimization. The movement participants, gauge anchors, held-out endpoints, and matched /// controls must be pairwise-disjoint under that one rule, and admission checks every pair -/// the trainer can see. The digest rides the run evidence so the population identity stays -/// auditable. +/// the trainer can see. The digest is recorded in the run evidence so the population identity +/// stays auditable. #[derive(Debug, Copy, Clone)] pub(crate) struct TargetSplit<'run, N> { /// The versioned split rule's content digest. @@ -107,8 +107,8 @@ pub(crate) struct TargetSplit<'run, N> { /// One stratified gauge draw, each anchor row beside its duplicate class. /// -/// The pairing is a construction fact, so no consumer re-checks the two lengths and no zip -/// over a malformed draw can silently truncate. +/// The pairing is a construction fact: no consumer re-checks the two lengths, and no zip over a +/// malformed draw can silently truncate. #[derive(Debug, Copy, Clone)] pub(crate) struct GaugeDraw<'run, N> { rows: &'run [N], @@ -120,8 +120,8 @@ impl<'run, N> GaugeDraw<'run, N> { /// /// # Panics /// - /// This panics when the two slices disagree in length. Both come from one draw, so a - /// mismatch is a wiring defect. + /// This panics when the two slices disagree in length. Both come from one draw, and a + /// mismatch is therefore a wiring defect. #[cfg_attr( not(test), expect( @@ -153,12 +153,12 @@ impl<'run, N> GaugeDraw<'run, N> { /// The target objective's whole run configuration. /// -/// The declared constants ride beside the run-borrowed draws. +/// The declared constants accompany the run-borrowed draws. /// /// The split machinery owns every draw here. Gauge membership, the reference populations, and /// the covariate partition are decided before optimization by the one versioned rule the -/// split identity's digest names, and the trainer consumes the outcome. The constants ride the -/// same value, so a configuration cannot arrive half-declared. +/// split identity's digest names, and the trainer consumes the outcome. The constants travel in +/// the same value, and a configuration cannot arrive half-declared. #[derive(Debug, Copy, Clone)] pub(crate) struct TargetInputs<'run, N> { /// The declared constants. diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/mod.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/mod.rs index fc6a5c2b8c0..c6aa2df5471 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/mod.rs @@ -1,5 +1,6 @@ -//! The target objective's wiring through the training run, from the boundary freeze to the -//! per-evaluation evidence. +//! The target objective's wiring through the training run. +//! +//! From the boundary freeze to the per-evaluation evidence. //! //! A target-configured run trains the declared estimand beside the released families. At the //! phase boundary the run freezes every reference the estimand reads - the ruler's `σ₀` table @@ -14,7 +15,7 @@ //! //! The estimand exists at exactly two steps, zero and canonical, whatever step the step's //! round-robin trains the released families at. The pass therefore forwards its own row set - -//! the drawn unit endpoints beside the whole gauge - at both steps, and never rides the released +//! the drawn unit endpoints beside the whole gauge - at both steps, and never reuses the released //! batch frame. The estimand's zero-side calculus lives entirely inside that pass forward: the //! pass's zero values project under the frozen constraint's own clip law, forces are evaluated //! at the projected values, and the deposit composes them through the applied clip derivatives @@ -26,8 +27,8 @@ //! //! The activation is a value, never structure. A zero-activation run draws the same units and //! enforces the same band, then fits the same gauge and reads the same estimand - it adds -//! exactly zero force. The reference replicate is that run, not a build without the code path, -//! so absence and inertness stay distinguishable in the artifact record. +//! exactly zero force. The reference replicate is that run rather than a build without the code +//! path, which keeps absence and inertness distinguishable in the artifact record. mod evidence; mod inputs; @@ -94,8 +95,8 @@ where { /// Admits the target configuration against the run's structure. /// - /// Every check here is coordinate-free and runs at session construction, so an impossible - /// target run fails before its opening segment. The schedule must open the ladder, and the + /// Every check here is coordinate-free and runs at session construction. An impossible target + /// run therefore fails before its opening segment. The schedule must open the ladder, and the /// canonical step must exist within it. The plan must draw unit types, and the corpus must /// carry a weighted unit population under the declared unit law. The declared split /// populations must be pairwise-disjoint - the membership law that keeps the optimizer @@ -136,7 +137,7 @@ where ); if schedule.boundary() == schedule.steps().get() { - // The ladder never opens, so no zero-condition reference exists to freeze. + // The ladder never opens, and no zero-condition reference exists to freeze. return Err(TrainError::Ruler(InvalidRuler::MissingReference)); } let Some(&canonical_eta) = STEPS.get(options.canonical_step.get()) else { @@ -144,8 +145,8 @@ where step: options.canonical_step.get(), }); }; - // Distance equality must carry corrective force under the ruled shape constraint, so - // a penalty whose slope dies at a zero violation pairs only with a positive margin. + // Distance equality must carry corrective force: a penalty whose slope dies at a zero + // violation pairs only with a positive margin. if options.penalty.dead_at_equality() && options.margin.is_zero() { return Err(TrainError::PenaltyWithoutForceAtEquality); } @@ -162,14 +163,14 @@ where })); } - // A forceless corpus declares no unit population: the run resolves into the released - // vacuous taxonomy instead of reading an estimand over nothing. + // A forceless corpus declares no unit population: the run is vacuous instead of reading + // an estimand over nothing. if vacuous { return Err(TrainError::EmptyTargetPopulation); } - // The declared populations are pairwise-disjoint under the one split rule the digest - // names - E5's membership law. One scan covers every pair because each row records + // The declared populations are pairwise-disjoint under the one split rule. One scan + // covers every pair because each row records // the population that claimed it, and the first double claim names the overlap. The // movement participants are the force-bearing endpoints, where relation gradients // reach coordinates directly. @@ -215,9 +216,8 @@ where )?; // `W`: the split-time total unit weight over the whole declared population, derived - // under the declared unit law - the match closes nothing the ledger keeps open. Under - // the per-instance law every admitted instance of every group is one unit, and - // zero-weight units stay members with zero mass. + // under the declared unit law. Under the per-instance law every admitted instance of + // every group is one unit, and zero-weight units stay members with zero mass. let mut weight = DNonNegative::ZERO; match options.unit_law { UnitLaw::PerLinkInstance => { @@ -284,8 +284,8 @@ where /// Converts one step's unit draws into priced units, in the corpus row domain. /// /// The construction conditions on the declared unit law - under the per-instance law each - /// drawn edge is one unit. The ruler is gathered from the frozen table here, before any - /// re-indexing, so the term stays decoupled from the live scale machinery. The weight is + /// drawn edge is one unit. Gathering the ruler from the frozen table here, before any + /// re-indexing, keeps the term decoupled from the live scale machinery. The weight is /// the released factor census, and the inclusion probability is the draw law's full /// per-unit product over the unit's group size. pub(super) fn units( diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/phase.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/phase.rs index af28b4a86eb..429de6621fb 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/phase.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/objective/phase.rs @@ -174,11 +174,12 @@ where }) } - /// Projects the pass's zero readbacks in place under the frozen constraint, collecting - /// each row's applied clip derivative. + /// Projects the pass's zero readbacks in place under the frozen constraint. + /// + /// It collects each row's applied clip derivative on the way. /// /// The pass's zero values are subject to the same frozen constraint as the constitutive - /// field, so each row's readback projects under the identical clip law before any reading + /// field, and each row's readback projects under the identical clip law before any reading /// derives from it. The whole-field application stays the record's one writer: this /// per-row projection records nothing, and each applied derivative arrives typed in the /// pass's own row domain for the deposit's composition. @@ -206,8 +207,8 @@ where /// the live gauge alignment on those coordinates. The batch estimator folds over the /// priced units. The scale pull fans into the anchors. The zero side composes through the /// pass's own applied clip derivatives, and both gradient fields deposit through the - /// surrogate, so the estimand's value and both of its Jacobians belong to the pass's one - /// graph realization. + /// surrogate. The estimand's value and both of its Jacobians therefore belong to the pass's + /// one graph realization. /// /// # Errors /// @@ -242,8 +243,8 @@ where let units = context.units(&self.ruler, draws); let pass = LocalPass::new(&units, self.gauge.rows()); - // The padded forwards prove their whole readback finite, so the pass fields arrive as - // proven prefixes. Alignment padding trails the pass rows, so a padded point diverging + // The padded forwards prove their whole readback finite, and the pass fields arrive as + // proven prefixes. Alignment padding trails the pass rows: a padded point diverging // names the last participating row. let diverged = |offender: NonFinitePoint| { let local = TargetRowId::from_usize(offender.id.as_usize().min(pass.rows.len() - 1)); @@ -252,7 +253,7 @@ where }) }; - // The two-step forwards. Each step's values read back from its own tensor, so every + // The two-step forwards. Each step's values read back from its own tensor, and every // reading the estimator takes shares a graph with the tensor its gradient deposits // through. let canonical_tensor = forward.model.forward(forward.columns.input_gather( @@ -325,7 +326,7 @@ where } } - // Both deposits ride one scalar. + // Both deposits sum into one scalar. let surrogate = deposit(canonical_tensor, &canonical_field, forward.device) + deposit(zero_tensor, &zero_gradient_field, forward.device); @@ -376,11 +377,11 @@ where /// Reads the final model's zero field into the enforcement record. /// - /// The loop's last optimizer update lands after its own step's enforcement application, so - /// this closing application reads the returned model's field once more: the record then - /// covers every update of the interval, with `steps` - one past the last step index - as - /// the closing enforcement point. Without it, the final update could leave the returned - /// field outside the radius while the record reads clean. + /// The loop's last optimizer update comes after its own step's enforcement application. + /// This closing application therefore reads the returned model's field once more: the + /// record then covers every update of the interval, with `steps` - one past the last step + /// index - as the closing enforcement point. Without it, the final update could leave the + /// returned field outside the radius while the record reads clean. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/options.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/options.rs index 5b4a00fc5c9..e4761c7293d 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/options.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/options.rs @@ -1,7 +1,7 @@ //! The validated constants a training run is declared with. //! //! A run's configuration settles before its first step and stays fixed across the whole run. -//! The types here validate that configuration at construction, so the run consumes plain +//! The types here validate that configuration at construction, and the run consumes plain //! values and re-checks nothing step to step. use core::num::NonZero; diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/admission.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/admission.rs index aa46c208ddd..4e509d63927 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/admission.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/admission.rs @@ -17,8 +17,8 @@ use crate::salt::{ /// /// The decision whether the boundary can freeze a radius is structural: force, review coverage, and /// the presence of an opening segment to measure after are properties of the index, the verdicts, -/// and the schedule, not of coordinates, so an impossible boundary fails here instead of after the -/// opening segment. +/// and the schedule rather than of coordinates. An impossible boundary therefore fails here instead +/// of after the opening segment. /// /// The columns, the semantic graph, and the support anchors share one corpus row domain - a /// wiring contract checked in debug builds, since all of them come from one generation. @@ -51,7 +51,7 @@ where ); let force = ForceClasses::measure(inputs.attraction); - // A measured radius needs the semantic-only baseline in front of the boundary; measuring on the + // A measured radius needs the semantic-only baseline in front of the boundary. Measuring on the // untrained init map would freeze a meaningless radius. if force.proximal && options.schedule.boundary() == 0 { return Err(TrainError::UnbaselinedRadius); @@ -93,10 +93,10 @@ impl ForceClasses { } } -/// Whether any reviewed-Proximal verdict covers a group that exerts Proximal force. +/// Returns whether any reviewed-Proximal verdict covers a group that exerts Proximal force. /// /// This is the coordinate-free core of the boundary measurement: the calibration's pair weights are -/// positive exactly on these groups' instances, so a positive measured mass exists if and only if +/// positive exactly on these groups' instances, and a positive measured mass exists if and only if /// this holds. fn reviewed_proximal_force( index: &AttractionIndex, @@ -118,7 +118,7 @@ fn reviewed_proximal_force( }) } -/// Whether a group can exert any force. +/// Returns whether a group can exert any force. /// /// Instances exist and the strength multiplier passes them through. const fn exerts_force(group: &AttractionGroup) -> bool { diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/boundary.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/boundary.rs index 1fd1bb8bd18..6cf5b548d64 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/boundary.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/boundary.rs @@ -18,9 +18,10 @@ use crate::{ }, }; -/// What one phase boundary produces - the composed relation energy and the boundary evidence, -/// with the measured zero-condition frame on a non-vacuous run, for the target freeze to -/// consume against the identical coordinates. +/// What one phase boundary produces. +/// +/// The composed relation energy and the boundary evidence, with the measured zero-condition frame +/// on a non-vacuous run, for the target freeze to consume against the identical coordinates. pub(super) type BoundaryOutcome = ( Option, BoundaryEvidence, @@ -36,7 +37,7 @@ where /// /// Measures the reviewed-Proximal `z` population over the forwarded frame and composes the /// relation energy. The caller owns the frame's forward - the boundary shares one frame - /// between this freeze and the target objective's - and its vacuous early-out, so this path + /// between this freeze and the target objective's - and its vacuous early-out, and this path /// always has force to measure. pub(super) fn freeze_radius( &self, @@ -55,7 +56,7 @@ where let (frozen, radius) = match calibration.radius() { Some(radius) => (radius, FrozenRadius::Measured { radius }), // The entry check admits this run only with reviewed coverage. Reaching here means - // the two mass walks disagree, so this returns an error rather than composing from + // the two mass walks disagree, and this returns an error rather than composing from // nothing. None => return Err(TrainError::MissingProximalReviews), }; diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/draw.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/draw.rs index cff7f1b9230..5b84f3ce99f 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/draw.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/draw.rs @@ -13,10 +13,19 @@ use crate::salt::projector::scale::LocalScales; /// Zero through the opening segment, round-robin across [`STEPS`] once the ladder opens. /// /// A vacuous run pins the zero step throughout. With no relation force the objective is identical -/// at every step, so lens variation could teach the modulation head nothing but batch-sampling -/// noise. A zero condition instead leaves the head's condition weights with exactly zero gradient, -/// so every projected step of a forceless corpus is bit-identical - the flat ladder is a -/// certificate, not an accident. +/// at every lens step, and lens variation could teach the modulation head nothing but +/// batch-sampling noise. A zero condition instead leaves the head's condition weights with exactly +/// zero gradient. The trainer's Adam applies no weight decay, and a parameter with zero gradient +/// and zero moments does not move under it while the optimizer arithmetic is finite and preserves +/// the zero. That Adam narrows its step count to `i32` for the bias corrections, and the argument +/// therefore covers step counts below `2³¹`. The head's bias receives gradient and can train +/// between optimizer updates, and every condition shares it. From the standard initialization (a +/// zero `FiLM` map and bias, fresh moments) a forceless run therefore projects, at any one model +/// state, the same map from representation to coordinate at every lens step, and the flat ladder +/// is a certificate rather than an accident. Bit identity between the projected lens steps holds +/// within one deterministic execution of one batch shape, as the projector's introduction states. +/// A supplied model with nonzero condition weights, or a resumed optimizer with nonzero moments, is +/// outside that argument. #[expect( clippy::integer_division_remainder_used, reason = "the step round-robin is an index modulus" @@ -29,12 +38,12 @@ pub(super) const fn step(step_index: usize, boundary: usize, vacuous: bool) -> u } } -/// Assembles one step's drawn populations at their step. +/// Assembles one step's drawn populations at their lens step. /// /// # Panics /// /// This panics when relation draws happen before a scale-bearing tick. The boundary always runs -/// one, so a miss is a wiring defect. +/// one, and a miss is therefore a wiring defect. pub(super) fn assemble_batch( populations: Populations<'_, N, E>, step_index: usize, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/mod.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/mod.rs index bb8d926efc7..c9e80ce09ff 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/mod.rs @@ -4,15 +4,15 @@ //! shared step loop both phases execute. //! //! A [`Session`] is everything the loop derives deterministically from the borrowed inputs and -//! options; [`Training`] is the mutable state a step advances - model, optimizer, scheduler, +//! options. [`Training`] is the mutable state a step advances - model, optimizer, scheduler, //! evidence. The split is what makes the phase boundary first-class: the opening segment and //! the ladder run the same loop body over the same session, and a resumed ladder rebuilds its //! session from the same artifacts while the training state arrives from the checkpoint. //! //! Refresh products (mined negatives, per-step scale tables) never cross a [`Session::run`] call: -//! the boundary step opens with an unconditional refresh, so a ladder segment re-derives them from -//! the model it starts with. That property is what makes a checkpointed resume bit-equal to the -//! straight run on a deterministic backend. +//! the boundary step opens with an unconditional refresh, and a ladder segment re-derives them +//! from the model it starts with. That property is what makes a checkpointed resume bit-equal to +//! the straight run on a deterministic backend. use core::ops::Range; @@ -122,8 +122,8 @@ where }; let mut plan = options.plan; if vacuous { - // No group exerts force, so relation draws would be dead - // weight at every step; the ladder still runs for the lens + // No group exerts force, and relation draws would be dead + // weight at every step. The ladder still runs for the lens // conditioning. plan.relation_types = 0; } @@ -169,17 +169,19 @@ where }) } - /// Runs the loop over one step range and returns the segment's terminal state: the - /// advanced training state, or the target refusal carrying everything measured before it. + /// Runs the loop over one step range and returns the segment's terminal state. + /// + /// The state is the advanced training state, or the target refusal carrying everything measured + /// before it. /// /// The body is phase-agnostic. The opening segment passes `0..boundary` and the ladder passes /// `boundary..steps`, while the boundary work (radius freeze, unconditional refresh) triggers /// on the step index alone. /// - /// The loop is the only place a step's loss exists before the run ends, so it reports every + /// The loop is the only place a step's loss exists before the run ends, and it reports every /// step to `progress` against the whole schedule rather than against its own range, which is /// one phase of it. This selects the snapshot sample the refresh ticks report once per phase - /// from the observer's stated appetite. The choice consumes no randomness, so the run's draws + /// from the observer's stated appetite. The choice consumes no randomness, and the run's draws /// are the same under every observer. pub(super) fn run, R: Rng + ?Sized, P: Progress>( &mut self, @@ -209,7 +211,7 @@ where let (energy, boundary, frame) = self.boundary(&model, step_index, device)?; self.evaluation.options.relation = energy; // The boundary record enters the run record the moment the radius freeze - // completes, so a target freeze refusal takes the completed measurement + // completes, and a target freeze refusal takes the completed measurement // with it. evidence.boundary = Some(boundary); phase = match self.freeze_phase(frame, step_index) { @@ -278,7 +280,7 @@ where TargetPass::Step(step) => Some(step), TargetPass::Refused(cause) => { // The phase seals its accumulated record into the run record, and the - // whole record rides the refusal. + // whole record accompanies the refusal. evidence.target = Some( phase .take() @@ -317,9 +319,9 @@ where /// Seals the target phase into its run evidence, when a phase exists. /// - /// The loop's final optimizer update landed after its own step's enforcement application, - /// so the record closes over the returned model's zero field before the evidence seals: - /// every update of the interval is read, the last one included. + /// The loop's final optimizer update comes after its own step's enforcement application. + /// The record therefore closes over the returned model's zero field before the evidence + /// seals: every update of the interval is read, the last one included. fn seal_target>( &self, phase: Option>, @@ -365,8 +367,8 @@ where /// Reads a scale-bearing tick's boundary-drift fraction, when the boundary froze a radius. /// - /// The boundary froze against the low step, so each scale-bearing tick re-asks the - /// freeze-time question of its own low-step frame: what share of reviewed mass now sits + /// The boundary froze against the low step, and each scale-bearing tick therefore re-asks + /// the freeze-time question of its own low-step frame: what share of reviewed mass now sits /// at or inside the frozen radius. fn drift_fraction( &self, @@ -388,7 +390,7 @@ where /// Runs one step's target pass, when a phase exists. /// /// The pass reads the draws in the corpus domain before assembly re-indexes the released - /// families away from it. The evidence reading rides the tick cadence and consumes the + /// families away from it. The evidence reading follows the tick cadence and consumes the /// pass's own live fit and projected zero field: the fit becomes the recorded /// objective-shape reading, and the bridge ends derive from the whole-corpus fields inside /// the reading itself. @@ -457,8 +459,8 @@ where /// Runs the phase boundary's measurement. /// /// Forwards the boundary's zero-condition frame once and freezes the Proximal radius - /// against it. The frame returns beside the freeze's evidence, so the loop's target - /// freeze reads the identical coordinates. + /// against it. The frame returns beside the freeze's evidence, and the loop's target + /// freeze therefore reads the identical coordinates. /// /// On a vacuous run - no attraction force at all - nothing exists to measure or compose. /// The evidence records the fact and the relation term stays absent. No frame returns diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/training.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/training.rs index ef5053c5a0d..4914f1f94e8 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/session/training.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/session/training.rs @@ -61,8 +61,9 @@ pub(crate) fn scheduler(schedule: TrainingSchedule) -> CosineAnnealingLrSchedule /// The terminal state of one run segment. /// /// A segment either completes with the advanced training state or ends at the target refusal. -/// The record measured before the refusal rides the refusal as a value, so no unwinding call -/// can drop it. +/// The record measured before the refusal accompanies the refusal as a returned value rather +/// than as an error, and the return path therefore carries it to the caller intact. A panic in a +/// later caller drops it like any other value. // No Debug: the optimizer adaptor inside `Training` does not implement it. #[expect( clippy::large_enum_variant, @@ -72,8 +73,10 @@ pub(crate) fn scheduler(schedule: TrainingSchedule) -> CosineAnnealingLrSchedule pub(crate) enum RunOutcome> { /// The segment completed and the training state advanced. Completed(Training), - /// The target objective refused, so the run publishes no activation candidate and - /// everything measured before the refusal rides it. + /// The target objective refused. + /// + /// The run publishes no activation candidate, and everything measured before the refusal + /// accompanies it. Refused(TargetRefusal), } diff --git a/libs/@local/graph/atlas/src/salt/projector/train/fit/tests.rs b/libs/@local/graph/atlas/src/salt/projector/train/fit/tests.rs index 049a59139c9..c3c66fdfc80 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/fit/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/fit/tests.rs @@ -3,8 +3,8 @@ //! End-to-end convergence, the phase boundary's radius policy, the lens schedule, and the refresh //! telemetry. //! -//! The corpus is two four-node semantic clusters whose representations share a cluster pattern, so -//! the model can learn the separation the semantic edges describe. Landmarks on one row per +//! The corpus is two four-node semantic clusters whose representations share a cluster pattern, +//! and the model can learn the separation the semantic edges describe. Landmarks on one row per //! cluster keep the frame from collapsing or drifting. #![expect( @@ -66,6 +66,7 @@ use crate::{ }, }; +/// The CPU device every training fixture in this file runs on, resolved once. static DEVICE: LazyLock = LazyLock::new(|| Device::Cpu.pin(0).resolve()); impl FitOutcome { @@ -98,6 +99,9 @@ fn proximal_corpus(verdicts: Vec) -> Corpus { ) } +/// A small projector architecture that trains in test time. +/// +/// Width 8, one residual block, four role dimensions, one condition dimension. fn architecture() -> Architecture { Architecture { width: nz!(8), @@ -108,6 +112,7 @@ fn architecture() -> Architecture { } } +/// A fresh training projector of the fixture architecture initialized from seed 7. fn model() -> Projector { Projector::new(architecture(), &*DEVICE, rng(7)) } @@ -148,10 +153,15 @@ where total / count } -/// Probed at N=25 and N=300, seeds 11 and 23, with the landmark coefficient zeroed (the corpus's -/// only nonzero pinning force here - `anchor` is already zero in `options()` and this fixture -/// supplies no anchors): separation survives at every point (within ~0.0003-0.012, between ~29-96, -/// both seeds), so the semantic gradient itself drives the separation this test names and measures. +/// Keeps the semantic clusters separated at every probe point with the landmark force zeroed. +/// +/// Probed at 25 and 300 steps, seeds 11 and 23, with the landmark coefficient zeroed (the corpus's +/// only nonzero pinning force here: `anchor` is already zero in `options()` and this fixture +/// supplies no anchors), separation survives at every point (within ~0.0003-0.012, between ~29-96, +/// both seeds). The separation this test measures therefore does not depend on the landmark pinning +/// force. Both repulsion terms stay active in that probe. Isolating the semantic attraction +/// gradient alone is the work of the few-step certificate +/// `few_steps_semantic_gradient_pulls_cluster_mates_together`, which zeros every other coefficient. #[test] fn training_separates_the_semantic_clusters() { let corpus = semantic_corpus(); @@ -188,6 +198,10 @@ fn training_separates_the_semantic_clusters() { ); } +/// Keeps every anchored row within its anchor radius under a dominant landmark coefficient. +/// +/// Under a dominant landmark coefficient, every anchored row of the trained frame lies within its +/// anchor radius of its target at the zero step. #[test] fn landmark_support_keeps_the_frame() { let corpus = semantic_corpus(); @@ -223,12 +237,14 @@ fn landmark_support_keeps_the_frame() { } } +/// Pulls cluster mates closer within ten steps of semantic gradient alone. +/// /// A ten-step run certifies the mechanism `training_separates_the_semantic_clusters` measures at /// convergence: the semantic gradient already pulls cluster mates closer well short of the full /// schedule that compounds the effect into a separated layout. (One optimizer step is not enough /// for a stable direction: Adam's first update is close to the coordinatewise sign of the gradient, -/// so a single step can move an individual row either way even though the mean cluster displacement -/// already improves. Across ten steps the true gradient direction dominates.) +/// and a single step can therefore move an individual row either way even though the mean cluster +/// displacement already improves. Across ten steps the true gradient direction dominates.) #[test] fn few_steps_semantic_gradient_pulls_cluster_mates_together() { let corpus = semantic_corpus(); @@ -273,21 +289,23 @@ fn few_steps_semantic_gradient_pulls_cluster_mates_together() { ); } +/// Points each anchored row toward its target within ten steps under a dominant landmark force. +/// /// A ten-step run under a dominant landmark coefficient certifies the mechanism /// `landmark_support_keeps_the_frame` measures at convergence: the force already points each /// anchored row toward its target well short of the full run. (One optimizer step is not enough: /// Adam's first update is close to the coordinatewise sign of the gradient rather than its true -/// direction, so a single step can send an anchored row away from its target even under a dominant -/// coefficient. Across ten steps the true gradient direction dominates.) +/// direction, and a single step can therefore send an anchored row away from its target even under +/// a dominant coefficient. Across ten steps the true gradient direction dominates.) #[test] fn few_steps_landmark_force_points_anchors_at_their_targets() { let corpus = semantic_corpus(); let before = project(&model(), &corpus, non_negative!(0.0)); let mut options = options(schedule(nz!(10), 10, nz!(2))); // A dominant landmark coefficient pins the anchored rows. Neither repulsion term aims at a - // fixed target point, so zeroing them keeps their sampling noise from swinging the one row this + // fixed target point: zeroing them keeps their sampling noise from swinging the one row this // certificate reads. The relation term has no evidence in this corpus and goes to zero with - // them. The semantic term's type admits no zero, so the 8:1 landmark dominance carries the + // them. The semantic term's type admits no zero: the 8:1 landmark dominance carries the // isolation. options.coefficients = Coefficients::new( Positive::ONE, @@ -363,6 +381,12 @@ fn equal_seeds_train_equal_frames() { ); } +/// Freezes a positive measured radius for a proximal corpus with one review. +/// +/// A proximal corpus with one review freezes a positive measured radius at the boundary step with a +/// one-type calibration and evaluated stability certificate. Phase A and the ladder's zero step +/// exert no relation force while positive steps do. Telemetry ticks fall at `[0, 4, 6, 8]`, and +/// drift fractions read at `[6, 8]` as mass shares. #[test] fn boundary_freezes_a_measured_radius_and_opens_the_ladder() { let corpus = proximal_corpus(vec![proximal_verdict()]); @@ -396,9 +420,8 @@ fn boundary_freezes_a_measured_radius_and_opens_the_ladder() { assert!(entry.mass > d_non_negative!(0.0)); assert!(entry.quantiles.is_some()); - // Phase A is semantic-only; the ladder's zero step stays so; and - // every positive step exerts relation force (the Proximal energy - // is strictly positive). + // Phase A is semantic-only, the ladder's zero step stays so, and every positive step exerts + // relation force (the Proximal energy is strictly positive). let losses = &fitted.evidence.losses; assert!(losses[..6].iter().all(|loss| loss.relation == 0.0)); assert_eq!(losses[6].relation, 0.0, "the ladder opens at the zero step"); @@ -417,7 +440,7 @@ fn boundary_freezes_a_measured_radius_and_opens_the_ladder() { .collect(); assert_eq!(ticks, [0, 4, 6, 8]); - // The certificate rides the calibration as an evaluation of the same freeze population. + // The certificate accompanies the calibration as an evaluation of the same freeze population. let certificate = boundary .calibration .stability() @@ -432,8 +455,8 @@ fn boundary_freezes_a_measured_radius_and_opens_the_ladder() { ); // The drift report reads at every scale-bearing tick: the boundary tick and the ones - // after it. The first entry is the freeze-time reading of the freeze frame itself, so it - // sits at or above the radius fraction, and every reading is a mass share. + // after it. The first entry is the freeze-time reading of the freeze frame itself and + // therefore lies at or above the radius fraction, and every reading is a mass share. let fraction_steps: Vec = fitted .evidence .fractions @@ -451,6 +474,10 @@ fn boundary_freezes_a_measured_radius_and_opens_the_ladder() { assert!(fitted.evidence.fractions[0].fraction >= d_non_negative!(0.25)); } +/// A proximal corpus without reviews fails with `MissingProximalReviews`. +/// +/// A proximal corpus without reviews fails with `MissingProximalReviews`, whose message tells the +/// operator to confirm Proximal types. #[test] fn missing_reviewed_radius_names_the_fix() { let corpus = proximal_corpus(Vec::new()); @@ -508,12 +535,16 @@ fn forceless_corpus_trains_vacuously() { ); } +/// Projects bit-identical maps and a zero displacement field for a forceless corpus. +/// +/// A forceless corpus projects bit-identical maps at lens steps 0, 0.5 and 1 and measures a zero +/// displacement field at every tick, because the condition weights never receive gradient. #[test] fn vacuous_run_trains_a_flat_ladder() { let corpus = semantic_corpus(); - // The ladder opens at step 3, but a forceless corpus pins the - // zero step through it: the condition weights never receive - // gradient, so every step projects the identical map. + // The ladder opens at step 3, but a forceless corpus pins the zero step through it: the + // condition weights never receive gradient, and every step therefore projects the identical + // map. let options = options(schedule(nz!(9), 3, nz!(4))); let fitted = fit( model(), @@ -563,6 +594,10 @@ fn measured_radius_requires_an_opening_segment() { assert_eq!(error, TrainError::UnbaselinedRadius); } +/// A purely coincident relation policy fails with `CoincidentWithoutProximal`. +/// +/// A corpus whose only relation policy is purely coincident fails with `CoincidentWithoutProximal`, +/// since coincident force alone cannot set the proximal radius. #[test] fn coincident_without_proximal_force_refuses() { let coincident_policy = RelationPolicy { @@ -597,13 +632,16 @@ fn coincident_without_proximal_force_refuses() { assert_eq!(error, TrainError::CoincidentWithoutProximal); } +/// Measures exactly zero maximum displacement at every opening-segment tick. +/// +/// Through the opening segment every telemetry tick covers all rows and measures a maximum +/// displacement of exactly zero, because the zero-initialized condition weight never moves. #[test] fn phase_a_ticks_measure_a_frozen_lens() { - // Through the opening segment the FiLM condition weight is - // zero-initialized and receives an exactly-zero gradient at the - // zero step, so Adam never moves it and the two lens extremes - // produce bit-identical frames: the measured displacement is - // exactly zero, not approximately. + // Through the opening segment the FiLM condition weight is zero-initialized and receives an + // exactly-zero gradient at the zero step. Adam therefore never moves it, and the two lens + // extremes produce bit-identical frames: the measured displacement is exactly zero, not + // approximately. let corpus = semantic_corpus(); let options = options(schedule(nz!(4), 4, nz!(2))); let fitted = fit( @@ -629,6 +667,10 @@ fn phase_a_ticks_measure_a_frozen_lens() { } } +/// An infinite component fails the fit with `RefreshError::Diverged` naming its row. +/// +/// An infinite component in corpus row 3 fails the fit with `RefreshError::Diverged` naming that +/// row. #[test] fn non_finite_representation_fails_the_first_tick() { let mut corpus = semantic_corpus(); @@ -650,6 +692,10 @@ fn non_finite_representation_fails_the_first_tick() { assert_eq!(row.as_u64(), 3, "the error names the diverged corpus row"); } +/// Equates a chunked forward pass with the whole-corpus pass. +/// +/// A forward pass in chunks of three rows equals the whole-corpus pass, since rows project +/// independently. #[test] fn chunked_forwards_match_the_whole_corpus_pass() { let corpus = semantic_corpus(); @@ -677,6 +723,11 @@ fn chunked_forwards_match_the_whole_corpus_pass() { ); } +/// Admits and rejects `TrainingSchedule::new` inputs at the boundary and rate bounds. +/// +/// `TrainingSchedule::new` accepts a boundary inside the run and a minimum rate at or below the +/// initial rate (zero included) and rejects a boundary beyond the run or a minimum above the +/// initial rate. #[test] fn schedule_validates_its_domain() { // Out-of-range rates are unconstructible: the initial rate as a `PositiveUnitFraction`, the @@ -861,6 +912,7 @@ fn forked_ladders_share_the_frozen_radius() { ); } +/// Resuming with a different step count fails with `ScheduleChanged` carrying both schedules. #[test] fn resumed_ladder_rejects_a_changed_schedule() { let corpus = proximal_corpus(vec![proximal_verdict()]); @@ -911,6 +963,7 @@ impl RecordingProgress { } } + /// The `(step, steps, loss)` observations recorded so far, in report order. fn steps(&self) -> Vec<(usize, usize, LossBreakdown)> { self.steps .lock() @@ -918,6 +971,7 @@ impl RecordingProgress { .clone() } + /// The snapshots recorded so far, each as its sampled coordinates and landmark count. fn snapshots(&self) -> Vec<(Vec, usize)> { self.snapshots .lock() @@ -927,7 +981,6 @@ impl RecordingProgress { } impl Progress for RecordingProgress { - /// The fixture watches training steps, so nothing crosses into owning machinery. type Detached = NoProgress; fn detach(&self) -> NoProgress { @@ -953,6 +1006,10 @@ impl Progress for RecordingProgress { } } +/// Reports steps `0..12` against the whole schedule with the evidence records' losses. +/// +/// The progress stream reports steps `0..12`, each against the whole twelve-step schedule, with the +/// same loss values in the same order as the evidence records. #[test] fn every_training_step_reports_the_loss_it_records() { let corpus = proximal_corpus(vec![proximal_verdict()]); @@ -988,6 +1045,10 @@ fn every_training_step_reports_the_loss_it_records() { ); } +/// Reports one concatenated `0..12` step stream across the opening segment and the resumed ladder. +/// +/// The opening segment and the resumed ladder report one concatenated `0..12` step stream against +/// the whole schedule rather than restarting the count at the boundary. #[test] fn phases_report_against_the_whole_schedule() { let corpus = proximal_corpus(vec![proximal_verdict()]); @@ -1028,6 +1089,11 @@ fn phases_report_against_the_whole_schedule() { assert!(reported.iter().all(|&(_, steps, _)| steps == 12)); } +/// Delivers one four-row snapshot per tick, landmarks first, without changing the losses. +/// +/// An observer with appetite four receives one four-row snapshot per telemetry tick with the two +/// landmark rows first, the first and last snapshots differ, and the watched run's losses are +/// bit-equal to an unwatched run's. #[test] fn a_watching_observer_sees_the_placement_move_and_changes_nothing() { let corpus = proximal_corpus(vec![proximal_verdict()]); @@ -1055,7 +1121,7 @@ fn a_watching_observer_sees_the_placement_move_and_changes_nothing() { .expect("the boundary fixture trains") .trained(); - // Each snapshot comes from the refresh's own frame, so exactly one exists per tick and the + // Each snapshot comes from the refresh's own frame: exactly one exists per tick, and the // telemetry counts the same ticks. Each snapshot reports the sample the observer requested, // with the corpus's two landmark rows first. let snapshots = observer.snapshots(); @@ -1067,8 +1133,9 @@ fn a_watching_observer_sees_the_placement_move_and_changes_nothing() { .all(|&(ref positions, landmarks)| positions.len() == 4 && landmarks == 2) ); - // The sample consumes no randomness, so watching cannot move the - // run: the two runs' losses are bit-equal, step for step. + // The sample consumes no randomness and this observer's callbacks only record, which leaves + // the two runs' draws the same. On the test backend their losses come out bit-equal, step for + // step. assert_eq!(watched.evidence.losses, unwatched.evidence.losses); // The placement moves under the observer's eye rather than @@ -1080,6 +1147,7 @@ fn a_watching_observer_sees_the_placement_move_and_changes_nothing() { /// Runs a closure under a warn-level subscriber and returns everything it logged. fn captured_warnings(run: impl FnOnce()) -> String { + /// A shared byte buffer the subscriber writes its formatted events into. #[derive(Clone, Default)] struct Capture(alloc::sync::Arc>>); @@ -1121,7 +1189,7 @@ fn captured_warnings(run: impl FnOnce()) -> String { String::from_utf8(bytes).expect("formatted log output is UTF-8") } -/// A certificate literal whose decision is the given `pass`, dyadic throughout. +/// Builds a stability certificate with decision `pass`. fn certificate(pass: bool) -> StabilityCertificate { StabilityCertificate { quantile: open_unit_fraction!(0.25), @@ -1158,6 +1226,10 @@ fn spread_calibration(spread: f32, pass: bool) -> ProximalCalibration { ) } +/// Warns on a failing stability certificate naming `reviewed_mass_stability_bound`. +/// +/// A failing stability certificate with a tight spread warns naming `reviewed_mass_stability_bound` +/// and reports `leave_one_out_spread` beside it, without the spread warning. #[test] fn a_failing_certificate_warns_with_its_check_name() { let calibration = spread_calibration(0.25, false); @@ -1169,12 +1241,16 @@ fn a_failing_certificate_warns_with_its_check_name() { output.contains("fails its evaluated stability bound"), "{output}" ); - // The leave-one-type-out spread rides beside the warning. + // The leave-one-type-out spread is reported beside the warning. assert!(output.contains("leave_one_out_spread"), "{output}"); // The tight spread crossed no spread warning. assert!(!output.contains("leave_one_out_radius_spread"), "{output}"); } +/// Warns on a wide leave-one-out spread naming `leave_one_out_radius_spread` alone. +/// +/// A passing certificate whose leave-one-out spread exceeds one transition width warns naming +/// `leave_one_out_radius_spread` and not the stability bound. #[test] fn a_spread_beyond_one_temperature_warns_with_its_check_name() { let calibration = spread_calibration(4.0, true); @@ -1192,6 +1268,7 @@ fn a_spread_beyond_one_temperature_warns_with_its_check_name() { ); } +/// A passing certificate with a tight spread produces no warning output. #[test] fn a_passing_certificate_with_a_tight_spread_warns_nothing() { let calibration = spread_calibration(0.25, true); @@ -1201,10 +1278,11 @@ fn a_passing_certificate_with_a_tight_spread_warns_nothing() { assert_eq!(output, ""); } -/// The field-derived constants the identity declares re-derive from the recorded boundary -/// field, and the enforcement record covers exactly the ladder's interval. The -/// neighbour-dependent constants ride the ruler tables, whose re-freeze carries its own -/// certificate below. +/// Re-derives the identity's field-derived constants from the recorded boundary field. +/// +/// The field-derived constants the identity declares re-derive from the recorded boundary field, +/// and the enforcement record covers exactly the ladder's interval. The neighbour-dependent +/// constants are carried by the ruler tables, whose re-freeze has its own certificate below. #[test] fn the_identity_constants_re_derive_from_the_recorded_boundary_field() { let corpus = target_corpus(); @@ -1236,7 +1314,7 @@ fn the_identity_constants_re_derive_from_the_recorded_boundary_field() { assert_eq!(field.len(), ROWS); // Every field-derived constant re-derives from the recorded field, bit for bit. The - // recorded field is finite by construction, so the reading needs no scan. + // recorded field is finite by construction: the reading needs no scan. let spread = target.boundary_field.rms_spread(); #[expect( clippy::cast_possible_truncation, @@ -1280,8 +1358,8 @@ fn the_identity_constants_re_derive_from_the_recorded_boundary_field() { assert_eq!(target.enforcement.last_application, Some(12)); assert_eq!(target.row_maxima.len(), ROWS); - // At the boundary step the zero field is the snapshot itself, so the common-mode fit - // reads the exact identity scale and neither displacement nor saturation reads anything. + // At the boundary step the zero field is the snapshot itself: the common-mode fit reads the + // exact identity scale, and neither displacement nor saturation reads anything. let boundary = &target.evaluations[0]; assert_eq!(boundary.zero_similarity.scale().get(), 1.0); let mut anchor_rows = 0; @@ -1310,18 +1388,19 @@ fn the_identity_constants_re_derive_from_the_recorded_boundary_field() { assert_eq!(fitted.evidence.losses[6].target, target.estimands[0]); } -/// A live gauge fit refusal ends the run as the refused outcome: no activation candidate, no -/// target claim, and the whole run record - boundary and target records sealed in - rides the -/// refusal, cut at the last completed reading before the failed one. +/// Ends the run as the refused outcome on a live gauge fit refusal, run record attached. +/// +/// A live gauge fit refusal ends the run as the refused outcome. The run publishes no activation +/// candidate and makes no target claim, and the whole run record (boundary and target records +/// sealed in) accompanies the refusal, cut at the last completed reading before the failed one. #[test] fn gauge_fit_refusal_keeps_the_recorded_evidence() { let corpus = target_corpus(); let draws = target_draws(); let options = options(schedule(nz!(12), 6, nz!(4))); let mut declared = target_options(1.0); - // At the boundary step no relation gradient has flowed, the steps read bit-identical - // coordinates, and the residual is exactly zero, so a bar below any real deformation - // refuses the first fit after a post-boundary update. + // the 10⁻¹² limit accepts the zero-residual boundary fit and rejects the deformation from the + // first post-boundary update. declared.residual_bar = Some(positive!(1e-12)); let outcome = fit( @@ -1343,7 +1422,7 @@ fn gauge_fit_refusal_keeps_the_recorded_evidence() { TargetRefusalCause::Gauge(GaugeRefusal::ResidualAboveBar { .. }) ); - // The run record rides the refusal whole. The boundary record entered it when the radius + // The refusal carries the run record whole. The boundary record entered it when the radius // freeze completed at step 6, the run's standing self-measurement is cut at the last // completed reading (losses through step 6, ticks at steps 0, 4, and 6 - the refusing // step ran none), and the refusing step's loss never computed. @@ -1379,8 +1458,8 @@ fn gauge_fit_refusal_keeps_the_recorded_evidence() { /// A refusal at a post-boundary tick step cuts between the tick's two readings. /// -/// The refresh telemetry precedes the target pass, so the refusing step's tick enters the -/// record, while the per-evaluation evidence follows the refusing fit and never records. The +/// The refresh telemetry precedes the target pass: the refusing step's tick enters the record, +/// while the per-evaluation evidence follows the refusing fit and never records. The /// schedule places the first fit after an optimizer update on a tick step - boundary 7, /// interval 4, step 8 - and that one step therefore witnesses both sides of the cut. #[test] @@ -1413,7 +1492,7 @@ fn tick_step_refusal_records_telemetry_and_no_evaluation() { ); // The refusing step is a tick step (8 % 4 == 0), and its telemetry entered the record - // before the target pass refused. The loop records a step's loss after the pass, so the + // before the target pass refused. The loop records a step's loss after the pass: the // refusing step's loss never pushed. let record = refusal.evidence; let ticks: Vec = record.telemetry.iter().map(|tick| tick.step).collect(); @@ -1434,17 +1513,18 @@ fn tick_step_refusal_records_telemetry_and_no_evaluation() { } /// A boundary freeze refusal ends the run as the refused outcome before any target step runs. -/// The boundary record enters the run record the moment the radius freeze completes, before -/// the target freeze is attempted. The refusal therefore carries it beside the opening -/// segment's accumulated readings, and no target record exists because the phase never froze. +/// +/// The boundary record enters the run record the moment the radius freeze completes, before the +/// target freeze is attempted. The refusal therefore carries it beside the opening segment's +/// accumulated readings, and no target record exists because the phase never froze. #[test] fn boundary_freeze_refusal_carries_the_boundary_evidence() { let corpus = target_corpus(); let draws = target_draws(); let options = options(schedule(nz!(12), 6, nz!(4))); let mut declared = target_options(1.0); - // A spread floor no healthy constellation reaches, so the gauge freeze refuses at the - // boundary - a data-dependent refusal admission cannot see. + // A spread floor no healthy constellation reaches: the gauge freeze refuses at the + // boundary, a data-dependent refusal admission cannot see. declared.gauge_spread_factor = Some(positive!(1e6)); let outcome = fit( @@ -1473,8 +1553,7 @@ fn boundary_freeze_refusal_carries_the_boundary_evidence() { .expect("the radius freeze completed before the refusal"); assert_eq!(boundary.step, 6); assert_matches!(boundary.radius, FrozenRadius::Measured { .. }); - // The phase never froze, so no target record exists - absent structurally, never - // dropped. + // the failed freeze prevents creation of the target record. assert!(record.target.is_none()); // The opening segment's record as accumulated. Losses reach through step 5 and ticks // ran at steps 0 and 4 (the boundary step's own tick follows the boundary and never @@ -1484,10 +1563,12 @@ fn boundary_freeze_refusal_carries_the_boundary_evidence() { assert!(record.fractions.is_empty()); } -/// The frozen ruler re-freezes bit-identically from the recorded boundary field, and the -/// recorded tables equal the re-freeze's own - the reading no field alone determines, since -/// one `Z_K` admits many neighbour tables. The freeze is bit-deterministic from its inputs, -/// so the recorded trio suffices to reconstruct the exact ruler every reading divided by. +/// Re-freezes the ruler bit-identically from the recorded boundary field and tables. +/// +/// The frozen ruler re-freezes bit-identically from the recorded boundary field, and the recorded +/// tables equal the re-freeze's own - the reading no field alone determines, since one `Z_K` admits +/// many neighbour tables. The freeze is bit-deterministic from its inputs, and the recorded trio +/// therefore suffices to reconstruct the exact ruler every reading divided by. #[test] fn the_ruler_re_freezes_from_the_recorded_boundary_field() { let corpus = target_corpus(); @@ -1547,10 +1628,11 @@ fn the_ruler_re_freezes_from_the_recorded_boundary_field() { ); } -/// The activation is a value, not structure. A zero-activation run reads a live estimand -/// stream while contributing exactly zero force: the recorded per-step target loss - the term -/// the composite loss descends - is 0.0 at every step under either penalty, and only a live -/// activation descends it. +/// The activation is a value, not structure. +/// +/// A zero-activation run reads a live estimand stream while contributing exactly zero force: the +/// recorded per-step target loss - the term the composite loss descends - is 0.0 at every step +/// under either penalty, and only a live activation descends it. #[test] fn zero_activation_zero_force() { let corpus = target_corpus(); @@ -1742,10 +1824,12 @@ fn every_split_population_overlap_refuses_at_admission() { ); } -/// A reopened resume checkpoint carries the written parameters, the scheduler at the boundary, -/// the identical schedule, and the caller's generator stream. Identity is asserted on the -/// decoded state, per the writer's own contract - the bytes are not canonical (the optimizer -/// record is a map), the decoded state is. +/// Reopens a resume checkpoint with parameters, scheduler, schedule and generator intact. +/// +/// A reopened resume checkpoint carries the written parameters, the scheduler at the boundary, the +/// identical schedule, and the caller's generator stream. Identity is asserted on the decoded +/// state, per the writer's own contract - the bytes are not canonical (the optimizer record is a +/// map), the decoded state is. #[test] fn resume_checkpoint_round_trip() { let corpus = target_corpus(); @@ -1794,8 +1878,8 @@ fn resume_checkpoint_round_trip() { state.training.scheduler.to_record::() ); - // The model record is a named-struct tree, so its serialization is deterministic - the - // writer's non-canonical clause covers the optimizer map alone. + // The model record is a named-struct tree and its serialization is therefore deterministic: + // the writer's non-canonical clause covers the optimizer map alone. let recorder = NamedMpkBytesRecorder::::new(); let written = recorder .record(state.training.model.into_record(), ()) @@ -1809,11 +1893,11 @@ fn resume_checkpoint_round_trip() { ); } -/// A structurally valid resume record around the given overrides. +/// A structurally valid resume record, which the tests override one field at a time. fn resume_record() -> ResumeRecord { ResumeRecord { model: model().into_record(), - // A fresh optimizer carries no moments until its first step, so the empty map is the + // A fresh optimizer carries no moments until its first step: the empty map is the // boundary state of a zero-length opening segment. optimizer: <_>::default(), scheduler: 5, @@ -1826,12 +1910,19 @@ fn resume_record() -> ResumeRecord { } } +/// Encodes a resume record with the checkpoint's recorder. +/// +/// The recorder is the full-precision named-`MessagePack` one the checkpoint uses. fn record_bytes(record: ResumeRecord) -> Vec { NamedMpkBytesRecorder::::new() .record(record, ()) .expect("the test record encodes") } +/// Restores the generator state and the schedule exactly from an encoded resume record. +/// +/// Opening an encoded resume record restores the generator state exactly and the schedule's steps, +/// boundary and refresh interval. #[test] fn open_checkpoint_schedule_round_trip() { let bytes = record_bytes(resume_record()); @@ -1852,6 +1943,10 @@ fn open_checkpoint_schedule_round_trip() { assert_eq!(state.schedule.refresh_interval().get(), 4); } +/// A record whose minimum rate exceeds the initial rate fails with `InvalidSchedule`. +/// +/// A record whose minimum learning rate exceeds the initial rate fails to open with +/// `CheckpointError::InvalidSchedule`. #[test] fn open_checkpoint_invalid_schedule() { let mut record = resume_record(); @@ -1871,6 +1966,10 @@ fn open_checkpoint_invalid_schedule() { ); } +/// A record whose scheduler position is off the boundary fails with `SchedulerPosition`. +/// +/// A record whose scheduler position is not the boundary fails to open with +/// `CheckpointError::SchedulerPosition` naming the position and the boundary. #[test] fn open_checkpoint_scheduler_off_boundary() { let mut record = resume_record(); diff --git a/libs/@local/graph/atlas/src/salt/projector/train/metrics.rs b/libs/@local/graph/atlas/src/salt/projector/train/metrics.rs index be96dec770a..cf941e9af78 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/metrics.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/metrics.rs @@ -32,7 +32,7 @@ use crate::{ /// A degree decile, `D1` (lowest participating degrees) through `D10` (highest). /// /// `Option` is one row's participation state: rows without attraction evidence have no -/// decile at all, so no sentinel value exists to misread as a bucket. +/// decile at all, and no sentinel value exists to misread as a bucket. #[derive(Debug, Copy, Clone, PartialEq, Eq)] #[repr(u8)] #[expect( @@ -59,11 +59,13 @@ impl Decile { reason = "the index ranges over the ten variants" )] const ALL: [Self; Self::COUNT] = core::array::from_fn(const |index| { - // SAFETY: `index` ranges over `0..COUNT`, so `index + 1` ranges over `1..=COUNT` - - // exactly the enum's `repr(u8)` discriminants `D1 = 1` through `D10 = COUNT`, which - // `variant_count` ties to the variant list itself. + // SAFETY: `Decile` is `repr(u8)` with discriminants `D1 = 1` through `D10 = COUNT`, and + // `variant_count` ties `COUNT` to the variant list itself. `index` ranges over + // `0..COUNT`, and `index + 1` therefore ranges over `1..=COUNT`, exactly those + // discriminants. unsafe { core::mem::transmute::(index as u8 + 1) } }); + /// The decile count, tied to the variant list. const COUNT: usize = mem::variant_count::(); /// Returns the 0-based bucket position, [`D1`](Self::D1) at zero. @@ -92,7 +94,7 @@ where /// # Panics /// /// This panics when an attraction edge references a row at or beyond `rows`. The index and the - /// row domain come from one generation, so a mismatch is a wiring defect. + /// row domain come from one generation, and a mismatch is therefore a wiring defect. #[must_use] pub(crate) fn new(index: &AttractionIndex, rows: usize) -> Self where @@ -231,7 +233,7 @@ impl BudgetBreakdown { /// /// A row participates in a type when any attraction instance of that type touches it, and repeated /// instances count once. Construction happens once per training run and every telemetry tick reuses -/// the result, so the per-tick cost is the participant lists rather than the edge lists. +/// the result, and the per-tick cost is the participant lists rather than the edge lists. #[derive(Debug)] pub(crate) struct TypeParticipants { types: Vec<(OntologyRowId, Box<[N]>)>, @@ -292,8 +294,8 @@ impl DisplacementMoments { pub(crate) const fn record(&mut self, displacement: NonNegative) { self.count += 1; self.sum += DNonNegative::from(displacement); - // The widened square is exact, so adding it matches the fused form bit for bit, and a - // sum of at most 2⁶⁴ squares of `f32`-born values stays far inside the `f64` range. + // The widened square is exact: adding it matches the fused form bit for bit. A sum of + // at most 2⁶⁴ squares of `f32`-born values stays far inside the `f64` range. self.sum_squares += displacement.square_wide(); self.maximum = self.maximum.max(displacement); } @@ -365,9 +367,11 @@ pub(crate) const EXPONENT_BUCKETS: usize = 256; /// A displacement histogram over the `f32` exponent grid. /// /// Bucket `b` counts displacements whose biased exponent is `b`: bucket 0 holds exact zeros and -/// subnormals, and bucket `b` for `1 ≤ b ≤ 254` holds values in `[2^(b - 127), 2^(b - 126))`. The -/// format's own grid needs no configured edges and resolves nine decades to within a factor of two, -/// which is the resolution the telemetry questions ask at. +/// subnormals, bucket `b` for `1 ≤ b ≤ 254` holds values in `[2^(b - 127), 2^(b - 126))`, and +/// bucket 255 holds `+∞`. The format's own grid needs no configured edges and resolves every +/// normal magnitude to within a factor of two, which is the resolution the telemetry questions ask +/// at. Bucket 0 spans zero and the whole subnormal range, whose positive values differ by up to +/// a factor of `2²³ − 1`. #[derive(Debug, Clone, PartialEq)] pub(crate) struct DisplacementHistogram { counts: [u64; EXPONENT_BUCKETS], @@ -427,7 +431,7 @@ impl Default for DisplacementHistogram { /// One refresh tick's displacement field, per reporting bucket. /// -/// The overall and per-decile buckets carry full histograms; the per-type buckets carry summary +/// The overall and per-decile buckets carry full histograms. The per-type buckets carry summary /// moments only, because a corpus has thousands of relation types and the per-type question - is a /// type moving nodes it has little evidence for - reads from location and spread, not shape. Rows /// without attraction evidence enter the overall bucket only, because the lens can move them @@ -447,8 +451,8 @@ impl DisplacementSummary { /// # Panics /// /// This panics when the frames disagree in length or a participant row lies outside them. The - /// frames, the participants, and the deciles all describe one corpus, so a mismatch is a wiring - /// defect. + /// frames, the participants, and the deciles all describe one corpus, and a mismatch is + /// therefore a wiring defect. #[must_use] pub(crate) fn measure( low: &FinitePointField, diff --git a/libs/@local/graph/atlas/src/salt/projector/train/mod.rs b/libs/@local/graph/atlas/src/salt/projector/train/mod.rs index a9696ed4f1a..f963120caba 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/mod.rs @@ -1,7 +1,7 @@ //! Training-step machinery for the conditioned projector. //! -//! One training step draws a minibatch over the built artifacts and projects its rows at the step's -//! relation-lens step. The step then evaluates the composite objective against the detached +//! One training step draws a minibatch over the built artifacts and projects its rows at the +//! training step's lens step. The step then evaluates the composite objective against the detached //! coordinates and measures the relation forces per node for the budget diagnostics. Its return //! value is one backward-ready scalar whose gradient carries exactly the combined per-node field //! through the shared model parameters. @@ -20,14 +20,14 @@ //! //! - Semantic attraction scales by `W / m` (total positive edge weight over drawn pairs), the //! unbiased estimator of the full weighted attraction. -//! - Ordinary repulsion scales by `W / m` as well, so the ordinary coefficient over the semantic -//! one reads directly as the repulsion-to-attraction balance. +//! - Ordinary repulsion scales by `W / m` as well: the ordinary coefficient over the semantic one +//! reads directly as the repulsion-to-attraction balance. //! - Hard-negative repulsion scales by `N / m` (corpus rows over drawn query rows), the unbiased //! estimator of the pooled mined-frame total. //! - Relation attraction scales by `G / g` (total relation groups over drawn groups), the unbiased //! estimator of the capped relation objective. That objective is the specified per-type clipped //! total over the same force-mass population the boundary calibration measures its radius over. -//! Changing the per-type factor re-derives both surfaces together, so they move in lockstep by +//! Changing the per-type factor re-derives both surfaces together: they move in lockstep by //! contract. //! - Support terms scale by their pool size over the drawn count. //! - The target objective divides each drawn unit by its full first-order inclusion probability diff --git a/libs/@local/graph/atlas/src/salt/projector/train/refresh.rs b/libs/@local/graph/atlas/src/salt/projector/train/refresh.rs index da7a00a8e17..847a0eae4e2 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/refresh.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/refresh.rs @@ -86,7 +86,7 @@ impl Error for RefreshError where N: fmt::Debug + fmt::Display {} pub(crate) struct RefreshOutcome { /// The low step's forwarded frame. /// - /// The tick's own artifacts consume it in place. It rides out for the boundary-drift + /// The tick's own artifacts consume it in place. It is returned for the boundary-drift /// report, which re-measures the reviewed mass fraction over the same step the radius /// froze on. pub frame: Box>, @@ -103,15 +103,17 @@ pub(crate) struct RefreshOutcome { /// The corpus rows a run reports into [`Progress::projector_snapshot`]. /// /// An observer's appetite ([`Progress::projector_sample_size`]) buys a fixed set of rows, chosen -/// before the loop and reported at every tick, so a watcher sees the same points moving rather than -/// a fresh sample each time. Landmark rows come first, because they are the skeleton the placement -/// hangs on and a renderer draws them apart. They take at most half the budget, so a landmark-rich -/// corpus still shows its interior. The rest is an even stride over the corpus rows no landmark -/// holds, so the two shares partition the sample by role: every reported point past the landmark -/// prefix is an ordinary row. +/// before the loop and reported at every tick, and a watcher sees the same points moving rather +/// than a fresh sample each time. Landmark rows come first, because they are the skeleton the +/// placement hangs on and a renderer draws them apart. They take at most half the budget wherever +/// the interior has rows for the other half, and a landmark-rich corpus still shows its interior. +/// The rest is an even stride over the corpus rows no landmark holds, and the two shares partition +/// the sample by role: every reported point past the landmark prefix is an ordinary row. /// -/// The choice is deterministic by construction and consumes no randomness: an observer cannot move -/// the run's draws, so a run publishes the same placement whether or not anything watches. +/// The choice is deterministic by construction and consumes no randomness, and an observer's +/// appetite therefore cannot move the run's batch draws. Whether the published placement is +/// bit-identical with and without a watcher depends also on the observer's callbacks and on the +/// backend's execution, which the selection does not govern. #[derive(Debug, Default)] pub(super) struct SnapshotSample { /// The sampled rows, with the landmark share first and the strided share after it. @@ -131,8 +133,8 @@ where /// own admission rejects them, and a sample is not the place to discover it. pub(super) fn select(rows: usize, landmarks: &[SupportAnchor], budget: usize) -> Self { if budget == 0 || rows == 0 { - // The zero-budget path allocates nothing and sorts nothing, so every later report is a - // no-op. + // The zero-budget path allocates nothing and sorts nothing, and every later report is + // a no-op. return Self { rows: Vec::new(), landmarks: 0, @@ -158,10 +160,9 @@ where .map(|rank| anchored[rank]) .collect(); - // The corpus is not walked to find its unheld rows: the - // `rank`-th of them sits `rank` places along plus one for every - // landmark at or before it, and the ranks arrive in order, so - // one pass over the sorted landmarks resolves every pick. + // The corpus is not walked to find its unheld rows: the `rank`-th of them is `rank` places + // along plus one for every landmark at or before it, and the ranks arrive in order. + // One pass over the sorted landmarks therefore resolves every pick. let mut passed = 0; sample.extend(even_ranks(interior, interior_share).map(|rank| { while passed < anchored.len() && anchored[passed].as_usize() <= rank + passed { @@ -191,8 +192,16 @@ where /// Picks `count` of `len` positions, evenly spread across the sequence. /// /// The walk is a Bresenham accumulator. Every position adds `count` and every crossing of `len` -/// takes one, so a `count` at or below `len` picks exactly `count` positions at an even spacing. -/// The walk needs no division, which is also why the spacing is exact rather than rounded. +/// takes one. For a `count` at or below `len` whose sum `len + count` is representable in `usize`, +/// the walk picks exactly `count` positions at an even spacing: the accumulator peaks below that +/// sum, and an overflowing addition would panic under overflow checks or wrap past a crossing. The +/// walk needs no division, which is also why the spacing is exact rather than rounded. +/// +/// # Panics +/// +/// Constructing the iterator never panics. Advancing it can panic under overflow checks when the +/// accumulator's addition of `count` overflows `usize`. A representable sum `len + count` rules +/// that out. fn even_ranks(len: usize, count: usize) -> impl Iterator { let mut accumulator = 0; (0..len).filter(move |_rank| { @@ -231,10 +240,10 @@ where /// /// `with_scales` selects the post-boundary shape, where the tick forwards every step and /// measures each one into a scale table. Without it the tick forwards only the two extremes. - /// The opening semantic-only segment and the vacuous-relation run consume no scale tables, so a - /// middle-step forward is dead weight. + /// The opening semantic-only segment and the vacuous-relation run consume no scale tables, and + /// a middle-step forward is dead weight there. /// - /// The tick is where the whole corpus exists in coordinates, so it reports `sample`'s rows of + /// The tick is where the whole corpus exists in coordinates, and it reports `sample`'s rows of /// the low step's frame to `progress`. That is the same frame the miner and the displacement /// summary read, retained no longer than they retain it. /// @@ -289,7 +298,7 @@ where /// Projects the whole corpus at one step, in bounded row slices. /// /// `forward_rows` bounds each slice's row count, and with it the peak device memory of a corpus -/// forward; the frame it returns matches a single whole-corpus pass because the model maps rows +/// forward. The frame it returns matches a single whole-corpus pass because the model maps rows /// independently of each other. The match is value-level, not bit-level: slices of different row /// counts are dispatches of different shapes, and a backend that selects kernels per shape may /// move the last bit of a coordinate between them. diff --git a/libs/@local/graph/atlas/src/salt/projector/train/step.rs b/libs/@local/graph/atlas/src/salt/projector/train/step.rs index e91f8d48afb..21600876e47 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/step.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/step.rs @@ -5,12 +5,12 @@ //! [`objective`](Evaluation::objective) projects the batch rows and hands the coordinates to //! [`evaluate`](Evaluation::evaluate), which computes the composite objective in two regimes. The //! hand-gradient families (semantic attraction, ordinary and hard repulsion, relation attraction) -//! evaluate value and per-node coordinate gradient against the detached coordinate frame; the +//! evaluate value and per-node coordinate gradient against the detached coordinate frame. The //! evaluation measures the relation field per node against the semantic one for the budget //! diagnostics, and the combined field re-enters the parameter graph through the surrogate scalar, //! whose single backward pass deposits exactly that field. The support families (temporal anchors, -//! landmarks) ride ordinary autodiff on the coordinate tensor - they carry no budget diagnostics - -//! and add onto the same scalar. +//! landmarks) differentiate through ordinary autodiff on the coordinate tensor - they carry no +//! budget diagnostics - and add onto the same scalar. //! //! Relation-inactive nodes - every node when the batch carries no relation edges, and any node //! whose accumulated relation gradient is exactly zero - contribute their semantic gradient alone @@ -45,7 +45,7 @@ use crate::{ /// The step's evaluated loss values, one per objective family. /// -/// Values are the scaled batch sums the step actually descends; families absent from the batch +/// Values are the scaled batch sums the step actually descends. Families absent from the batch /// report zero. #[derive(Debug, Copy, Clone, PartialEq, Default)] pub struct LossBreakdown { @@ -101,7 +101,7 @@ pub(crate) struct Objective { /// /// The corpus input columns, the numerical contract, and the reporting decile axis. /// -/// Bound once per training run; the loop composes the frozen relation energy into `options` at the +/// Bound once per training run. The loop composes the frozen relation energy into `options` at the /// phase boundary. #[derive(Debug)] pub(crate) struct Evaluation<'run, N> { @@ -144,8 +144,8 @@ where /// /// `coordinates` are the batch rows' projections in the batch's local row order, optionally /// followed by alignment padding. Trailing rows beyond the batch's own are the materialized - /// input's padding twins (see [`ROW_ALIGNMENT`]), which no population references, so they - /// carry exactly zero force. + /// input's padding twins (see [`ROW_ALIGNMENT`]), which no population references. They carry + /// exactly zero force. /// /// The coordinate producer stays exchangeable: [`objective`](Self::objective) forwards the /// model's projection, while tests drive hand-built frames through this method directly. @@ -302,7 +302,7 @@ where let rows = frame.len(); // One scratch field serves every type. The pass reads and re-zeroes only the rows a type - // touches, so it costs the edge lists rather than types times batch rows. + // touches, and it costs the edge lists rather than types times batch rows. let mut relation_field = GradientField::new(rows); let mut scratch = GradientField::new(rows); let mut contributions: Vec<(BatchRowId, OntologyRowId, Vec2)> = Vec::new(); @@ -389,9 +389,9 @@ where /// Reads the detached coordinate frame back to the host as a proven-finite field. /// -/// The finiteness scan covers the whole readback, alignment padding included, so the returned -/// [`FinitePointField`] carries every row of the tensor in row order and downstream views need -/// no rescan. +/// The finiteness scan covers the whole readback, alignment padding included. The returned +/// [`FinitePointField`] therefore carries every row of the tensor in row order and downstream views +/// need no rescan. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/projector/train/tests.rs b/libs/@local/graph/atlas/src/salt/projector/train/tests.rs index 87ce0bc03da..841c0a22b6f 100644 --- a/libs/@local/graph/atlas/src/salt/projector/train/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/train/tests.rs @@ -2,7 +2,6 @@ //! //! Deterministic seeded draws, estimator scales, batch-local re-indexing, hand-computed objective //! fields verified through autodiff, budget-clip wiring, and the reporting buckets. - #![expect( clippy::float_cmp, reason = "dyadic fixture values compute exactly in f32 and bit-exact assertions are the \ @@ -63,20 +62,26 @@ use crate::{ }, }; +/// The CPU device every tensor fixture in this file lives on, resolved once. static DEVICE: LazyLock = LazyLock::new(|| Device::Cpu.pin(0).resolve()); +/// Builds a [`NodeRowId`] from a literal, keeping the pair fixtures short. macro_rules! node { ($id:expr) => { NodeRowId::new($id) }; } +/// Builds a [`BatchRowId`](crate::salt::projector::loss::BatchRowId) from a literal. +/// +/// The expected local-domain pairs use it. macro_rules! batch { ($id:expr) => { crate::salt::projector::loss::BatchRowId::new($id) }; } +/// Seeds a [`Xoshiro256PlusPlus`] generator from `seed`. fn rng(seed: u64) -> Xoshiro256PlusPlus { Xoshiro256PlusPlus::seed_from_u64(seed) } @@ -140,6 +145,9 @@ fn instance( } } +/// Builds the relation indexes over `rows` corpus rows from certified `policies` and instances. +/// +/// The attraction options are the defaults, whose fixture values satisfy every contract. fn relation_indexes( rows: usize, policies: &[RelationPolicy], @@ -156,8 +164,8 @@ fn relation_indexes( /// The affinity energy of the dyadic fixtures. /// -/// `a = 1, b = 1, ε = 0.5`, so at squared distance one both logarithm arguments are exactly one -/// (zero value) and the derivative mass is exactly `0.25`. +/// `a = 1, b = 1, ε = 0.5`: at squared distance one both logarithm arguments are exactly one +/// (zero value), and the derivative mass is exactly `0.25`. fn affinity() -> AffinityEnergy { AffinityEnergy::new( AffinityCurve::new(1.0, 1.0).expect("the fixture curve is valid"), @@ -168,8 +176,8 @@ fn affinity() -> AffinityEnergy { /// The relation energy of the dyadic fixtures: Proximal radius one at temperature one half. /// -/// `z = 1` therefore sits exactly on the radius with derivative `sigmoid(0) = 0.5`. The scale guard -/// is `0.5`, so unit normalization comes from local scales of `0.5`. +/// `z = 1` therefore lies exactly on the radius with derivative `sigmoid(0) = 0.5`. The scale guard +/// is `0.5`: unit normalization comes from local scales of `0.5`. fn relation_energy() -> RelationEnergy { RelationEnergy::new( CoincidentEnergy::new(non_negative!(0.25), positive!(1.0)), @@ -179,14 +187,15 @@ fn relation_energy() -> RelationEnergy { .expect("the fixture radii are ordered") } +/// Support options with a unit Huber threshold and a `0.5` radius floor. fn support_options() -> SupportOptions { SupportOptions::new(positive!(1.0), positive!(0.5)) } /// Coefficients used by the objective fixtures. /// -/// `lambda_S = 0.5` pairs with a semantic scale of two for a unit semantic factor, and `lambda_N = -/// 2` doubles the ordinary term so the two families are distinguishable in the combined field. +/// `λ_S = 0.5` pairs with a semantic scale of two for a unit semantic factor, and `λ_N = 2` +/// doubles the ordinary term, which keeps the two families distinguishable in the combined field. fn coefficients() -> Coefficients { Coefficients::new( positive!(0.5), @@ -198,6 +207,9 @@ fn coefficients() -> Coefficients { ) } +/// Assembles the objective options with the given relation energy and budget. +/// +/// The affinity, coefficients and support come from the shared fixtures. fn options(relation: Option, budget: Budget) -> ObjectiveOptions { ObjectiveOptions { affinity: affinity(), @@ -208,6 +220,7 @@ fn options(relation: Option, budget: Budget) -> ObjectiveOptions } } +/// A budget with a `0.25` floor, high enough that no fixture evaluation clips. fn fixture_budget() -> Budget { Budget { floor: positive!(0.25), @@ -266,7 +279,7 @@ fn unused_deciles() -> DegreeDeciles { /// A run context for driving [`Evaluation::evaluate`] with a hand-built frame. /// -/// The columns are empty because `evaluate` never reads them; only [`Evaluation::objective`] +/// The columns are empty because `evaluate` never reads them. Only [`Evaluation::objective`] /// projects through them. fn frame_evaluation( options: ObjectiveOptions, @@ -285,9 +298,9 @@ fn frame_evaluation( #[test] fn degree_deciles_rank_participating_rows() { // One relation with edges (0,1), (0,2), (0,3), (4,5) over seven rows. Row 0 has degree three, - // rows 1-5 degree one, and row 6 none. Participating degrees sorted: [1, 1, 1, 1, 1, 3], n - // = 6. Rank of degree 1 is 5 (entries at or below), so its decile is - // (5-1)*10/6 = 6; rank of degree 3 is 6, decile (6-1)*10/6 = 8. + // rows 1-5 degree one, and row 6 none. Participating degrees sorted: [1, 1, 1, 1, 1, 3], + // n = 6. Rank of degree 1 is 5 (entries at or below): its decile is (5 − 1)·10/6 = 6. Rank of + // degree 3 is 6, decile (6 − 1)·10/6 = 8. let indexes = relation_indexes( 7, &[proximal_policy(11)], @@ -372,8 +385,9 @@ fn draws_are_deterministic_at_a_fixed_seed() { ); } -/// The allocator choice is bit-inert: `_in` through a different allocator draws and assembles -/// identically. +/// Draws and assembles identically through a different allocator. +/// +/// The allocator choice is bit-inert for `_in`. /// /// Equal seeds through `draw`/`assemble` (global) and `draw_in`/`assemble_in` (system) produce /// equal populations and batches, family by family - the allocator parameter places storage and @@ -479,6 +493,10 @@ fn allocator_seam_draws_and_assembles_identically() { assert_eq!(assembled.eta, assembled_in.eta); } +/// Draws no relation edges at `eta = 0` and the one relation group at `eta = 1`. +/// +/// At `eta = 0` the sampler draws no relation edges and reports a zero relation scale, and at +/// `eta = 1` the same seed draws the one relation group. #[test] fn draw_skips_the_relation_family_at_a_zero_step() { let graph = semantic_graph(4, &[(0, 1, 0.5)]); @@ -597,7 +615,7 @@ fn draw_computes_the_estimator_scales() { assert_eq!(populations.relation.len(), 1); assert_eq!(populations.relation_scale, 2.0); - // Two of three landmarks drawn, so 3 / 2 = 1.5. + // the inverse sampling fraction gives 3 / 2 = 1.5. assert_eq!(populations.landmarks.len(), 2); assert_eq!(populations.landmark_scale, 1.5); @@ -681,6 +699,10 @@ fn draw_collects_pooled_mined_pairs() { assert_eq!(populations.hard_scale, 1.0); } +/// Maps corpus rows `{2, 5, 9}` to locals `{0, 1, 2}` in every family with their scales. +/// +/// Assembling populations over corpus rows `{2, 5, 9}` maps them in ascending order to locals +/// `{0, 1, 2}` in every family and gathers the matching entries of the local-scale table. #[test] fn assemble_reindexes_into_the_local_domain() { // Corpus rows {2, 5, 9} participate; ascending order maps them to @@ -730,6 +752,10 @@ fn assemble_reindexes_into_the_local_domain() { ); } +/// Panics with the documented message on relation edges without a local-scale table. +/// +/// Assembling a batch that carries relation edges but no local-scale table panics with the +/// documented message. #[test] #[should_panic(expected = "relation edges need the step's local scales")] fn assemble_rejects_relation_edges_without_scales() { @@ -746,18 +772,22 @@ fn assemble_rejects_relation_edges_without_scales() { drop(Batch::assemble(populations, None)); } +/// Reads exactly zero losses and a `-0.5` surrogate on unit-distance dyadic pairs. +/// +/// On a semantic pair and an ordinary pair at unit distance under the dyadic affinity, the loss +/// values are exactly zero, the surrogate is exactly `-0.5`, the coordinate gradient equals the +/// hand-derived field, and the budget pass records no nodes because no relation edges exist. #[test] fn objective_matches_the_hand_computed_semantic_field() { // Rows {0..3}: a semantic pair (0, 1) and an ordinary pair (2, 3), // both at unit distance. With the dyadic affinity the values are // exactly zero and the derivative mass is exactly 0.25 per pair. // - // Semantic factor: λ_S · scale = 0.5 · 2 = 1, so the pair - // gradient is difference · (2 · 1 · 0.25) = (-0.5, 0) at row 0. - // Ordinary factor: λ_N · scale = 2 · 1 = 2, so the pair - // gradient is (0, -1) · (2 · 2 · -0.25) = (0, 1) at row 2. + // Semantic factor: λ_S · scale = 0.5 · 2 = 1: the pair gradient is + // difference · (2 · 1 · 0.25) = (-0.5, 0) at row 0. Ordinary factor: λ_N · scale = 2 · 1 = 2: + // the pair gradient is (0, -1) · (2 · 2 · -0.25) = (0, 1) at row 2. // - // Surrogate: + = 0.5 - 1 = -0.5 exactly. + // Surrogate: ⟨y₁, g₁⟩ + ⟨y₃, g₃⟩ = 0.5 - 1 = -0.5 exactly. let mut populations = empty_populations(non_negative!(0.0)); populations.semantic = vec![NodePair::new(node!(0), node!(1))]; populations.semantic_scale = 2.0; @@ -800,6 +830,9 @@ fn relation_fixture() -> ( (indexes, scales) } +/// Assembles the two-row relation batch of `relation_fixture` at step `eta`. +/// +/// It first pins the group weights and edge confidence the hand derivations assume. fn relation_batch( indexes: &RelationIndexes, scales: &LocalScales, @@ -827,11 +860,11 @@ fn relation_batch( Batch::assemble(populations, Some(scales)) } -/// The relation field rides whole and every bucket records its measurement. +/// The relation field is applied whole and every bucket records its measurement. #[test] fn objective_applies_the_relation_field_and_records_the_buckets() { - // Coordinates (0,0), (1,0) give d = 1. Local scales 0.5 with guard 0.5 give unit normalization, - // so z = 1 on the Proximal radius. + // Coordinates (0,0), (1,0) give d = 1. Local scales 0.5 with guard 0.5 give unit + // normalization: z = 1 on the Proximal radius. // // Semantic gradient at row 0: (-0.5, 0), norm 0.5 = baseline. // Relation factor: η · λ_R · scale · c · ν · strength @@ -862,11 +895,16 @@ fn objective_applies_the_relation_field_and_records_the_buckets() { assert_eq!(types[0].1.nodes(), 2); assert_eq!(types[0].1.mean_ratio(), Some(1.0)); - // Both endpoints have attraction degree one: rank 2 of 2 - // participating rows, decile (2-1)*10/2 = 5. + // Both endpoints have attraction degree one: rank 2 of 2 participating rows, decile + // (2 − 1)·10/2 = 5. assert_eq!(metrics.deciles()[5].nodes(), 2); } +/// Reads the two-row relation loss as `0.5 · ln 2` within `1e-6` with zero semantic loss. +/// +/// The relation loss of the two-row fixture equals +/// `factor · proximal · temperature · softplus(0) = 0.5 · ln 2` within `1e-6`, while the semantic +/// loss stays zero. #[test] fn objective_reports_the_relation_loss_value() { // The relation value is factor · proximal weight · temperature · @@ -916,11 +954,10 @@ fn relation_gradients_are_linear_in_the_lens() { #[test] fn support_terms_ride_autodiff_outside_the_budget() { - // One landmark holding row 1 at (2, 0) while it sits at (1, 0): - // residual d = 1 smoothed to √(1.25) - 0.5, normalized by - // radius 0.5 + ε 0.5 = 1, inside the unit Huber threshold. - // Its gradient flows through autodiff; row 0 keeps exactly its - // semantic gradient, certifying the two terms stay separate. + // One landmark holding row 1 at (2, 0) while it lies at (1, 0): residual d = 1 smoothed to + // √1.25 - 0.5, normalized by radius 0.5 + ε 0.5 = 1, inside the unit Huber threshold. Its + // gradient flows through autodiff. Row 0 keeps exactly its semantic gradient, certifying the + // two terms stay separate. let mut populations = empty_populations(non_negative!(0.0)); populations.semantic = vec![NodePair::new(node!(0), node!(1))]; populations.semantic_scale = 2.0; @@ -957,6 +994,7 @@ fn support_terms_ride_autodiff_outside_the_budget() { ); } +/// A NaN coordinate in the frame makes `evaluate` return `StepError::Diverged` naming that row. #[test] fn evaluate_rejects_non_finite_coordinates() { let mut populations = empty_populations(non_negative!(0.0)); @@ -976,6 +1014,11 @@ fn evaluate_rejects_non_finite_coordinates() { ); } +/// Fills the biased-exponent buckets and exact moments for five recorded displacements. +/// +/// Recording displacements `0, 0.5, 1, 1.5, 2` fills the biased-exponent buckets +/// `0, 126, 127, 127, 128` and the moments report count 5, sum 5, sum of squares 7.5 and maximum 2 +/// exactly. #[test] fn displacement_histogram_buckets_by_exponent() { // Buckets are f32 biased exponents: 0.5 → 126, 1.0 and 1.5 → @@ -1005,6 +1048,10 @@ fn displacement_histogram_buckets_by_exponent() { assert_eq!(moments.maximum(), 2.0); } +/// Lists each relation type once, ascending, with deduplicated ascending participants. +/// +/// `TypeParticipants` lists each relation type once in ascending ontology-row order with its +/// participating rows deduplicated and ascending. #[test] fn type_participants_deduplicate_and_order_rows() { let indexes = relation_indexes( @@ -1035,11 +1082,16 @@ fn type_participants_deduplicate_and_order_rows() { ); } +/// Fills the overall, decile and type buckets from three rows moving by `1`, `0` and `5`. +/// +/// Measuring three rows moving by `1`, `0` and `5` fills the overall histogram and moments, the +/// decile bucket only with the two participating rows, and the type bucket with its participants' +/// displacements. #[test] fn displacement_summary_reports_every_axis() { - // Rows 0 and 1 participate in relation 11 (degree one each, upper - // rank two of two participants: decile (2-1)*10/2 = 5); row 2 has - // no attraction evidence and enters the overall bucket only. + // Rows 0 and 1 participate in relation 11 (degree one each, upper rank two of two + // participants: decile (2 − 1)·10/2 = 5). Row 2 has no attraction evidence and enters the + // overall bucket only. // Displacements: row 0 moves by exactly 1, row 1 not at all, and // row 2 by exactly 5 (a 3-4-5 triangle). let indexes = relation_indexes(3, &[proximal_policy(11)], vec![instance(0, 11, 0, 1)]); @@ -1092,11 +1144,12 @@ fn displacement_summary_reports_every_axis() { /// Corpus rows of the padding fixtures. /// /// Exactly 32 rows participate in the gradient certificate's batch. At 32 every tensor of both the -/// padded and the unpadded graph reaches the CPU backend's SIMD dispatch threshold, so both graphs -/// compute with the same element-wise kernels. Below it the backend mixes dispatch paths, and the -/// comparison then measures the backend's reciprocal estimate (the SIMD reciprocal is a hardware -/// approximation that the autodiff division backward consumes) rather than the padding. +/// padded and the unpadded graph reaches the CPU backend's SIMD dispatch threshold, and both graphs +/// therefore compute with the same element-wise kernels. Below it the backend mixes dispatch paths, +/// and the comparison then measures the backend's reciprocal estimate (the SIMD reciprocal is a +/// hardware approximation that the autodiff division backward consumes) rather than the padding. const PADDING_ROWS: usize = 32; +/// Component count of the padding fixture's representation storage. const PADDING_CAPACITY: usize = PADDING_ROWS * PROJECTOR_DIMENSIONS; /// Builds the padding fixtures' input columns. @@ -1141,7 +1194,7 @@ fn padding_column_view<'corpus>( /// Nudges every parameter off its initialization. /// /// The identity-contract layers initialize to zero and would block gradient flow into the deep -/// block parameters, leaving the padding certificate comparing zeros with zeros; a deterministic +/// block parameters, leaving the padding certificate comparing zeros with zeros. A deterministic /// ramp makes every parameter's gradient generically nonzero. struct Perturb; @@ -1165,13 +1218,17 @@ impl ModuleMapper for Perturb { let shape = tensor.shape(); let device = tensor.device(); let ramp = Tensor::from_data(TensorData::new(ramp, shape), &device); - // The sum is an interior autodiff node; re-rooting it as a - // required-gradient leaf is what lets gradients accumulate at - // the perturbed parameter. + // The sum is an interior autodiff node. Re-rooting it as a required-gradient leaf is + // what lets gradients accumulate at the perturbed parameter. Param::from_mapped_value(id, (tensor + ramp).detach().require_grad(), mapper) } } +/// Pads four rows to the alignment by replicating the last row and fills the condition column. +/// +/// `input` pads four participating rows up to the row alignment by replicating the last row's +/// representation and role, fills the condition column with the step, and `input_aligned` at +/// alignment one returns the unpadded four-row frame. #[test] fn input_pads_the_gathered_rows_to_the_alignment() { // Rows {0, 1, 2, 5} participate: four rows pad to the alignment, @@ -1226,6 +1283,10 @@ fn input_pads_the_gathered_rows_to_the_alignment() { assert!(condition.iter().all(|&eta| eta == 0.5)); } +/// Evaluates a padded frame to the exact-cover loss with zero gradient on the padded rows. +/// +/// A two-row semantic batch evaluated on a four-row frame whose tail twins the last row yields the +/// same loss as the exact-cover frame and deposits exactly zero gradient on the padded rows. #[test] fn padded_frame_adds_zero_force() { // The two-row semantic batch against a four-row frame whose tail @@ -1259,6 +1320,7 @@ fn padded_frame_adds_zero_force() { assert_eq!(gradient, [-0.5, 0.0, 0.5, 0.0, 0.0, 0.0, 0.0, 0.0]); } +/// Evaluating a two-row batch against a one-row frame panics with the documented cover message. #[test] #[should_panic(expected = "cover the batch rows")] fn evaluate_rejects_a_frame_smaller_than_the_batch() { @@ -1274,8 +1336,8 @@ fn evaluate_rejects_a_frame_smaller_than_the_batch() { /// The coordinate leaf's gradient under the surrogate's backward pass, as flat values. /// -/// Unlike [`leaf_gradient`], the leaf arrives already built, so a padded shape wider than the -/// batch keeps its padded rows in the reading. +/// Unlike [`leaf_gradient`], the leaf arrives already built, and a padded shape wider than the +/// batch therefore keeps its padded rows in the reading. fn frame_gradient(leaf: &Tensor, surrogate: &Tensor) -> Vec { leaf.grad(&surrogate.backward()) .expect("the surrogate reaches the coordinate leaf") @@ -1294,21 +1356,16 @@ fn frame_values(frame: &Tensor) -> Vec { .expect("coordinates are f32") } +/// Projects and evaluates bit-equal results across padded and unpadded materializations. +/// +/// With every loss family active over enough rows to clear the CPU backend's SIMD threshold, the +/// padded and unpadded materializations project bit-equal coordinates for participating rows, +/// evaluate to bit-equal losses, and a leaf at the padded shape carries exactly zero force on every +/// padded row while the participating rows carry force. #[test] fn padding_zero_force_at_simd_scale() { - // Semantic, ordinary, relation, and landmark families all - // participate, so every loss path crosses the padded frame. The - // padded and unpadded materializations of the same batch project - // bit-equal coordinates for the participating rows and evaluate - // to bit-equal loss values, and the padded rows carry exactly - // zero force: a coordinate leaf of the padded shape deposits an - // exactly-zero gradient on every padded row, read from the one - // graph that computes it. - // - // The batch covers all [`PADDING_ROWS`] corpus rows so the - // tensors clear the CPU backend's SIMD dispatch threshold - // (see the constant's documentation) - the certificate compares - // the padding, not the backend's kernel election. + // PADDING_ROWS keeps both shapes above the CPU backend's SIMD dispatch threshold. Using the + // same kernel for each keeps the comparison specific to padding. let indexes = relation_indexes( PADDING_ROWS, &[proximal_policy(7)], @@ -1399,7 +1456,7 @@ fn padding_zero_force_at_simd_scale() { ); // A coordinate leaf at the padded materialization's own values: the padded tail deposits - // an exactly-zero gradient, so padding adds no force at SIMD-dispatch scale. + // an exactly-zero gradient, and padding therefore adds no force at SIMD-dispatch scale. let padded_rows = PADDING_ROWS.next_multiple_of(ROW_ALIGNMENT.get()); let leaf_frame = leaf(&padded_values, padded_rows); let mut leaf_metrics = BudgetBreakdown::new(); @@ -1427,6 +1484,7 @@ struct RecordingSnapshots { } impl RecordingSnapshots { + /// A recorder whose `projector_sample_size` answers `appetite`. fn new(appetite: usize) -> Self { Self { appetite, @@ -1434,6 +1492,7 @@ impl RecordingSnapshots { } } + /// The snapshots recorded so far, each as its sampled coordinates and landmark count. fn snapshots(&self) -> Vec<(Vec, usize)> { self.snapshots .lock() @@ -1443,7 +1502,6 @@ impl RecordingSnapshots { } impl Progress for RecordingSnapshots { - /// The fixture watches snapshots, so nothing crosses into owning machinery. type Detached = NoProgress; fn detach(&self) -> NoProgress { @@ -1462,7 +1520,7 @@ impl Progress for RecordingSnapshots { } } -/// A frame whose every row sits at its own row index, so a report names the rows it sampled. +/// A frame whose every row lies at its own row index: a report names the rows it sampled. fn identity_frame(rows: usize) -> Vec { (0..rows) .map(|row| { @@ -1512,6 +1570,10 @@ fn landmark(row: usize) -> SupportAnchor { } } +/// Lists both landmarks first and then an even stride under a budget of four over eight rows. +/// +/// A budget of four over eight rows with two landmarks lists both landmarks first and then an even +/// stride over the remaining rows. #[test] fn a_snapshot_sample_leads_with_landmarks_and_strides_the_rest() { let landmarks = [landmark(4), landmark(0)]; @@ -1535,19 +1597,24 @@ fn a_snapshot_sample_never_repeats_a_landmark_row() { assert_eq!(landmarks, 2); } +/// Fills only five of ten slots from sixty landmarks and the rest from the interior. +/// +/// Sixty landmarks against a budget of ten fill only five slots, and the other five come from the +/// non-landmark interior. #[test] fn landmarks_take_at_most_half_a_snapshot_budget() { let landmarks: Vec> = (0..60).map(landmark).collect(); let sample = SnapshotSample::select(100, &landmarks, 10); let (rows, landmarks) = sampled_rows(&sample, 100, 10); - // Half the budget holds the skeleton; the interior keeps the rest, - // so a landmark-rich corpus still shows more than its anchors. + // Half the budget holds the skeleton and the interior keeps the rest: a landmark-rich corpus + // still shows more than its anchors. assert_eq!(landmarks, 5); assert_eq!(rows.len(), 10); assert!(rows[5..].iter().all(|&row| row >= 60), "{rows:?}"); } +/// A budget larger than the corpus samples every row exactly once, landmark first. #[test] fn a_budget_beyond_the_corpus_reports_every_row_once() { let sample = SnapshotSample::select(5, &[landmark(2)], 4_096); diff --git a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/mod.rs b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/mod.rs index f11a6ae0666..6331fd9ac30 100644 --- a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/mod.rs @@ -2,10 +2,10 @@ //! //! At the end of semantic-only training the data sets the Proximal radius. The calibration measures //! the locally normalized distance `z` over the pairs of every reviewed-Proximal relation type and -//! freezes the radius at the 25th percentile, so the Proximal energy pulls on the outlying three +//! freezes the radius at the 25th percentile: the Proximal energy pulls on the outlying three //! quarters. The low quartile anchors the boundary in the population the semantic baseline already //! satisfies - reviewed pairs the embedding placed together - and everything beyond it feels the -//! pull, so the lens moves reviewed geometry instead of policing its fringe. "Outlying three +//! pull: the lens moves reviewed geometry instead of policing its fringe. "Outlying three //! quarters" counts in the units that matter, since the force the training loop will actually apply //! to a pair weights that pair's `z`, //! @@ -15,15 +15,15 @@ //! ``` //! //! where `min(cap, n) / n` is the pair's inclusion probability once the relation sampler draws its -//! type (the sampler draws types uniformly, so the type-level factor is constant and drops out of +//! type (the sampler draws types uniformly, and the type-level factor is constant and drops out of //! the percentile), `c` is effective confidence, `ν` the degree normalization, and `p_P · h` the //! group's Proximal class weight and strength multiplier. The sampler factor keeps a high-volume -//! type from buying the radius with edge count; the degree factor keeps hub-heavy types from +//! type from buying the radius with edge count. The degree factor keeps hub-heavy types from //! inflating it - a pair into a high-degree hub exerts proportionally little force on the layout, //! and its pull on the percentile shrinks in the same proportion. Both factors come from the built -//! artifacts, so the measurement cannot drift from what training consumes. The `min(cap, n) / n` +//! artifacts, and the measurement cannot drift from what training consumes. The `min(cap, n) / n` //! factor is the relation objective's own per-type clip. This calibration and the training sampler -//! move in lockstep by contract, so changing the factor re-derives both surfaces together. +//! move in lockstep by contract: changing the factor re-derives both surfaces together. //! //! The calibration measures `z = d / √((ρ_i + ε)(ρ_j + ε))` in the relation loss's own //! normalization convention, using the same local scales and the same scale guard the relation @@ -42,13 +42,13 @@ use super::{PlacementClass, ResolvedVerdict}; /// The weighted-quantile fraction at which the Proximal radius freezes. /// -/// Both the pooled radius and every leave-one-out radius freeze at this fraction, so the two +/// Both the pooled radius and every leave-one-out radius freeze at this fraction, and the two /// surfaces cannot drift apart. The per-type evidence quartiles are descriptive and keep their own -/// literals; their first entry coinciding with this fraction is today's policy choice, not a shared -/// definition. +/// literals. Their first entry coinciding with this fraction is a policy choice rather than a +/// shared definition. /// /// The fraction itself is a policy choice with no derivation behind it. It freezes engagement -/// demand at a quantile the semantic baseline already achieves, so demand is weakest where the +/// demand at a quantile the semantic baseline already achieves: demand is weakest where the /// relation signal adds most. const RADIUS_FRACTION: OpenUnitFraction = OpenUnitFraction::new(0.25).expect("0.25 lies inside (0, 1)"); @@ -68,8 +68,8 @@ use crate::{ /// Validated calibration parameters. /// /// `cap` is the relation sampler's per-type edge cap, `epsilon` the relation energy's scale -/// guard, and `temperature` the Proximal transition temperature the composed energy runs with; -/// all three must be the values the training loop runs with, or the measured radius and its +/// guard, and `temperature` the Proximal transition temperature the composed energy runs with. +/// All three must be the values the training loop runs with, or the measured radius and its /// stability certificate describe a different population and tolerance than the loss acts on. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct CalibrationOptions { @@ -81,8 +81,8 @@ pub(crate) struct CalibrationOptions { impl CalibrationOptions { /// Creates calibration parameters. /// - /// The domains ride in the types: [`Positive`] is the domain the relation energy accepts - /// for both the scale guard and the temperature. + /// Every parameter carries its domain in its type: [`Positive`] is the domain the relation + /// energy accepts for both the scale guard and the temperature. #[must_use] pub(crate) const fn new(cap: NonZero, epsilon: Positive, temperature: Positive) -> Self { Self { @@ -110,7 +110,7 @@ pub(crate) struct TypeCalibration { /// /// [`None`] when nothing else carries mass. The spread of these values across types is the /// review-sufficiency reading: a tight cluster means the radius does not hinge on any - /// single review, a wide one names the review that owns it. + /// single review, and a wide one names the review that owns it. pub radius_without: Option, } @@ -133,11 +133,11 @@ pub(crate) struct ProximalCalibration { } impl ProximalCalibration { - /// Returns the calibration for a run with no measured population. Its radius and certificate - /// are absent. + /// Returns the calibration for a run with no measured population. /// - /// A forceless corpus carries no reviewed-Proximal geometry at all, so the record states - /// the absence directly rather than measuring an empty population. + /// Its radius and certificate are absent. A forceless corpus carries no reviewed-Proximal + /// geometry at all, and the record states the absence directly rather than measuring an empty + /// population. pub(crate) const fn vacuous() -> Self { Self { radius: None, @@ -146,12 +146,12 @@ impl ProximalCalibration { } } - /// A calibration assembled from the given readings, for fixtures. + /// Assembles a calibration from the given readings, for fixtures. /// /// # Panics /// /// This panics when `radius` and `stability` disagree on presence: the certificate is an - /// evaluated reading of the same positive-mass population the radius froze from, so a + /// evaluated reading of the same positive-mass population the radius froze from, and a /// fixture carrying one without the other states an impossible record. #[cfg(test)] pub(crate) fn fixture( @@ -180,8 +180,8 @@ impl ProximalCalibration { /// # Panics /// /// This panics when an edge references a row outside the frame. The index and the frame - /// describe one corpus, so a mismatch is a wiring defect. A pair whose reading or force - /// weight falls outside its validated domain panics for the same reason: the coordinates, + /// describe one corpus, and a mismatch is therefore a wiring defect. A pair whose reading or + /// force weight falls outside its validated domain panics for the same reason: the coordinates, /// scales, and force factors that produce them are all validated upstream. pub(crate) fn new( verdicts: &[ResolvedVerdict], @@ -286,11 +286,11 @@ impl ProximalCalibration { } } - /// How far a single omitted type moves the pooled radius. + /// Returns how far a single omitted type moves the pooled radius. /// /// The maximum of `|R_{-t} - R|` over the types with a leave-one-out reading. [`None`] /// without a pooled radius or when no other type carries mass. A tight spread means the - /// radius does not hinge on any single review, a wide one names the review that owns it. + /// radius does not hinge on any single review, and a wide one names the review that owns it. pub(crate) fn leave_one_out_spread(&self) -> Option { let radius = self.radius?.widen(); self.types @@ -300,14 +300,17 @@ impl ProximalCalibration { .max() } + /// Returns the frozen radius `u_P`, or [`None`] when no reviewed pair carries mass. pub(crate) const fn radius(&self) -> Option { self.radius } + /// Returns the per-type evidence, ascending by relation row. pub(crate) const fn types(&self) -> &[TypeCalibration] { &self.types } + /// Returns the stability certificate, present exactly when [`radius`](Self::radius) is. pub(crate) const fn stability(&self) -> Option<&StabilityCertificate> { self.stability.as_ref() } @@ -326,7 +329,7 @@ struct PairReading { /// /// This is the single home of the calibration weight formula. The pooled percentile, the /// per-type quantiles, the stability certificate, and the per-refresh fraction report all -/// read their populations through it, so the readings cannot drift apart. +/// read their populations through it, and the readings cannot drift apart. fn group_readings<'group, N, E>( group: &'group AttractionGroup, frame: ScaledFrame<'group, N>, @@ -354,7 +357,7 @@ where let normalization = scales.normalization(source, target, options.epsilon); PairReading { - // Never NaN and never negative; a quotient past the working range saturates, so one + // Never NaN and never negative. A quotient past the working range saturates, and one // extreme pair reads as the ceiling instead of poisoning the quantile. z: distance.saturating_div(normalization), weight: DNonNegative::new( @@ -372,7 +375,7 @@ where /// /// Each entry re-reads the radius quantile with its own rows excluded, over the shared pooled /// population in walk order. The surviving mass is summed over the surviving entries rather -/// than subtracted from the total, so the threshold cannot drift from the walked mass by +/// than subtracted from the total, and the threshold cannot drift from the walked mass by /// cancellation. /// /// # Panics @@ -412,8 +415,8 @@ fn assign_radius_without( /// ([`ProximalCalibration::new`]), re-measured over the given frame and scales: the per-refresh /// drift report re-asks the freeze-time question of a later frame, on the same step the freeze /// measured. At the freeze frame itself the reading is the smallest mass share the atom -/// structure realizes at or above the radius fraction, so later readings drift against that -/// first entry rather than against the fraction constant. +/// structure realizes at or above the radius fraction. Later readings therefore drift against +/// that first entry rather than against the fraction constant. /// /// Returns [`None`] when no reviewed pair carries mass. /// @@ -421,8 +424,8 @@ fn assign_radius_without( /// /// This panics when an edge references a row outside the frame - the same one-corpus wiring /// contract as [`ProximalCalibration::new`], and the same validated-domain contract on every pair's -/// reading and -/// force weight. It also panics when a weight sum overflows, which is a defect of the weights. +/// reading and force weight. It also panics when a weight sum overflows, which is a defect of the +/// weights. pub(crate) fn reviewed_fraction_within( verdicts: &[ResolvedVerdict], index: &AttractionIndex, @@ -456,7 +459,7 @@ where } // The total's finish is what refuses a population whose weight sums overflowed. The within - // mass sums a subsequence of the total's own non-negative addends, so the share lies within + // mass sums a subsequence of the total's own non-negative addends, and the share lies within // accumulation rounding of [0, 1]. let total = total .finish() @@ -489,7 +492,7 @@ fn weighted_quantile( } // Reachable only through cumulative rounding shaving the last step - // below the threshold; the answer is the distribution's maximum + // below the threshold. The answer is the distribution's maximum // either way. last.expect("a positive total implies entries") } diff --git a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/mod.rs b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/mod.rs index 724d46146ef..6d7c2bb0c3b 100644 --- a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/mod.rs @@ -11,7 +11,7 @@ //! materiality tolerance, and `δ` the false-pass budget. The bound holds conditional on the //! independence licence: pair draws independent with fixed weights. The sentence is not //! `Pr(correct | pass) ≥ 1 − δ`, because the conditional claim needs a marginal over the latent -//! distribution that nothing here supplies. It also makes no claim about the latent band's +//! distribution, and the certificate carries none. It also makes no claim about the latent band's //! width: an atom inside the band boundary collapses the empirical gap while the latent gap //! stays arbitrary. For a consumer, a pass certifies the recorded radius lies within one //! temperature of the latent weighted-population quantile except on a δ-probability event - @@ -30,7 +30,7 @@ //! `ε = q` boundary (which is what makes the floor inclusive) and the upper event through the //! fixed strict-threshold variables at `u−`. A pass therefore needs //! `n_eff ≥ ln(2/δ)/(2q²)` - the legibility floor - before the gap is even consulted. A -//! hundred low-weight pairs therefore never outrank ten balanced ones by count alone. +//! hundred low-weight pairs never outrank ten balanced ones by count alone. //! //! `ε*`, `n*`, and the attained bit are derived after the decision and persist as evidence for //! readers: `ε* = sup{ε ∈ (0, q] : G(ε) ≤ τ}`, `n* = ln(2/δ)/(2·ε*²)`, and @@ -66,8 +66,10 @@ const FALSE_PASS_BUDGET: OpenUnitFraction = /// The materiality multiplier `κ` in `τ = κ·T`. /// -/// A radius error inside the sigmoid transition's own width does not change what the Proximal -/// energy does to a pair, so one temperature is the materiality unit. +/// One temperature is the chosen tolerance. A radius error of one `T` shifts the Proximal pull +/// `sigmoid((z − radius) / T)` by the transition's own blur (at `z = radius` it moves from `1/2` +/// to `1/(1+e)`), and the certificate declares an error of that size immaterial rather than +/// without effect. const MATERIALITY_MULTIPLIER: DPositive = DPositive::new(1.0).expect("the materiality unit is positive"); @@ -96,8 +98,8 @@ pub(crate) enum StabilityBound { /// The reviews arm's evaluated certificate, persisted beside the boundary calibration. /// -/// Every constant the decision consumed rides in the record, so the artifact names its own -/// regime and a future compatible arm evaluates fresh rather than copying a scalar. +/// Every constant the decision consumed is in the record: the artifact names its own regime, and +/// a future compatible arm evaluates fresh rather than copying a scalar. #[derive(Debug, Clone, PartialEq)] pub(crate) struct StabilityCertificate { /// The quantile level `q` the estimator freezes at. @@ -110,8 +112,9 @@ pub(crate) struct StabilityCertificate { pub temperature: DPositive, /// The materiality tolerance `τ = κ·T`. pub tau: DPositive, - /// The effective support `(Σw)² / Σw²` of the positive-mass population - the derivation's - /// `n_eff`. + /// The effective support `(Σw)² / Σw²` of the positive-mass population. + /// + /// The derivation's `n_eff`. pub effective_support: DPositive, /// The raw pair count, reader context only: the decision never consults it. pub pairs: usize, @@ -121,8 +124,8 @@ pub(crate) struct StabilityCertificate { pub epsilon_zero: DPositive, /// The empirical interval width `G(ε₀) = Q̂(q+ε₀) − Q̂(q−ε₀)`, clamped-endpoint semantics. /// - /// Past the floor (`ε₀ > q`) the lower level clamps at the positive-mass minimum, so the - /// reading degrades toward the population's full width rather than vanishing; the decision + /// Past the floor (`ε₀ > q`) the lower level clamps at the positive-mass minimum, and the + /// reading degrades toward the population's full width rather than vanishing. The decision /// has already failed on the floor conjunct there. pub gap: DNonNegative, /// The evaluated bound, or the record that none exists. @@ -167,16 +170,16 @@ impl Support { Self { values, cumulative } } - /// The total positive mass `W`. + /// Returns the total positive mass `W`. const fn mass(&self) -> f64 { self.cumulative.last().copied().unwrap_or(0.0) } - /// The quantile at `level`, with the clamped-endpoint convention. + /// Returns the quantile at `level`, with the clamped-endpoint convention. /// /// Levels at or below zero read the positive-mass minimum and levels at or above one the - /// positive-mass maximum; between them the production walk's first crossing of - /// `level · W` decides. + /// positive-mass maximum. Between them the production walk's first crossing of `level · W` + /// decides. fn quantile(&self, level: f64) -> NonNegative { if level <= 0.0 { return self.values[0]; @@ -189,20 +192,20 @@ impl Support { let position = self .cumulative .partition_point(|&cumulative| cumulative < threshold); - // Cumulative rounding can shave the last prefix below `level · W` for levels near one; - // the walk's answer is the maximum either way. + // Cumulative rounding can shave the last prefix below `level · W` for levels near one. + // The walk's answer is the maximum either way. self.values[position.min(self.values.len() - 1)] } - /// The empirical interval width `G(ε) = Q̂(q+ε) − Q̂(q−ε)`. + /// Returns the empirical interval width `G(ε) = Q̂(q+ε) − Q̂(q−ε)`. fn gap(&self, level: f64, epsilon: f64) -> DNonNegative { - // The first-crossing quantile is nondecreasing in its level - the threshold is monotone - // in the level, `partition_point` is monotone in the threshold, and the values ascend, - // therefore the width is non-negative and the absolute value is exact on it. + // The first-crossing quantile is nondecreasing in its level: the threshold is monotone + // in the level, `partition_point` is monotone in the threshold, and the values ascend. + // Therefore the width is non-negative, and the absolute value is exact on it. (self.quantile(level + epsilon).widen() - self.quantile(level - epsilon).widen()).abs() } - /// The effective support `(Σw)² / Σw²`, computed on normalized weights. + /// Returns the effective support `(Σw)² / Σw²`, computed on normalized weights. /// /// Normalizing by `W` first keeps the squares away from underflow: `Σp²` is at least the /// reciprocal of the row count. @@ -231,12 +234,12 @@ impl Support { /// `sorted` is the calibration's pooled `(z, weight)` population, ascending by `z` in the same /// order the frozen radius walked, `type_masses` the per-type total masses, `pairs` the raw pair /// count, and `temperature` the generation's frozen `T`. Zero-weight rows and zero-mass types -/// are excluded here, so callers pass their populations unfiltered. +/// are excluded here, and callers pass their populations unfiltered. /// /// # Panics /// /// This panics when no row carries positive mass. The caller evaluates the certificate exactly -/// when the boundary froze a measured radius, which requires positive mass, so an empty support +/// when the boundary froze a measured radius, which requires positive mass, and an empty support /// is a wiring defect. pub(crate) fn evaluate( sorted: impl IntoIterator, @@ -259,8 +262,8 @@ pub(crate) fn evaluate( let effective_support = support.effective(); // In domain with no check: the confidence is a positive constant and the effective support - // a validated positive, so the quotient stays positive - a reciprocal of a finite value - // never rounds to zero - and the square root of a positive value is positive. + // a validated positive. The quotient stays positive - a reciprocal of a finite value never + // rounds to zero - and the square root of a positive value is positive. let epsilon_zero = DPositive::new_unchecked((confidence / (2.0 * effective_support)).sqrt()); let gap = support.gap(quantile.get(), epsilon_zero.get()); let pass = epsilon_zero <= quantile && gap <= tau; @@ -313,7 +316,7 @@ pub(crate) fn evaluate( /// Returns [`None`] when the safe set is empty. /// /// `G` is a nondecreasing step function of `ε` whose value can change only where `q − ε` or -/// `q + ε` crosses a prefix-mass fraction, so the candidates are those crossings plus the domain +/// `q + ε` crosses a prefix-mass fraction, and the candidates are those crossings plus the domain /// endpoint `q`. Between consecutive candidates `G` is constant, and each candidate is evaluated /// exactly by the same walk that evaluates `G(ε₀)` - the sup and its membership bit come from /// evaluation, never from continuity reasoning. diff --git a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/tests.rs b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/tests.rs index 8a7b5066332..380cc7d44ae 100644 --- a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/stability/tests.rs @@ -25,6 +25,10 @@ fn row(z: f32, weight: f64) -> (NonNegative, DNonNegative) { ) } +/// Compares `Support::quantile` with the production cumulative walk at every level `k/32`. +/// +/// `Support::quantile` agrees with the production cumulative walk at every level `k/32` over a +/// population with atoms on the level boundaries. #[test] fn quantile_matches_the_production_walk() { // Mixed dyadic weights, total exactly one, with atoms at the level boundaries. @@ -49,6 +53,10 @@ fn quantile_matches_the_production_walk() { } } +/// Ignores zero-weight rows in every quantile and clamps to the positive-mass extremes. +/// +/// Zero-weight rows below and inside the positive support change no quantile, and the clamps at +/// levels zero and one are the positive-mass minimum and maximum. #[test] fn quantile_ignores_zero_weight_rows_exactly_as_the_production_walk_does() { // A zero-weight row below the positive support and one inside it. @@ -80,6 +88,7 @@ fn quantile_ignores_zero_weight_rows_exactly_as_the_production_walk_does() { assert_eq!(support.quantile(1.0), non_negative!(2.0)); } +/// Adding a zero-weight row leaves every certificate field equal. #[test] fn zero_weight_rows_change_no_certificate_field() { let with_zero = [ @@ -98,14 +107,16 @@ fn zero_weight_rows_change_no_certificate_field() { assert_eq!(evaluated_with, evaluated_without); } +/// Sixty-four unit weights give an effective support of exactly 64. #[test] fn effective_support_of_balanced_dyadic_weights_is_the_count() { - // Unit weights over a power-of-two count keep every share and square dyadic, so the - // effective support is the count exactly. + // Unit weights over a power-of-two count keep every share and square dyadic: the effective + // support is the count exactly. let support = Support::new(balanced(&[1.0; 64])); assert_eq!(support.effective(), d_positive!(64.0)); } +/// Weights `(3, 1)` give an effective support of exactly `8/5`. #[test] fn effective_support_downweights_concentration() { // Weights (3, 1): shares (3/4, 1/4), squares sum 5/8, effective 8/5 - all dyadic-exact. @@ -116,8 +127,8 @@ fn effective_support_downweights_concentration() { #[test] fn thin_reviews_fail_the_legibility_floor() { // The thin-reviews shape is one resolving verdict with three balanced pairs. The floor at - // q = 0.25, δ = 0.05 is 29.51 effective pairs, so the run fails on the floor conjunct - // before the gap is consulted. + // q = 0.25, δ = 0.05 is 29.51 effective pairs: the run fails on the floor conjunct before + // the gap is consulted. let certificate = evaluate( balanced(&[1.0, 1.0, 1.0]), [d_non_negative!(3.0)], @@ -141,7 +152,7 @@ fn thin_reviews_fail_the_legibility_floor() { #[test] fn healthy_balanced_pairs_pass_and_the_suite_can_show_it() { // Sixty-four balanced pairs at one point: ε₀ = √(ln40/128) ≈ 0.1698 clears the floor and - // the gap is zero everywhere, so the certificate passes - the check provably can. + // the gap is zero everywhere. The certificate passes, which shows the check can. let certificate = evaluate( balanced(&[1.0; 64]), [d_non_negative!(64.0)], @@ -157,8 +168,8 @@ fn healthy_balanced_pairs_pass_and_the_suite_can_show_it() { assert_eq!(certificate.gap, d_non_negative!(0.0)); assert!(certificate.pass); - // The gap never exceeds τ on (0, q], so the sup is the domain endpoint and attained, and - // n* is the legibility floor itself. + // The gap never exceeds τ on (0, q]: the sup is the domain endpoint and attained, and n* is + // the legibility floor itself. assert_eq!( certificate.bound, StabilityBound::Finite { @@ -205,10 +216,14 @@ fn the_eight_point_shape_persists_an_unattained_bound() { ); } +/// Reads an `Unattainable` bound and a full-width gap from a prefix fraction exactly on `q`. +/// +/// A prefix fraction that falls exactly on `q` makes the bound `Unattainable`, the certificate +/// fail, and the gap read the population's full width. #[test] fn an_atom_boundary_at_the_level_is_unattainable() { - // The unit-weight shape (0, 10, 10, 10) puts the first prefix fraction exactly at q, so - // the interval width is ten for every ε in the domain - a cliff wider than τ that no mass + // The unit-weight shape (0, 10, 10, 10) puts the first prefix fraction exactly at q: the + // interval width is ten for every ε in the domain, a cliff wider than τ that no mass // stabilizes. let certificate = evaluate( balanced(&[0.0, 10.0, 10.0, 10.0]), @@ -219,15 +234,15 @@ fn an_atom_boundary_at_the_level_is_unattainable() { assert_eq!(certificate.bound, StabilityBound::Unattainable); assert!(!certificate.pass); - // Past the floor the lower endpoint clamps at the positive-mass minimum, so the persisted - // gap reads the population's full width. + // Past the floor the lower endpoint clamps at the positive-mass minimum: the persisted gap + // reads the population's full width. assert_eq!(certificate.gap, d_non_negative!(10.0)); } #[test] fn the_gap_at_epsilon_zero_uses_the_clamped_lower_endpoint() { - // The unit-weight shape (1, 2, 3, 4) gives ε₀ ≈ 0.679, past q, so the lower level clamps - // at the minimum and the upper level 0.929 walks to the maximum. The gap is exactly three. + // The unit-weight shape (1, 2, 3, 4) gives ε₀ ≈ 0.679, past q: the lower level clamps at + // the minimum and the upper level 0.929 walks to the maximum. The gap is exactly three. let certificate = evaluate( balanced(&[1.0, 2.0, 3.0, 4.0]), [d_non_negative!(4.0)], @@ -244,6 +259,10 @@ fn the_gap_at_epsilon_zero_uses_the_clamped_lower_endpoint() { assert!(!certificate.pass); } +/// Echoes the certificate's inputs and lets a reader recompute `ε₀` and `n*`. +/// +/// The certificate echoes its quantile, `δ`, `κ`, temperature, `τ`, pair count and mass, and the +/// recorded fields let a reader recompute `ε₀` and `n*`. #[test] fn the_certificate_records_its_regime() { let certificate = evaluate( @@ -272,10 +291,14 @@ fn the_certificate_records_its_regime() { ); } +/// Reads type effective supports of `1.6` and exactly two from two mass profiles. +/// +/// Type masses `(3, 1, 0)` give a type effective support of `1.6`, ignoring the zero-mass type, and +/// balanced masses `(2, 2)` read exactly two. #[test] fn type_effective_support_ignores_zero_mass_types() { - // Masses (3, 1, 0) total four with squares ten, so the support is 1.6 and the zero-mass - // type is not a type for the concentration reading. + // Masses (3, 1, 0) total four with squares ten: the support is 16/10 = 1.6, and the + // zero-mass type is not a type for the concentration reading. let certificate = evaluate( balanced(&[1.0; 4]), [ diff --git a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/tests.rs b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/tests.rs index bb067dc7ed4..efd2a37b319 100644 --- a/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/verdict/calibrate/tests.rs @@ -62,6 +62,9 @@ fn instance( } } +/// Builds the attraction index over `rows` from certified `policies` and instances. +/// +/// The attraction options are the defaults. fn attraction_index( rows: usize, policies: &[RelationPolicy], @@ -77,6 +80,7 @@ fn attraction_index( .attraction } +/// A resolved proximal verdict for `relation`. fn proximal_verdict(relation: u64) -> ResolvedVerdict { ResolvedVerdict { relation: OntologyRowId::new(relation), @@ -84,19 +88,26 @@ fn proximal_verdict(relation: u64) -> ResolvedVerdict { } } -/// Scales of 0.75 with the guard 0.25 make every normalization exactly one. +/// Builds a local scale from a literal test value. +/// +/// # Panics /// -/// Measured `z` therefore equals raw 2D distance. +/// Panics unless `value` is finite and non-negative. fn scale(value: f32) -> NonNegative { NonNegative::new(value).expect("test scales are finite and non-negative") } +/// Builds local scales with unit normalization. +/// +/// The scale `0.75` plus the guard `0.25` is exactly one, and measured `z` therefore equals the +/// raw 2D distance. fn unit_scales(rows: usize) -> LocalScales { LocalScales::new(IdSlice::from_boxed_slice( vec![scale(0.75); rows].into_boxed_slice(), )) } +/// Calibration options with per-type cap `cap`, quantile `0.25` and temperature `0.5`. fn options(cap: usize) -> CalibrationOptions { CalibrationOptions::new( NonZero::new(cap).expect("test limits are positive"), @@ -105,13 +116,17 @@ fn options(cap: usize) -> CalibrationOptions { ) } +/// Measures `z = 1`, mass `0.5` and radius one from one pair at distance five. +/// +/// One pair at distance five, with scales that make the normalization exactly five, measures +/// `z = 1`, mass `0.5` and radius one, with no leave-one-out radius for the only type. #[test] fn z_is_measured_in_the_loss_normalization_by_hand() { - // One disjoint pair with degrees 1 each, so ν = 1/√(2 · 2) = 0.5. + // One disjoint pair with degrees 1 each: ν = 1/√(2 · 2) = 0.5. let index = attraction_index(2, &[proximal_policy(5)], vec![instance(0, 5, 0, 1)]); - // d = 5 (a 3-4-5 triangle). Normalization = √((0.75 + 0.25) · (24.75 + 0.25)) = √(25) = 5, so z - // = 1 exactly. + // d = 5 (a 3-4-5 triangle). Normalization = √((0.75 + 0.25) · (24.75 + 0.25)) = √25 = 5: + // z = 1 exactly. let coordinates = [Vec2::new(0.0, 0.0), Vec2::new(3.0, 4.0)]; let scales = LocalScales::new(IdSlice::from_boxed_slice(Box::new([ scale(0.75), @@ -139,6 +154,10 @@ fn z_is_measured_in_the_loss_normalization_by_hand() { assert_eq!(outcome.types[0].radius_without, None); } +/// Reads total mass two and a low-quartile radius from four equal-weight pairs. +/// +/// Four equal-weight pairs at `z = 1, 2, 3, 4` give total mass two and a radius at the low quartile +/// `z = 1`. #[test] fn radius_is_the_weighted_p25() { // The fixture has four disjoint pairs (all ν = 0.5, weight 0.5) at z = 1, 2, 3, 4. @@ -178,6 +197,11 @@ fn radius_is_the_weighted_p25() { assert_eq!(outcome.types[0].radius_without, None); } +/// Lets a per-type cap decide which type owns the radius. +/// +/// A per-type cap of two evens an eight-pair type against a two-pair type so the low-`z` type owns +/// the radius, while cap eight lets volume buy the radius. The leave-one-out radii name the owner +/// in both cases. #[test] fn cap_bounds_a_high_volume_type() { // Type 5: eight disjoint pairs at z = 5. Type 9: two disjoint @@ -204,8 +228,8 @@ fn cap_bounds_a_high_volume_type() { let scales = unit_scales(20); let verdicts = [proximal_verdict(5), proximal_verdict(9)]; - // Cap 2: type 5's pairs sample at 2/8, so both types weigh 1.0 and - // type 9's first pair crosses the p25 threshold (0.5). + // Cap 2: type 5's pairs sample at 2/8. Both types then weigh 1.0, and type 9's first pair + // crosses the p25 threshold (0.5). let capped = ProximalCalibration::new( &verdicts, &index, @@ -234,10 +258,14 @@ fn cap_bounds_a_high_volume_type() { assert_eq!(uncapped.radius, Some(non_negative!(5.0))); } +/// Weighs a four-leaf hub below three disjoint pairs and follows the peers' radius. +/// +/// A hub with four leaves has more pairs but less mass than three disjoint pairs, the pooled radius +/// follows the peers' `z = 1`, and each type's leave-one-out radius is the other's atom. #[test] fn hubs_are_discounted_by_degree() { // Type 5 has one hub (node 0) linked to four leaves at z = 2 per pair. Within the group the - // hub's degree is 4, so ν = 1/√(5 · 2). Type 9: three disjoint pairs at z = 1 with ν = 1/2. + // hub's degree is 4: ν = 1/√(5 · 2). Type 9: three disjoint pairs at z = 1 with ν = 1/2. let instances = vec![ instance(0, 5, 0, 1), instance(1, 5, 0, 2), @@ -296,6 +324,10 @@ fn hubs_are_discounted_by_degree() { assert_eq!(peers.radius_without, Some(non_negative!(2.0))); } +/// Reads no radius and one zero-mass type entry from verdicts without attraction mass. +/// +/// A proximal verdict for a relation with no attraction group and an overlay verdict for one with a +/// group yield no radius and a single zero-mass type entry. #[test] fn missing_groups_and_foreign_classes_contribute_nothing() { let index = attraction_index(2, &[proximal_policy(5)], vec![instance(0, 5, 0, 1)]); @@ -331,6 +363,7 @@ fn missing_groups_and_foreign_classes_contribute_nothing() { assert_eq!(outcome.types[0].radius_without, None); } +/// With no resolved verdicts the calibration has no radius and no types. #[test] fn no_verdicts_yield_no_radius() { let index = attraction_index(2, &[proximal_policy(5)], vec![instance(0, 5, 0, 1)]); @@ -354,6 +387,10 @@ fn no_verdicts_yield_no_radius() { ); } +/// Reads the within-radius fraction inclusively at, between and below the atoms. +/// +/// The within-radius fraction reads the mass at or below a radius inclusively: `0.25` at the +/// `z = 1` atom, `0.5` between atoms, zero below the population and `None` with no reviewed mass. #[test] fn the_fraction_instrument_re_measures_the_freeze_population() { // The p25 fixture: four disjoint pairs (weight 0.5 each, total 2.0) at z = 1, 2, 3, 4. @@ -475,6 +512,10 @@ fn the_calibration_carries_its_stability_certificate() { assert_eq!(vacuous.stability, None); } +/// Reads a pooled radius of one and a leave-one-out spread of four from two equal types. +/// +/// With two types at `z = 5` and `z = 1` of equal mass, the pooled radius is one and the +/// leave-one-out spread is the larger movement, four. A vacuous calibration has no spread. #[test] fn the_leave_one_out_spread_reads_the_owning_review() { // Type 5 has two pairs at z = 5 (mass 1.0) and type 9 two pairs at z = 1 (mass 1.0). diff --git a/libs/@local/graph/atlas/src/salt/projector/verdict/mod.rs b/libs/@local/graph/atlas/src/salt/projector/verdict/mod.rs index 4c856d7c933..5e4ee3546db 100644 --- a/libs/@local/graph/atlas/src/salt/projector/verdict/mod.rs +++ b/libs/@local/graph/atlas/src/salt/projector/verdict/mod.rs @@ -6,11 +6,11 @@ //! by content hash, never derived by the pipeline. //! //! A type verdict for a store-native type names the exact type version whose rendered card the -//! reviewer saw, and resolution is version-precise, so the verdict binds only that version's -//! ontology row. No review covers the other versions of the same type, and they take their policy +//! reviewer saw. Resolution is version-precise: the verdict binds only that version's ontology +//! row. No review covers the other versions of the same type, and they take their policy //! from lower-precedence sources. A reviewed version absent from the corpus snapshot resolves to no -//! row at all, which resolution reports as evidence rather than an error, because snapshots -//! legitimately move past reviewed versions. Verdicts for foreign-corpus types carry no store +//! row at all. Resolution reports that outcome as evidence rather than an error: snapshots move +//! past reviewed versions in ordinary operation. Verdicts for foreign-corpus types carry no store //! identity and are likewise evidence. The record keeps the review, and nothing in this corpus //! answers to it. //! @@ -19,10 +19,9 @@ //! ontology row. The primary consumer is the training loop's phase boundary, which calibrates the //! Proximal radius from the reviewed-Proximal types' attraction pairs. //! -//! The reader parses and validates pair-level verdicts (a placement class for one concrete entity -//! pair) but does not yet resolve them. No exporter emits them, so nothing has pinned the wire form -//! of their entity references against real bytes. Their resolution lands with the first exporter -//! that produces one. +//! The reader parses, validates, and retains pair-level verdicts (a placement class for one +//! concrete entity pair) without resolving them. Their entity references have no pinned wire form, +//! because no exporter emits them. pub(crate) mod calibrate; @@ -53,7 +52,7 @@ pub enum InvalidReviewedVerdicts { Schema { found: Box }, /// A type verdict is not strictly after its predecessor in relation order. /// - /// Which also covers duplicated relations. + /// This also covers duplicated relations. UnorderedTypeVerdicts { index: usize }, /// A type verdict repeats an earlier verdict's versioned URL. DuplicateVersion { index: usize }, @@ -61,7 +60,7 @@ pub enum InvalidReviewedVerdicts { EmptyTypeVerdictField { index: usize, field: &'static str }, /// A pair verdict is not strictly after its predecessor in `(left, right)` order. /// - /// Which also covers duplicated pairs. + /// This also covers duplicated pairs. UnorderedPairVerdicts { index: usize }, /// A pair verdict carries an empty string field. EmptyPairVerdictField { index: usize, field: &'static str }, @@ -114,7 +113,7 @@ impl Error for InvalidReviewedVerdicts { /// A human-confirmed placement class. /// /// An `excluded` review records a supervised exclusion rather than a placement. The exporter omits -/// those reviews, so no fourth variant exists here. +/// those reviews, and no fourth variant exists here. #[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Deserialize)] #[serde(rename_all = "lowercase")] pub(crate) enum PlacementClass { @@ -141,7 +140,7 @@ pub(crate) struct TypeVerdict { pub reviewer: String, /// The exact type version whose card the reviewer saw - the resolution key. /// - /// Only types with a store identity record one; a verdict for a foreign-corpus type carries + /// Only types with a store identity record one. A verdict for a foreign-corpus type carries /// [`None`], can never resolve to an ontology row, and enters the unresolved evidence. pub versioned_url: Option, } @@ -177,7 +176,7 @@ struct Document { /// /// Construction checks the whole wire contract (declared schema, type verdicts strictly ascending /// by relation with unique versioned URLs, pair verdicts strictly ascending by `(left, right)`, and -/// no empty identity fields), so consumers read verdicts without re-checking. +/// no empty identity fields), and consumers read verdicts without re-checking. #[derive(Debug, Clone, PartialEq)] pub(crate) struct ReviewedVerdicts { // Construction validates the wire document into owned, normalized verdicts once per fit, and @@ -204,7 +203,6 @@ impl ReviewedVerdicts { } for (index, verdict) in document.type_verdicts.iter().enumerate() { - // This silly validation would collapse for (field, value) in [ ("relation", &verdict.relation), ("reviewer", &verdict.reviewer), @@ -279,15 +277,15 @@ impl ReviewedVerdicts { /// `ontology` is the type table in ontology row order, keyed by the corpus's own id type. Each /// verdict's versioned URL derives the id naming it in that id space, and a matching table /// position resolves the verdict to that row. Verdicts naming no id - an unreviewed version, a - /// foreign identity form, or a positional id space - land in + /// foreign identity form, or a positional id space - are reported in /// [`unresolved`](ResolvedVerdicts::unresolved). #[must_use] pub(crate) fn resolve(&self, ontology: &IdSlice) -> ResolvedVerdicts<'_> where O: OntologyIdentity + Eq + Hash, { - // Validation rejected duplicate versioned URLs, so every - // verdict owns its map entry. + // Validation rejected duplicate versioned URLs, and every + // verdict therefore owns its map entry. let targets: HashMap = self .type_verdicts .iter() diff --git a/libs/@local/graph/atlas/src/salt/projector/verdict/tests.rs b/libs/@local/graph/atlas/src/salt/projector/verdict/tests.rs index 67bdc425060..a1b99d2f3b4 100644 --- a/libs/@local/graph/atlas/src/salt/projector/verdict/tests.rs +++ b/libs/@local/graph/atlas/src/salt/projector/verdict/tests.rs @@ -32,10 +32,19 @@ fn table_entry(base: &str, version: u32) -> ArchivedOntologyTypeUuid { )) } +/// Base URL of the `delivers` link type, reviewed as proximal in the fixtures. const DELIVERS: &str = "https://hash.ai/@h/types/entity-type/delivers/"; +/// Base URL of the Block Protocol `link` type. +/// +/// It sorts before the HASH types by relation string. const LINK: &str = "https://blockprotocol.org/@blockprotocol/types/entity-type/link/"; +/// Base URL of the `yields` link type, reviewed as coincident in the fixtures. const YIELDS: &str = "https://hash.ai/@h/types/entity-type/yields/"; +/// Parses a contract-conforming document into three type verdicts and the source digest. +/// +/// A contract-conforming document parses into three type verdicts with their classes, relations, +/// reviewer and versioned URLs, no pair verdicts, and the recorded source digest. #[test] fn shipped_shape_parses() { let json = document( @@ -78,6 +87,7 @@ fn shipped_shape_parses() { ); } +/// A document with a different schema tag fails with `Schema` carrying the tag found. #[test] fn foreign_schema_is_rejected() { let json = document(&verdict("overlay", LINK, 1), "") @@ -89,6 +99,10 @@ fn foreign_schema_is_rejected() { ); } +/// Unknown fields and an unknown class fail at the JSON layer with `Json`. +/// +/// An unknown field at the document or row level and an unknown class (`excluded`) fail at the JSON +/// layer with `Json`. #[test] fn unknown_fields_and_classes_are_rejected() { // A field this reader does not know is a schema evolution nobody taught it, at the document and @@ -109,7 +123,7 @@ fn unknown_fields_and_classes_are_rejected() { Err(InvalidReviewedVerdicts::Json(_)), ); - // The exporter omits `excluded` reviews, so they are never a class. + // The exporter omits `excluded` reviews: `excluded` is never a class. let excluded = document(&verdict("excluded", LINK, 1), ""); assert_matches!( ReviewedVerdicts::from_slice(excluded.as_bytes()), @@ -117,6 +131,7 @@ fn unknown_fields_and_classes_are_rejected() { ); } +/// An unversioned URL and an uppercase digest fail at the JSON layer with `Json`. #[test] fn malformed_versioned_urls_and_digests_are_rejected() { let unversioned = document( @@ -135,10 +150,13 @@ fn malformed_versioned_urls_and_digests_are_rejected() { ); } +/// Unordered or repeated type verdicts fail with `UnorderedTypeVerdicts` naming the index. +/// +/// Type verdicts out of ascending relation order, or repeating a relation, fail with +/// `UnorderedTypeVerdicts` naming the offending index. #[test] fn unordered_and_duplicate_type_verdicts_are_rejected() { - // LINK sorts before DELIVERS by relation string, so this order is - // descending. + // LINK sorts before DELIVERS by relation string: this order is descending. let unordered = document( &[ verdict("proximal", DELIVERS, 1), @@ -182,6 +200,10 @@ fn repeated_versioned_url_is_rejected() { ); } +/// Empty type and pair verdict fields fail with the variant naming the field. +/// +/// An empty reviewer or relation fails with `EmptyTypeVerdictField` naming the field, and an empty +/// pair-verdict field with `EmptyPairVerdictField`. #[test] fn empty_identity_fields_are_rejected() { let empty_reviewer = document( @@ -221,6 +243,7 @@ fn empty_identity_fields_are_rejected() { ); } +/// Pair verdicts out of order fail with `UnorderedPairVerdicts` naming the offending index. #[test] fn unordered_pair_verdicts_are_rejected() { let json = document( @@ -237,6 +260,9 @@ fn unordered_pair_verdicts_are_rejected() { ); } +/// Only the verdict whose versioned URL the ontology table holds resolves. +/// +/// A verdict at another version of a held type and one for an absent type stay unresolved. #[test] fn resolution_is_version_precise() { let json = document( @@ -249,8 +275,8 @@ fn resolution_is_version_precise() { ); let verdicts = ReviewedVerdicts::from_slice(json.as_bytes()).expect("the fixture conforms"); - // The snapshot holds two versions of the reviewed type; only the - // reviewed version resolves. LINK v9 is absent entirely. + // The snapshot holds two versions of the reviewed type, and only the reviewed version + // resolves. LINK v9 is absent entirely. let ontology = [ table_entry(YIELDS, 1), table_entry(DELIVERS, 1), @@ -270,10 +296,14 @@ fn resolution_is_version_precise() { assert_eq!(unresolved[0].relation, format!("hash:{LINK}")); } +/// Reorders verdicts by ontology row against an opposite document order. +/// +/// Resolution reorders verdicts by ontology row even when the document and the table order them +/// oppositely. #[test] fn resolved_verdicts_ascend_by_row() { - // Document order is by relation string; the table reverses it, so - // resolution must re-order by row. + // Document order is by relation string and the table reverses it: resolution must + // re-order by row. let json = document( &[ verdict("overlay", LINK, 1), @@ -318,9 +348,8 @@ fn resolved_verdicts_ascend_by_row() { #[test] fn verdicts_without_a_store_identity_are_carried_as_evidence() { - // Foreign-corpus types (e.g. wikidata) record no versioned URL; - // the verdict parses, never resolves, and never conflicts with - // another identity-free verdict. + // Foreign-corpus types (e.g. wikidata) record no versioned URL. The verdict parses, never + // resolves, and never conflicts with another identity-free verdict. let json = document( &[ verdict("proximal", DELIVERS, 1), @@ -347,6 +376,7 @@ fn verdicts_without_a_store_identity_are_carried_as_evidence() { assert_eq!(unresolved, ["wikidata:P50", "wikidata:P69"]); } +/// An empty document parses and resolves to no resolved and no unresolved verdicts. #[test] fn empty_document_resolves_to_nothing() { let json = document("", ""); diff --git a/libs/@local/graph/atlas/src/salt/quality/clump.rs b/libs/@local/graph/atlas/src/salt/quality/clump.rs index 2042b21aacb..46bdf235f9a 100644 --- a/libs/@local/graph/atlas/src/salt/quality/clump.rs +++ b/libs/@local/graph/atlas/src/salt/quality/clump.rs @@ -1,27 +1,27 @@ //! Near-duplicate clumps over the 512-component neighbour table. //! -//! A clump is a connected component of the k-nearest-neighbour graph restricted to edges at cosine -//! distance at most ε: rows whose representations chain through near-identical neighbours share a -//! component label. Collapsing neighbour orderings onto clump ids relabels recall at component -//! granularity - a triage diagnostic and nothing stronger: ε chains can reach arbitrary diameter, -//! so a shared label certifies neither component compactness nor within-component placement, and -//! the collapsed readings never affect admission. +//! A [`Clumps`] label identifies a connected component formed by stored neighbour edges at cosine +//! distance at most ε. Collapsing recall onto these component labels relaxes row identity for +//! diagnostic comparison. Single-linkage chains can span distances much larger than ε. A shared +//! label certifies neither component compactness nor within-component placement, and collapsed +//! readings never affect admission. //! -//! The kNN restriction reads a subgraph of the full ε graph: a group larger than the table's k -//! connects only through chains of stored edges, so a true ε-ball component can split but never -//! spuriously merge - a split clump makes clump-granularity readings stricter, never looser. +//! When stored distances agree with the full ε graph, restricting to stored k-NN edges yields a +//! subgraph of that graph. Removing edges can split a connected component but cannot join different +//! components. Therefore the stored-edge labels refine the full-graph labels. For fixed +//! neighbourhood lists, this refinement can only reduce their collapsed overlap. A checked table +//! validates distance ranges and structure, not correspondence to the embedding matrix. //! -//! ε is a calibrated configuration value: [`DEFAULT_EPSILON`] carries the corpus evidence it was -//! pinned on, and the grouping is judged against measured corpus structure (group count, coverage, -//! size distribution) and against the flagged subgroups it is expected to resolve. The -//! [`calibration`](super::report::calibration) instrument re-derives the readings against any -//! published k-NN table. +//! [`DEFAULT_EPSILON`] records development-corpus calibration readings and their generation +//! dependence. [`calibration`](super::report::calibration) measures grouping shape +//! over a published k-NN table. Compare that shape and the subgroup readings before choosing a +//! threshold for another generation. //! //! [`ClumpAggregate`] is the collapsed counterpart of the plain recall reading: both neighbour -//! lists relabel onto clump ids and overlap as multisets, so same-component siblings satisfy each -//! other under the relabeling while a clump the map underrepresents earns only the credit it shows. -//! Under singleton labels the multiset overlap is exactly the shared-row count, so clump recall is -//! always at least plain recall and equals it when nothing clumps. +//! lists relabel onto clump ids and overlap as multisets. Each shared row still matches its own +//! label after relabeling, and additional same-label matches may appear. Therefore clump recall is +//! always at least plain recall over those same lists, with equality for singleton labels. A clump +//! the map underrepresents earns only the credit its observed multiplicity supplies. #![expect( clippy::min_ident_chars, reason = "k is the canonical neighbourhood-size name across the metric literature" @@ -34,34 +34,30 @@ use hashql_core::id::{Id, IdUnionFind, IdVec}; use super::super::knn::table::KnnView; use crate::math::UnitFraction; -/// The default clump threshold, as cosine distance over the 512-component representation. +/// The default cosine-distance threshold over the 512-component representation. /// -/// The value is calibrated, not derived, and the calibration is per generation rather than -/// universal. On two fits of the development corpus (985,932 rows, 30 stored neighbours per row) it -/// sits on a plateau: `2ea9cb45…` reads 131,773, 131,760, and 131,147 multi-row groups at ε = -/// 0.0012, 0.002, and 0.0028 - a 0.48% spread - while coverage grows from 48.7% to 60.9%, and -/// `c1d00be7…` reproduces every one of those readings within 0.02%. At 0.002, cosine similarity -/// 0.998, the first reads 131,760 groups covering 55.9% of the corpus at mean size 4.18. Below the -/// plateau exact duplicates stay split; above roughly 0.0045 the components percolate, the group -/// count falling while sizes grow without bound. +/// The value 0.002 corresponds to cosine similarity 0.998. Calibration depends on the generation, +/// including its stored neighbour graph. /// -/// The plateau belongs to those fits and not to the construction: generation `bfc67cbc…` has none. -/// It reads 85,794, 91,162, and 95,179 groups over the same three thresholds - a 10.9% rise across -/// the interval - with the curve 34.9%, 30.8%, and 27.4% below the others at those three thresholds -/// and coverage 39.0% to 53.3%, so there ε = 0.002 sits on a slope and neighbouring thresholds do -/// not produce the same grouping structure. A clump-granularity reading is therefore comparable -/// within one generation and not across generations; `report clumps` re-reads the curve for a new -/// fit in seconds. +/// Recorded sweeps over development-corpus fits with 985,932 rows and 30 stored neighbours per row +/// motivate this value. One fit (generation prefix `2ea9cb45`) records 131,773, 131,760 and 131,147 +/// multi-row groups at ε = 0.0012, 0.002 and 0.0028: about 0.48% variation while coverage grows +/// from 48.7% to 60.9%. At 0.002 it records 55.9% coverage and mean group size 4.18. A second fit +/// (generation prefix `c1d00be7`) has recorded readings differing by less than 0.02%. /// -/// An earlier audit structure (165K groups, 66% coverage, mean size near 4) came from a different -/// grouping construction and is not reproducible by ε-connected components over the k-NN table at -/// any threshold; it anchors the scale of this value, not the value itself. +/// A third fit (generation prefix `bfc67cbc`) records 85,794, 91,162 and 95,179 groups over those +/// thresholds, a 10.9% rise, with coverage from 39.0% to 53.3%. Its group counts are respectively +/// 34.9%, 30.8% and 27.4% below the first fit's. This generation has no comparable plateau over +/// that interval. A fixed ε does not establish comparable component structure across generations. +/// Use [`calibration`](super::report::calibration) to measure the grouping curve for a new table, +/// and compare subgroup readings before adopting the threshold. pub(crate) const DEFAULT_EPSILON: f32 = 0.002; -/// A dense clump labelling of the node-row domain. +/// Connected-component labels for rows joined by stored edges within a distance threshold. /// -/// Every row carries a clump id in `0..clumps`; ids are assigned in ascending order of each clump's -/// first row, so equal tables and thresholds label equally. A singleton row is its own clump. +/// Every row carries a dense clump id in `0..clumps`, ordered by each component's first row. A +/// singleton row is its own clump. For equal tables and thresholds, the partition and labels are +/// deterministic. #[derive(Debug, Clone)] pub(crate) struct Clumps { labels: IdVec, @@ -77,9 +73,15 @@ where { /// Groups the table's rows at the `epsilon` distance threshold. /// - /// An edge joins two rows when either row stores the other at cosine distance at most - /// `epsilon`; exact-duplicate embeddings (distance 0) group at every threshold. A non-finite - /// `epsilon` admits no edges. + /// An edge joins two rows when either row stores the other at distance at most `epsilon`. A + /// stored zero-distance edge joins at every non-negative threshold. NaN and negative thresholds + /// admit no edges, while positive infinity admits every stored edge. The row count must fit + /// u32, and the table must support row access over its complete domain. + /// + /// # Complexity + /// + /// For n rows and e stored edges, grouping takes O((n + e) · α(n)) time with union-find and + /// O(n) additional storage. Labels retain O(n) storage. pub(crate) fn from_knn(table: &KnnView<'_, N>, epsilon: f32) -> Self { let rows = table.rows(); let mut components = IdUnionFind::::new(rows); @@ -94,8 +96,7 @@ where } } - // Dense relabelling by first row: deterministic in the - // partition alone. + // first-row relabeling makes labels depend on the partition, not the union-find roots let mut labels = IdVec::from_elem(0_u32, rows); let mut label_of = IdVec::from_elem(u32::MAX, rows); let mut clumps = 0_u32; @@ -114,10 +115,14 @@ where Self::from_dense_labels(labels, clumps as usize, epsilon) } - /// Wraps a labelling that is already dense in first-row order. + /// Computes grouping counts from labels already dense in first-row order. + /// + /// Every label must lie in `0..count`, first appearances must ascend, and each component's row + /// count must fit u32. + /// + /// # Panics /// - /// The caller promises every label lies in `0..count` and that labels first appear in ascending - /// order; the fixture paths that use this assert both. + /// Panics when a label lies outside `0..count`. fn from_dense_labels(labels: IdVec, count: usize, epsilon: f32) -> Self { let mut sizes = vec![0_u32; count]; for &clump in &labels { @@ -140,7 +145,7 @@ where } } - /// Wraps a hand-built labelling for kernel and report tests. + /// Validates first-row label order and computes grouping counts for a fixture. /// /// # Panics /// @@ -199,8 +204,6 @@ where } /// Returns the count of clumps holding at least two rows. - /// - /// The calibration reading compared against measured corpus structure. #[inline] #[must_use] pub(crate) const fn groups(&self) -> usize { @@ -208,8 +211,6 @@ where } /// Returns the count of rows inside multi-row clumps. - /// - /// The coverage side of the calibration reading. #[inline] #[must_use] pub(crate) const fn grouped_rows(&self) -> usize { @@ -219,12 +220,14 @@ where /// Accumulated clump-granularity neighbourhood overlap. /// -/// One aggregate fixes a neighbourhood size at construction; queries accumulate through -/// [`observe`](Self::observe) and the recall reading divides the totals on demand. A query's -/// overlap is the multiset intersection of its two neighbourhoods' clump ids: each reference -/// neighbour is matched by a distinct map neighbour from the same clump, so siblings reshuffling -/// inside one clump keep full credit while a clump the map shows fewer members of earns exactly the -/// members shown. +/// One aggregate fixes a neighbourhood size k. A query's overlap is Σ min(r(c), m(c)) over +/// component labels c, where r(c) and m(c) count that label's occurrences in the reference and map +/// neighbourhoods. Each reference neighbour matches a distinct map neighbour from the same clump. +/// Reshuffling siblings keeps full credit, while a clump the map underrepresents earns only its +/// observed multiplicity. +/// +/// Observations and merges must keep the query count and query-times-k product within usize and the +/// matched total within u64. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) struct ClumpAggregate { k: usize, @@ -246,8 +249,8 @@ impl ClumpAggregate { /// Accumulates one query's pair of collapsed neighbourhoods. /// - /// Each slice holds the clump ids of the query's `k` nearest points in its space; both are - /// sorted in place, since the overlap is order-free. + /// Each slice holds the clump ids of the query's k nearest points in its space. Both slices are + /// sorted in place because overlap ignores order. /// /// # Panics /// diff --git a/libs/@local/graph/atlas/src/salt/quality/error.rs b/libs/@local/graph/atlas/src/salt/quality/error.rs index 1e7794e1751..6a203df7b56 100644 --- a/libs/@local/graph/atlas/src/salt/quality/error.rs +++ b/libs/@local/graph/atlas/src/salt/quality/error.rs @@ -1,5 +1,3 @@ -//! The quality runner's error. - use core::{error::Error, fmt}; use super::probe::{DeliveryError, ProbeError}; @@ -11,7 +9,7 @@ use crate::{ salt::{fit::prepare::identity::InvalidIdentityFile, knn::artifact::InvalidKnnFile}, }; -/// The quality run could not produce a report. +/// An artifact, probe or type-delivery failure that prevents a quality report. #[derive(Debug)] pub(crate) enum QualityRunError { /// Opening the k-NN artifact failed. diff --git a/libs/@local/graph/atlas/src/salt/quality/metric.rs b/libs/@local/graph/atlas/src/salt/quality/metric.rs index 3b3a97a7e86..1c2532817d2 100644 --- a/libs/@local/graph/atlas/src/salt/quality/metric.rs +++ b/libs/@local/graph/atlas/src/salt/quality/metric.rs @@ -1,29 +1,36 @@ //! Rank-based fidelity kernels over a shared comparison universe. //! -//! Every kernel here consumes neighbour orderings, never coordinates: a query's view of the -//! universe is a permutation of `0..m` listing the comparison points from nearest to farthest, one -//! permutation per space. Which spaces produced the orderings - the 2D map against the -//! 512-component representation, either against exact canonical distances - is the orchestration's -//! concern, so one set of kernels serves every space pair the suite reports. +//! A query's view of a comparison universe is a permutation of `0..m` listing its points +//! nearest-first, one permutation per space. The neighbourhood kernels consume these orderings or +//! the opposite ranks of their nearest points. Triplet aggregation consumes order-agreement +//! verdicts. The same kernels serve every space pair independently of its coordinate +//! representation. //! -//! For a query with reference ordering `R` and map ordering `M`, the per-query quantities at -//! neighbourhood size `k` are: +//! For reference ordering R and map ordering M over m points, let Xₖ be the first k points of X, +//! with integer 1 ≤ k ≤ ⌊m/2⌋. Use one-based ranks rᵣ and rₘ and a horizon h with k ≤ h ≤ m. The +//! per-query quantities are: //! -//! - shared neighbours: `|R_k intersect M_k|`, where `X_k` is the ordering's first `k` points; -//! recall at `k` is the shared count over `k`. -//! - trust penalty: `sum (rank_R(j) - k)` over the map neighbours `j in M_k \ R_k`, with ranks -//! 1-based - how far into the reference ordering the map's false neighbours live. -//! - continuity penalty: the mirror image, `sum (rank_M(j) - k)` over `j in R_k \ M_k` - how far -//! the map banishes true neighbours. -//! - intrusions and extrusions: the false neighbours whose rank excess passes a configured horizon, -//! separating foreign points from near-boundary reshuffling among close ranks. +//! - Shared neighbours: |Rₖ ∩ Mₖ|. Recall divides this count by k. +//! - Trust penalty: Σ(rᵣ(j) − k) over j ∈ Mₖ ∖ Rₖ, measuring how far false map neighbours rank +//! beyond the reference neighbourhood. +//! - Continuity penalty: Σ(rₘ(j) − k) over j ∈ Rₖ ∖ Mₖ, measuring how far missing reference +//! neighbours rank beyond the map neighbourhood. +//! - Intrusions: map neighbours with rᵣ(j) > h. Extrusions: reference neighbours with rₘ(j) > h. +//! Each rate divides its count by k. //! -//! [`NeighbourhoodAggregate`] accumulates these over queries and normalizes trustworthiness and -//! continuity onto `[0, 1]` (1 is a perfect map, 0 the worst permutation) by the worst-case penalty -//! `q · k · (2m - 3k + 1) / 2`: each of `q` queries can misplace at most `k` points, and their rank -//! excesses are largest when the false neighbours occupy the ordering's final `k` positions. Over a -//! universe of `m = n - 1` non-self points this reduces to the Venna-Kaski normalization `2 / (n k -//! (2n - 3k - 1))`. +//! Distinct opposite ranks maximize either penalty at the final k positions of the universe. Their +//! excesses sum to W = k · (2m − 3k + 1) / 2. The bound k ≤ ⌊m/2⌋ makes those positions disjoint +//! from the reference neighbourhood. Therefore W is the attainable worst per-query penalty, and +//! [`NeighbourhoodAggregate`] computes trustworthiness and continuity as 1 − P/(q · W) for q > 0 +//! queries with total penalty P. The readings lie in [0, 1], with one at zero penalty and zero at +//! the worst case. With q = n queries and m = n − 1 comparisons per query, the coefficient becomes +//! the Venna-Kaski normalization 2 / (n · k · (2n − 3k − 1)). +//! +//! Integer counts preserve exact aggregation order-independence while their carriers hold the +//! totals. [`NeighbourhoodAggregate::supports`] checks the count and normalization capacity for a +//! proposed total query count, but observation does not enforce that check. Conversion and division +//! round to f64. For the same valid rank inputs and supported totals, merge order leaves the +//! readings unchanged. #![expect( clippy::cast_precision_loss, clippy::cast_possible_truncation, @@ -41,8 +48,8 @@ use crate::math::UnitFraction; /// Reusable inverse-rank buffers for one comparison universe. /// -/// Sized once per universe; [`NeighbourhoodAggregate::observe`] fills both buffers per query, so a -/// suite pass allocates two `u32` rows regardless of query count. +/// Allocate for one universe and reuse across observations. Each buffer holds one u32 rank per +/// comparison point. pub(crate) struct RankScratch { reference_rank: Vec, map_rank: Vec, @@ -62,8 +69,13 @@ impl RankScratch { /// Accumulated neighbourhood agreement between two orderings. /// /// One aggregate fixes a universe size, a neighbourhood size, and an intrusion horizon at -/// construction; queries accumulate through [`observe`](Self::observe) and the metric readings +/// construction. Queries accumulate through [`observe`](Self::observe) and the metric readings /// divide the totals on demand. An aggregate over a single query is that query's own reading. +/// +/// Observations must describe valid permutations with u32 rank positions. Before accumulating or +/// merging, ensure [`supports`](Self::supports) accepts the resulting total query count. These are +/// correctness requirements, not checks performed by observation. The shape constructor alone does +/// not establish arithmetic capacity. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct NeighbourhoodAggregate { universe: usize, @@ -80,12 +92,12 @@ pub(crate) struct NeighbourhoodAggregate { impl NeighbourhoodAggregate { /// Creates an empty aggregate. /// - /// `universe` is the comparison-point count every observed ordering permutes; `k` the - /// neighbourhood size; `horizon` the 1-based rank beyond which a false neighbour counts as an - /// intrusion or extrusion rather than a reshuffle. + /// `universe` is the comparison-point count every observed ordering permutes. `k` is the + /// neighbourhood size, and `horizon` is the one-based rank beyond which a false neighbour + /// counts as an intrusion or extrusion. /// - /// Returns [`None`] unless `k ≤ universe / 2` (the trustworthiness normalizer is positive on - /// this domain) and `k ≤ horizon ≤ universe`. + /// Returns [`None`] unless k ≤ ⌊universe/2⌋ and k ≤ horizon ≤ universe. On this domain the k + /// largest rank excesses are attainable together and the normalizer is positive. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -117,8 +129,9 @@ impl NeighbourhoodAggregate { /// Creates an empty aggregate at the clamped intrusion horizon. /// - /// The mathematical horizon is `min(factor · k, universe)`. When the product exceeds - /// `usize`, it exceeds the universe as well, so the clamp is the universe. + /// The mathematical horizon is min(factor · k, universe). Saturating the product before + /// clamping preserves that value: a product exceeding usize also exceeds the universe. Returns + /// [`None`] when [`new`](Self::new) rejects the neighbourhood shape. #[must_use] pub(crate) const fn clamped( universe: usize, @@ -135,17 +148,16 @@ impl NeighbourhoodAggregate { Self::new(universe, k, horizon) } - /// Proves the aggregate's integer carriers hold `observations` queries' worst-case penalties. + /// Checks arithmetic capacity for `observations` total queries. + /// + /// For universe m, neighbourhood k and total query count q, the worst per-query penalty is W = + /// k · (2m − 3k + 1)/2. This check verifies that 2m, the unhalved penalty product, q · k and q + /// · W all fit usize. On 32-bit and 64-bit targets, q · W also bounds the u64 penalty totals. + /// Therefore these products and totals cannot wrap for an accepted load of valid observations. /// - /// Take `m` as the aggregate's universe and `k` its neighbourhood size. Penalties accumulate - /// into `u64` totals and read back through `usize` products. The recomputed products are the - /// doubled universe `2m` and the worst-case per-query penalty product `k·(2m − 3k + 1)` - /// before its exact halving into `worst`, then the pair count `q·k` and the normalizer - /// `q·worst` at `q = observations`. Each is checked with the carrier's own width, so an - /// accepted load cannot wrap on either target width. The checked normalizer also bounds the - /// `u64` penalty totals, and `usize` never exceeds `u64`. The worst case assumes each - /// observation ranks `k` distinct reference positions, which every observed ordering - /// satisfies: an ordering is a permutation, so one query's `k` opposite ranks are distinct. + /// The bound assumes k distinct opposite ranks from valid permutations, including through + /// [`observe_ranks`](Self::observe_ranks). It validates no observation and reserves no + /// capacity. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -176,12 +188,15 @@ impl NeighbourhoodAggregate { /// Accumulates one query's pair of orderings. /// /// Each slice lists the universe's points nearest-first in its space and must be a permutation - /// of `0..universe`; the query itself is not a universe point, so neither ordering lists it. + /// of `0..universe`. The query itself is not a universe point. `scratch` must have been created + /// for this same universe, and the resulting total query count must satisfy + /// [`supports`](Self::supports). /// /// # Panics /// - /// This panics when either ordering's length differs from the universe or names a point outside - /// it. + /// Panics when an ordering's length differs from the universe, a point lies outside the scratch + /// storage, or an accumulated rank lies outside the universe. With correctly sized scratch, + /// every out-of-universe point fails the storage bound. pub(crate) fn observe( &mut self, by_reference: &[u32], @@ -216,11 +231,14 @@ impl NeighbourhoodAggregate { /// Accumulates one query from each neighbourhood's opposite ranks. /// - /// `reference_ranks_of_map_neighbours` holds the reference-space ranks of the query's `k` - /// nearest map points, nearest-first, and `map_ranks_of_reference_neighbours` the mirror image; - /// ranks are 0-based positions in the universe. The metrics are functions of exactly these `2k` - /// ranks, so a caller that computes ranks by counting - without materializing whole orderings - - /// observes through here and [`observe`](Self::observe) reduces to it. + /// `reference_ranks_of_map_neighbours` holds the reference-space ranks of the query's k nearest + /// map points, and `map_ranks_of_reference_neighbours` holds the map-space ranks of its k + /// nearest reference points. Ranks are zero-based positions in the universe. Each slice must + /// contain distinct ranks derived from the same pair of valid orderings, and the resulting + /// total query count must satisfy [`supports`](Self::supports). + /// + /// The metrics depend on exactly these 2k ranks, irrespective of their order within each slice. + /// Use this form for ranks computed by counting, without materializing whole permutations. /// /// # Panics /// @@ -247,14 +265,20 @@ impl NeighbourhoodAggregate { ); } - /// Folds one query's opposite-rank pairs into the totals. + /// Adds one query's opposite-rank counts and penalties to the totals. + /// + /// Each iterator must supply k distinct ranks from the same pair of orderings, within the + /// aggregate's supported query load. + /// + /// # Panics + /// + /// Panics when a rank is outside the universe. fn accumulate( &mut self, reference_ranks_of_map_neighbours: impl Iterator, map_ranks_of_reference_neighbours: impl Iterator, ) { - // Positions are 0-based; the 1-based excess (rank - k) of a - // position p is (p - k) + 1. + // for zero-based position p, the one-based rank excess is p − k + 1 for rank in reference_ranks_of_map_neighbours { let reference_position = rank as usize; assert!( @@ -295,9 +319,13 @@ impl NeighbourhoodAggregate { /// Folds another aggregate's observations into this one. /// - /// Merging aggregates observed over disjoint query sets equals one aggregate observing their - /// union, so per-query aggregates roll up into per-subgroup readings and a reading over every - /// query without revisiting orderings. + /// The resulting total query count must satisfy [`supports`](Self::supports). + /// + /// # Properties + /// + /// For valid observations with supported totals and equal shapes, merging aggregates equals + /// observing the concatenation of their queries. A repeated query counts again. Disjoint query + /// sets yield the reading over their union. /// /// # Panics /// @@ -392,6 +420,11 @@ impl NeighbourhoodAggregate { Some((self.queries * self.k) as f64) } + /// Normalizes a supported total penalty, with one at zero penalty. + /// + /// The worst case per query is k · (2m − 3k + 1) / 2, the penalty of a neighbourhood whose k + /// members all rank last among the m comparisons. An empty aggregate reads one. `penalty` must + /// not exceed the supported total normalizer. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -403,7 +436,8 @@ impl NeighbourhoodAggregate { return UnitFraction::ONE; } - // The constructor bounds k ≤ universe / 2, so 2m - 3k + 1 > 0. + // construction gives 2m − 3k + 1 > 0. The supports precondition establishes that the + // products fit let worst_per_query = self.k * (2 * self.universe - 3 * self.k + 1) / 2; UnitFraction::new_unchecked(1.0 - penalty as f64 / (self.queries * worst_per_query) as f64) } @@ -411,11 +445,13 @@ impl NeighbourhoodAggregate { /// Accumulated order agreement over sampled triplets. /// -/// A triplet fixes an anchor and two comparison points, and the map preserves that triplet when it -/// orders the points' distances from the anchor as the reference space does, both spaces compared -/// under the shared `(distance, row)` total order - a pair coincident in both spaces is therefore -/// preserved, and one coincident in exactly one space is not. An aggregate over one anchor's pairs -/// is that anchor's own reading. +/// A triplet fixes an anchor and two distinct comparison points. The probe marks it preserved when +/// both spaces order those points alike by `(distance, row)`. Equal distances in both spaces agree +/// by row order. Equal distances in only one space agree when its row tiebreak matches the other +/// space's distance order. +/// +/// An aggregate over one anchor's pairs is that anchor's own reading. Observations and merges must +/// keep the triplet total within u64. #[derive(Debug, Copy, Clone, PartialEq, Eq, Default)] pub(crate) struct TripletAggregate { triplets: u64, @@ -460,8 +496,9 @@ impl TripletAggregate { return UnitFraction::ONE; } - // Preservation only ever counts a subset of the observed triplets, - // so the ratio lies ∈ [0, 1] by construction. + // Preserved triplets are a subset of observations. Within the u64 count capacity, + // conversion to f64 is monotone and the denominator is positive. Therefore the rounded + // ratio remains in [0, 1]. UnitFraction::new_unchecked(self.preserved as f64 / self.triplets as f64) } } diff --git a/libs/@local/graph/atlas/src/salt/quality/mod.rs b/libs/@local/graph/atlas/src/salt/quality/mod.rs index 31f9b981500..c9a82d6155b 100644 --- a/libs/@local/graph/atlas/src/salt/quality/mod.rs +++ b/libs/@local/graph/atlas/src/salt/quality/mod.rs @@ -1,25 +1,25 @@ -//! Map-fidelity metrics and release thresholds for the quality suite. +//! Map-fidelity measurements and admission thresholds. //! -//! The suite judges a projected map by how well small neighbourhoods survive the trip from the -//! canonical embedding space to 2D. It compares neighbour rankings between three spaces - the 2D -//! map, the 512-component training representation, and exact 3072-component canonical distances -//! over bounded probe sets - and reports recall, trustworthiness, continuity, intrusion rates, -//! triplet agreement, and density distortion, each for the whole probe and per subgroup. The -//! 512-versus-3072 comparison is the representation baseline the suite judges the map readings -//! against. +//! The suite judges neighbourhood preservation between the 2D map, the 512-component training +//! representation and the 3072-component canonical space. Map-versus-representation rankings cover +//! every non-anchor row. Comparisons involving canonical embeddings use a bounded shared sample. +//! The representation-versus-canonical reading supplies a baseline for the map's canonical reading +//! at that sampled scale. //! -//! [`metric`] holds the rank-based kernels: pure functions over neighbour orderings and distances, -//! independent of which spaces produced them. [`clump`] groups near-duplicate rows over the -//! 512-component neighbour table, which lets a reading collapse orderings onto clump ids and -//! separate placement error from reshuffling among near-identical siblings. [`probe`] orchestrates -//! the measurement from anchor and comparison sampling through canonical embeddings, three-space -//! rankings, and per-anchor reading grids. [`report`] renders the readings under configured -//! thresholds: whole-probe and per-subgroup metric rows, the subgroup degradation flags, and the -//! release verdict. +//! [`metric`] holds rank-based recall, trustworthiness, continuity, intrusion/extrusion and +//! triplet-agreement kernels. [`clump`] groups rows through near-duplicate edges in the stored +//! neighbour table. Collapsing recall onto these component labels measures overlap with row +//! identity relaxed, without certifying compactness or within-component placement. [`probe`] +//! samples anchors and comparisons, fetches canonical embeddings and produces per-anchor readings. +//! [`report`] aggregates these into whole-probe measurements, per-type neighbourhood rows and +//! subgroup flags, with density distortion from neighbourhood radii and a threshold verdict. +//! [`runner`] assesses a published generation against a dataset. //! -//! Every metric here is a function of rankings over a shared comparison universe. Probe-scoped -//! readings are exact over their probe sets and estimates of the corpus-wide quantity; the report -//! carries the probe sizes so a reading is never mistaken for a corpus-complete measurement. +//! Rankings use computed distances over their stated universe. Aggregates retain anchor-sampling +//! uncertainty even where ranking coverage is exact, and a sampled k-neighbourhood measures a +//! coarser scale than the same k over the corpus. Reports retain both universe sizes. Admission +//! checks the observed map-versus-representation metrics, density spread and sampled triplet +//! agreement. Canonical comparisons and subgroup flags remain report-only. pub(crate) mod clump; pub(crate) mod error; @@ -31,14 +31,11 @@ pub(crate) mod runner; #[cfg(test)] mod tests; -/// One quality metric of the admission probe's six-threshold set. -/// -/// Each variant names one control of the release battery, so a report's verdict and an observer's -/// reading identify a metric the same way. +/// A metric checked by the admission thresholds. #[derive(Debug, Copy, Clone, PartialEq, Eq)] #[repr(u8)] pub enum QualityMetric { - /// The neighbour backend's measured recall. + /// Shared map-versus-representation neighbourhoods. Recall, /// Neighbourhood trustworthiness. Trustworthiness, @@ -53,26 +50,22 @@ pub enum QualityMetric { } impl QualityMetric { - /// Every metric of the battery, in the order a report's controls carry them. - /// - /// An observer rendering the battery needs the set before the probe reports any of it, and the - /// readings arrive in one burst at the end of the probe, so this list carries the order rather - /// than arrival. + /// Every admission metric, in report-control order. #[expect( clippy::cast_possible_truncation, reason = "the index runs over the variant count, an order of magnitude inside u8" )] pub const ALL: [Self; core::mem::variant_count::()] = - // SAFETY: every variant is a unit variant of a `repr(u8)` enum. Its discriminants are then - // exactly the range `0..variant_count`, and `from_fn` calls the closure once per index of - // that range. + // SAFETY: a fieldless `repr(u8)` enum has u8 size and requires a valid discriminant. These + // six variants have implicit consecutive discriminants starting at zero. `from_fn` + // supplies exactly those indices, and each fits in u8. Therefore every transmute produces + // a valid variant. core::array::from_fn(const |index| unsafe { core::mem::transmute(index as u8) }); - /// The metric's name, in the vocabulary its own threshold key uses. + /// Returns the metric noun used in its threshold key. /// - /// Each name is the noun of the report's own key, so `minimum_recall` is `recall` and - /// `maximum_density_spread` is `density spread`. A rendered reading and the threshold that - /// moves it therefore name one control. + /// For example, `minimum_recall` uses `recall` and `maximum_density_spread` uses `density + /// spread`. #[must_use] pub const fn label(self) -> &'static str { match self { diff --git a/libs/@local/graph/atlas/src/salt/quality/probe/error.rs b/libs/@local/graph/atlas/src/salt/quality/probe/error.rs index f414612adb0..a0da4305de8 100644 --- a/libs/@local/graph/atlas/src/salt/quality/probe/error.rs +++ b/libs/@local/graph/atlas/src/salt/quality/probe/error.rs @@ -1,8 +1,6 @@ -//! Design, domain, and delivery failures that stop the probe. - use core::{error::Error, fmt, num::NonZero}; -/// The probe could not run. +/// A design or canonical-delivery failure that prevents probe completion. #[derive(Debug)] pub(crate) enum ProbeError { /// The corpus cannot host disjoint anchor and comparison samples. @@ -83,7 +81,7 @@ impl Error for ProbeError { } } -/// An unordered id-keyed delivery did not match its requests. +/// A failed or mismatched id-keyed delivery stream. #[derive(Debug)] pub(crate) enum DeliveryError { /// The stream failed. @@ -92,9 +90,8 @@ pub(crate) enum DeliveryError { Unrequested, /// The stream delivered one requested id twice. /// - /// A repeat is never a harmless echo. Its payload would replace one the reading has already - /// accepted, and nothing at this seam can tell a duplicate of the same bytes from a second, - /// different answer arriving under one id. + /// Every requested id permits one delivery. Repetition fails regardless of payload equality, + /// without selecting one answer over another. Repeated, /// The stream ended before covering every requested id. Missing { requested: usize, delivered: usize }, diff --git a/libs/@local/graph/atlas/src/salt/quality/probe/mod.rs b/libs/@local/graph/atlas/src/salt/quality/probe/mod.rs index 9a94b2a0625..da60dd16f3d 100644 --- a/libs/@local/graph/atlas/src/salt/quality/probe/mod.rs +++ b/libs/@local/graph/atlas/src/salt/quality/probe/mod.rs @@ -5,25 +5,31 @@ //! feed one kernel set: //! //! - The corpus pass ranks every non-anchor row against each anchor in the map and the -//! representation. The pass counts ranks instead of materializing sorted orderings, so the -//! map-versus-representation readings are exact at corpus scale while per-anchor memory stays -//! bounded by the largest neighbourhood. -//! - The sampled pass ranks a shared comparison universe - a bounded uniform sample whose canonical -//! embeddings arrive through the dataset's probe-scoped stream - in all three spaces, and reads -//! every space pair over that one universe. Canonical readings are never corpus-exact, because -//! the full canonical corpus stays at the source. The sampled pass measures the representation -//! baseline under the identical design, so comparing a map reading against it is like for like. +//! representation. Counting ranks over the full non-anchor universe avoids materializing sorted +//! orderings. Per-anchor ranking scratch grows with the largest neighbourhood. +//! - The sampled pass ranks a shared comparison universe in all three spaces. It fetches canonical +//! embeddings only for the sampled rows. The representation baseline and map-versus-canonical +//! reading share this universe, making their neighbourhood scales directly comparable. With fewer +//! comparisons than non-anchor rows, these rankings cover a sample of the corpus. //! -//! At equal `k`, a sampled reading covers a coarser neighbourhood than a corpus reading, because -//! the `k` nearest of a uniform sample of `m` rows sit at a corpus-scale depth of about `k · rows / -//! m` neighbours. The passes therefore answer different questions - fine placement against the -//! representation, coarse placement against the canonical space - and [`ProbeReadings`] keeps them -//! apart. +//! At equal k, a smaller uniform comparison sample measures a coarser neighbourhood. Among n +//! non-anchor rows, the expected full-universe rank of the k-th nearest of m sampled rows is: //! -//! Every ranking resolves distance ties by ascending row, so equal inputs produce equal readings. -//! The probe keeps readings per anchor ([`ReadingGrid`]), so whole-probe and per-subgroup roll-ups -//! merge cells instead of re-ranking. Anchors rank independently and in parallel; the corpus pass -//! performs `anchors · rows` representation-kernel evaluations and dominates the probe's runtime. +//! `k · (n + 1)/(m + 1)`, approximately `k · n/m`. +//! +//! [`ProbeReadings`] keeps the corpus and sampled grids separate. +//! +//! Rankings use computed distances, with row order breaking ties. Replaying requires equal corpus +//! and canonical values, options and initial generator state, together with the same numerical +//! environment. The distance kernels round to f32. Representation and canonical kernels accumulate +//! in f64, while squared map distances use f32 arithmetic. Finite coordinates alone still permit +//! overflow in that arithmetic. +//! +//! Anchors rank independently in parallel. Per-anchor cells ([`ReadingGrid`]) support whole-probe +//! and subgroup merges without re-ranking. For a anchors, n non-anchor rows and maximum +//! neighbourhood K, the corpus pass evaluates a · (n + K) representation distances and counts ranks +//! against K thresholds per scanned row. It shares an O(rows) anchor mask and uses O(K) ranking +//! scratch per worker, in addition to output grids. #![expect( clippy::cast_possible_truncation, reason = "the corpus row domain is checked against the crate's u32 row encoding at entry" @@ -71,10 +77,10 @@ mod readings; /// One generation's row-aligned probe inputs. /// -/// The slices describe the same rows in the same order; mapped `f32[N, 512]` and `f32[N, 2]` -/// artifacts yield the representation and coordinate slices directly. -/// [`with_clumps`](Self::with_clumps) attaches a clump grouping over the same rows when the probe -/// reads recall collapsed onto clump ids. +/// The inputs must describe the same rows in the same order, with unique byte-encoded source ids +/// and finite representations. Construction checks equal lengths. Coordinates have a finite-point +/// type, but squared map distances must also remain finite for finite radius statistics. +/// [`with_clumps`](Self::with_clumps) attaches labels over the same row domain. #[derive(Debug, Copy, Clone)] pub(crate) struct ProbeCorpus<'corpus, N> { node_ids: &'corpus IdSlice, @@ -88,8 +94,7 @@ impl<'corpus, N> ProbeCorpus<'corpus, N> { /// /// # Panics /// - /// This panics when the slices disagree about the row count; all three describe one generation, - /// so a mismatch is a wiring defect. + /// Panics when the inputs disagree about the row count. #[must_use] pub(crate) fn new( node_ids: &'corpus IdSlice, @@ -115,12 +120,11 @@ impl<'corpus, N> ProbeCorpus<'corpus, N> { } } - /// Attaches a clump grouping, enabling the collapsed corpus reading. + /// Attaches clump labels for corpus and sampled-baseline recall. /// /// # Panics /// - /// This panics when the grouping labels a different row count; both describe one generation, so - /// a mismatch is a wiring defect. + /// Panics when the grouping labels a different row count. #[must_use] pub(crate) fn with_clumps(mut self, clumps: &'corpus Clumps) -> Self { assert_eq!( @@ -141,23 +145,26 @@ impl<'corpus, N> ProbeCorpus<'corpus, N> { /// Matches an unordered delivery stream against the requested rows' ids. /// -/// Probe-scoped dataset streams owe no delivery order and identify their items only by source id, -/// so this function matches deliveries by id bytes and checks completeness - every requested id -/// exactly once, nothing else - before returning the payloads in `rows` order. It reads the -/// requests straight off the `(node_ids, rows)` pair, so no caller materializes a request list. +/// Matches source ids by their byte encoding and returns payloads in `rows` order. The requested +/// rows must identify distinct byte encodings, and the request count must fit u32. Success requires +/// every requested id exactly once, with no additional id. +/// +/// # Errors /// -/// This function enforces exactly-once rather than assuming it, because a violation damages the -/// reading it feeds. An unrequested id refuses, a short stream refuses, and a repeated id refuses -/// before its payload can replace the one already accepted. +/// Returns [`DeliveryError`] for a failed stream, an unrequested or repeated id, or incomplete +/// delivery. An identical repeated payload still fails. No payload collection is returned on +/// failure. +/// +/// # Panics +/// +/// Panics when a requested row lies outside `node_ids`. pub(super) async fn match_deliveries( node_ids: &IdSlice, rows: &[R], deliveries: impl Stream>, ) -> Result, DeliveryError> where - // Matching is on byte identity, not semantic order: `IntoBytes` totally - // orders exactly the encoding the stream echoes back, where an ordering - // bound on the id type would owe neither totality nor byte fidelity. + // matching compares the delivered byte encoding rather than an id type's semantic ordering I: zerocopy::IntoBytes + zerocopy::Immutable, R: Id, { @@ -175,8 +182,7 @@ where .await .map_err(DeliveryError::Dataset)? { - // `Err` from the search carries the insertion point, so a miss means - // the id was never requested. + // a failed search identifies a delivery outside the requested id set let position = order .binary_search_by(|&slot| key(slot).cmp(id.as_bytes())) .map_err(|_insertion| DeliveryError::Unrequested)?; @@ -200,13 +206,17 @@ where .collect()) } -/// Draws the probe's row sample: `anchors + comparisons` distinct rows in draw order. +/// Samples disjoint anchor and comparison rows in draw order. +/// +/// The result contains `anchors + comparisons` distinct rows, with anchors first. The count sum +/// must fit usize. Sampling is the first generator operation, preserving agreement with offline +/// coverage when population length, counts and initial generator state match. Use +/// [`probe_rng`](crate::salt::runner::probe_rng) for the fit runner's seed derivation. +/// +/// # Panics /// -/// The sample is its generator's first draw. The probe and the offline dump's coverage request -/// both draw through this function from equally seeded generators -/// ([`probe_rng`](crate::salt::runner::probe_rng)) over equal-length populations, so the two -/// agree on the sampled rows by construction, and a draw inserted ahead of this one would make -/// the probe request rows existing dumps never covered. +/// Panics when the requested count exceeds the population length, or the count sum overflows with +/// integer overflow checks enabled. pub(crate) fn probe_sample( mut rng: impl Rng, population: &IdSlice, @@ -217,6 +227,10 @@ pub(crate) fn probe_sample( } /// Fetches the sampled rows' canonical embeddings, in sample order. +/// +/// # Errors +/// +/// Returns [`ProbeError`] for a failed canonical stream or a delivery mismatch. async fn fetch_canonical<'data, D: Dataset>( dataset: &'data D, node_ids: &IdSlice, @@ -246,13 +260,19 @@ async fn fetch_canonical<'data, D: Dataset>( /// /// This samples anchor and comparison rows disjointly without replacement, then fetches both /// samples' canonical embeddings through the dataset's probe-scoped stream before any ranking -/// begins. +/// begins. The dataset must supply the same canonical values as the corpus represents, with finite +/// components. The requested count sum and all aggregate totals and normalization products must fit +/// their integer carriers. Design validation checks corpus size and neighbourhood shape, not those +/// arithmetic capacities. /// /// # Errors /// -/// Returns an error when the corpus cannot host the probe design, a neighbourhood size violates an -/// aggregate domain, the row count exceeds the crate's `u32` row encoding, or the canonical stream -/// fails, misdelivers, or ends short. +/// Returns [`ProbeError`] for an invalid probe design or a failed or mismatched canonical delivery. +/// +/// # Panics +/// +/// An overflowing design count or aggregate arithmetic can panic when integer overflow checks are +/// enabled. pub(crate) async fn probe( dataset: &D, corpus: ProbeCorpus<'_, D::NodeId>, @@ -318,8 +338,7 @@ pub(crate) async fn probe( let steps = options.neighbourhoods.len(); let mut triplet_columns = transpose_triplets(sampled.triplets); - // The pair-indexed arrays move into named fields through the enum, so - // reordering the pair schema cannot mismatch a reading with its field. + // use typed pair indices when assigning the named result fields let mut sampled_grids = transpose_pairs(sampled.cells) .map(|cells| Some(ReadingGrid::from_anchor_cells(cells, steps))); let mut sampled_grid = |pair: SpacePair| { @@ -419,7 +438,14 @@ impl SampledColumns { } } -/// Builds one empty aggregate per neighbourhood size over `universe`. +/// Builds one shape-validated empty aggregate per neighbourhood size. +/// +/// Arithmetic capacity for the eventual query count remains unchecked. +/// +/// # Errors +/// +/// Returns [`ProbeError::Neighbourhood`] for the first size outside the aggregate's domain over +/// `universe`. fn aggregate_template( universe: usize, options: &ProbeOptions, @@ -434,16 +460,16 @@ fn aggregate_template( .collect() } -/// Samples distinct comparison-index pairs, uniform over ordered pairs. +/// Samples ordered pairs of distinct comparison indices, with replacement. /// -/// A universe of fewer than two comparison points holds no ordered pair and yields none regardless +/// The comparison count must fit u32. Each pair is uniform over distinct indices, and pairs may +/// repeat across draws. A universe of fewer than two comparison points yields no pairs regardless /// of the requested count. pub(super) fn sample_pairs(mut rng: impl Rng, comparisons: usize, count: usize) -> Box<[[u32; 2]]> { let Some(choices) = NonZero::new(comparisons as u64) else { return Box::new([]); }; - // The second draw runs over a universe one smaller than the first, - // which is what leaves a single-point universe with no pair to draw. + // a single-point universe has no distinct second point let Some(second_choices) = NonZero::new(choices.get() - 1) else { return Box::new([]); }; @@ -451,11 +477,10 @@ pub(super) fn sample_pairs(mut rng: impl Rng, comparisons: usize, count: usize) core::iter::repeat_with(|| { let first = uniform_below(&mut rng, choices) as u32; let mut second = uniform_below(&mut rng, second_choices) as u32; - // Skip-over-self, in place of a rejection loop: `second` comes from - // a universe one smaller, and shifting the values at or above - // `first` up by one maps them onto everything except `first`. The - // largest shifted value is `comparisons - 1`, so the pair is - // distinct and in bounds by construction. + // A uniform index mapped bijectively onto a finite set remains uniform. The second draw + // covers 0..comparisons-1, and shifting at first maps that range onto every index except + // first. The largest result is comparisons-1, within the u32 domain. Therefore this draw is + // uniform over distinct second indices without rejection. if second >= first { second += 1; } diff --git a/libs/@local/graph/atlas/src/salt/quality/probe/options.rs b/libs/@local/graph/atlas/src/salt/quality/probe/options.rs index 9a0dcb69eed..746c5be6bab 100644 --- a/libs/@local/graph/atlas/src/salt/quality/probe/options.rs +++ b/libs/@local/graph/atlas/src/salt/quality/probe/options.rs @@ -1,4 +1,4 @@ -//! The probe's design parameters and their validation. +//! Sampling settings and corpus-size checks for a quality probe. use alloc::borrow::Cow; use core::num::NonZero; @@ -42,34 +42,30 @@ const DEFAULT_HORIZON_FACTOR: NonZero = // not measured. const DEFAULT_TRIPLET_PAIRS: usize = 64; -/// Pinned sampling and neighbourhood settings for one probe. +/// Sampling and neighbourhood settings for one probe. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) struct ProbeOptions { /// Sampled anchor rows: the queries every reading aggregates over. + /// + /// Uses 256 by default. pub anchors: NonZero = DEFAULT_ANCHORS, /// Sampled comparison rows: the shared universe the sampled pass ranks. /// - /// More rows sharpen the canonical readings toward finer neighbourhood scales and grow the - /// canonical fetch linearly. + /// Uses 4,096 by default. More rows measure finer neighbourhood scales at the same k and grow the canonical fetch linearly. pub comparisons: NonZero = DEFAULT_COMPARISONS, /// Neighbourhood sizes to read at, in reporting order. /// - /// The list must name at least one size. The trend across sizes is itself evidence: recall - /// rising with `k` is the near-tie reshuffling fingerprint. + /// Uses `[15, 30, 50]` by default. The list must name at least one size, each at most half both comparison universes. Recall rising with k can suggest near-boundary reshuffling, but the trend alone does not identify its cause. pub neighbourhoods: Cow<'static, [NonZero]> = Cow::Borrowed(DEFAULT_NEIGHBOURHOODS), /// Horizon multiplier for the intrusion and extrusion readings. /// - /// A false neighbour counts as an intrusion or extrusion when its 1-based opposite-space rank - /// passes `factor · k` (clamped to the universe), separating foreign points from reshuffling - /// near the neighbourhood boundary. + /// Uses 2 by default. A false neighbour counts as an intrusion or extrusion when its one-based opposite-space rank exceeds min(factor · k, universe), distinguishing distant ranks from swaps near the neighbourhood boundary. pub horizon_factor: NonZero = DEFAULT_HORIZON_FACTOR, /// Comparison-point pairs sampled for the triplet readings. /// - /// Every anchor reads the one shared pair sample, so the estimate's mean stays unbiased while - /// all anchors share one pair-driven variance. The reading's resolution therefore tracks this - /// count rather than the anchor-times-pair triplet total. Zero disables the readings - and with - /// them admission: the verdict demands the full battery, so a triplet-free probe is report-only - /// by construction. + /// Uses 64 by default. Each pair contains distinct comparison points, but pairs sample with replacement. Every anchor evaluates the same pairs. Conditional on the selected anchors and comparison rows, their mean is unbiased for agreement over all ordered pairs. The anchor-times-pair total is not a count of independent observations. + /// + /// Zero disables triplet sampling. The resulting report cannot pass admission because the triplet control requires observed triplets. pub triplet_pairs: usize = DEFAULT_TRIPLET_PAIRS, } @@ -81,13 +77,21 @@ const impl Default for ProbeOptions { /// Checks the probe design fits the corpus. /// -/// The design holds when the row count fits the `u32` probe domain, at least one neighbourhood size -/// is named, and the corpus can host the disjoint anchor and comparison samples. +/// Checks the u32 row domain, a nonempty neighbourhood list and room for disjoint samples. +/// +/// `anchors + comparisons` must fit usize. Neighbourhood shapes and aggregate arithmetic capacity +/// are separate conditions. +/// +/// # Errors +/// +/// Returns [`ProbeError`] for an oversized row domain, an empty neighbourhood list or insufficient +/// corpus rows, in that order. +/// +/// # Panics +/// +/// Panics on an overflowing anchor-plus-comparison count when integer overflow checks are enabled. pub(super) fn validate_design(rows: usize, options: &ProbeOptions) -> Result<(), ProbeError> { - // The corpus arrives as mapped slices, so its row count is a usize; - // the probe's own row ids, orderings, and pair samples all travel as - // u32. Checking the width once here makes every later narrowing cast - // lossless. + // the corpus row count bounds sampled row positions and ranks narrowed to u32 if u32::try_from(rows).is_err() { return Err(ProbeError::RowsExceedProbeDomain { rows }); } diff --git a/libs/@local/graph/atlas/src/salt/quality/probe/pass.rs b/libs/@local/graph/atlas/src/salt/quality/probe/pass.rs index b34acf7cdd5..5929c71f58c 100644 --- a/libs/@local/graph/atlas/src/salt/quality/probe/pass.rs +++ b/libs/@local/graph/atlas/src/salt/quality/probe/pass.rs @@ -1,11 +1,11 @@ //! Per-anchor ranking workers over the probe's shared inputs. //! -//! Each pass is a context binding its shared inputs once; running one ranks the anchors -//! independently and in parallel under one total order - distances by [`f32::total_cmp`], ties by -//! ascending row - and yields per-anchor cells cloned from a prevalidated template. The corpus pass -//! counts ranks against bounded threshold sets, so its per-thread memory follows the search depth, -//! never the corpus; the sampled pass sorts whole comparison universes, whose size the probe design -//! bounds. +//! Both passes rank anchors independently in parallel, ordering [`NonNegative`] distances first and +//! breaking ties by ascending row. They produce per-anchor cells from shape-validated templates. +//! The corpus pass counts ranks against bounded threshold sets, using O(K) ranking scratch for +//! search depth K and a shared corpus-sized anchor mask. The sampled pass sorts its whole +//! comparison universe, with O(m) ranking scratch for m comparisons. Output cells are additional to +//! that scratch. #![expect( clippy::cast_possible_truncation, reason = "the corpus row domain is checked against the crate's u32 row encoding at probe entry" @@ -15,17 +15,11 @@ reason = "k is the canonical neighbourhood-size name across the metric literature" )] -// PERF: at the default search depth (K = 50) the linear threshold- -// counting loops in the corpus pass cost cycles comparable to the -// 512-component distance kernel itself. The corpus pass is then -// plausibly 30-40% scalar counting. If the suite's runtime ever -// matters, the algorithmic fix comes before SIMD. Sort the K -// thresholds once per anchor. Then binary-search each candidate's -// insertion point for log K compares instead of K and suffix-sum a -// small histogram into the per-threshold counts. That stays exact and -// costs less than vectorizing compares whose order is lexicographic -// over distance and row rather than a plain float compare. Measure at -// live shape (1M rows x 256 anchors) before acting. +// PERF: threshold counting compares every candidate with K thresholds. A possible alternative sorts +// thresholds, binary-searches the first threshold greater than each candidate and accumulates those +// suffix increments into per-threshold counts. This uses O(log K) comparisons per candidate under +// the same lexicographic distance/row order. Measure counting and distance-kernel costs at live +// shape (1M rows and 256 anchors) before choosing this change or vectorizing comparisons. use alloc::{borrow::Cow, collections::BinaryHeap}; use core::{cmp::Ordering, num::NonZero}; @@ -110,12 +104,18 @@ fn push_bounded( }; if candidate < *farthest { - // `PeekMut` sifts the replacement into place on drop. + // PeekMut restores heap order on drop *farthest = candidate; } } /// Sorts universe indices nearest-first, ties by ascending row. +/// +/// `rows` must cover the distance array, whose length must fit u32. +/// +/// # Panics +/// +/// Panics when a compared row index lies outside `rows`. fn order_into( order: &mut Vec, distances: &[NonNegative], @@ -132,8 +132,7 @@ fn order_into( /// One anchor's corpus-pass output across the neighbourhood sizes. /// -/// Readings outlive the per-thread scratch arena, so they own plain heap storage; only the ranking -/// intermediates live in the arena. +/// Readings own their storage independently of the reusable ranking scratch. pub(super) struct AnchorReading { /// Rank aggregates, one per neighbourhood size. pub cells: Vec, @@ -146,6 +145,11 @@ pub(super) struct AnchorReading { } /// Shared inputs for ranking every anchor against every non-anchor row. +/// +/// Row-aligned inputs must cover the mask domain. `search` must reach every neighbourhood size and +/// fit the non-anchor universe. Templates must match the neighbourhood list and universe, with +/// capacity for the intended aggregate totals. These relationships are established by the probe's +/// construction, except for arithmetic capacity, which it does not check. pub(super) struct CorpusPass<'pass, N> { /// The representation matrix, in row order. pub representations: &'pass IdSlice>, @@ -155,7 +159,7 @@ pub(super) struct CorpusPass<'pass, N> { pub anchor_mask: &'pass DenseBitSet, /// Nearest rows kept per space: the largest neighbourhood size. pub search: usize, - /// Prevalidated empty aggregates, one per neighbourhood size. + /// Shape-validated empty aggregates, one per neighbourhood size. pub template: &'pass [NeighbourhoodAggregate], /// The neighbourhood sizes, in the template's order. pub neighbourhoods: &'pass [NonZero], @@ -170,6 +174,11 @@ where N: Id, { /// Ranks every anchor, yielding per-neighbourhood readings in anchor order. + /// + /// # Panics + /// + /// Evaluating the iterator panics when an anchor or scanned row exceeds an input's row domain, + /// or the search depth cannot supply a requested neighbourhood. pub(super) fn run<'call>( &'call self, anchor_rows: &'call [N], @@ -188,7 +197,13 @@ where /// /// The rank of a neighbour is the count of universe rows strictly nearer under the total order, /// accumulated against the opposite space's nearest [`search`](Self::search) rows during each - /// scan. The pass scans the representation matrix once and the coordinate frame twice. + /// scan. The pass scans the representation matrix once and the coordinate frame twice, with + /// additional distance evaluations for the retained thresholds. + /// + /// # Panics + /// + /// Panics on an out-of-domain row or when retained neighbours cannot cover a requested size. + /// Aggregate shape mismatches also panic during observation. fn anchor( &self, anchor: N, @@ -200,7 +215,7 @@ where let anchor_point = self.coordinates[anchor]; let anchor_embedding = &self.representations[anchor]; - // The map's nearest rows, from the first coordinate scan. + // the first coordinate scan supplies map neighbours whose reference ranks are needed let mut heap = BinaryHeap::new_in(&*scratch); for row in negated_anchor_mask { @@ -215,8 +230,8 @@ where let nearest = heap.into_sorted_vec(); - // Representation scan: the reference nearest rows, and each map - // neighbour's reference rank counted against its distance. + // one representation scan finds its nearest rows and counts the opposite ranks of map + // neighbours let mut thresholds = Vec::with_capacity_in(nearest.len(), &*scratch); thresholds.extend(nearest.iter().map(|member| Ranked { distance: anchor_embedding.cosine_distance(&self.representations[member.row]), @@ -248,7 +263,7 @@ where let reference_nearest = heap.into_sorted_vec(); let reference_ranks = counts.clone(); - // Second coordinate scan: each reference neighbour's map rank. + // the second coordinate scan counts the map ranks of reference neighbours thresholds.clear(); thresholds.extend(reference_nearest.iter().map(|member| Ranked { distance: anchor_point.distance_squared(self.coordinates[member.row]), @@ -279,8 +294,7 @@ where for (aggregate, &k) in cells.iter_mut().zip(self.neighbourhoods) { aggregate.observe_ranks(&reference_ranks[..k.get()], &counts[..k.get()]); radii.push(RadiusPair { - // The map scans rank by squared distance; the radius is - // the distance itself. + // rankings use squared distance, but density ratios use Euclidean radii map: nearest[k.get() - 1].distance.sqrt(), representation: reference_nearest[k.get() - 1].distance, }); @@ -293,10 +307,15 @@ where } } - /// Reads the clump-collapsed cells from the anchor's nearest lists, empty without a grouping. + /// Computes collapsed recall from nearest-first row lists. /// - /// Both nearest lists arrive nearest-first from the anchor's scans, so the collapsed reading - /// costs two small label sweeps per neighbourhood size. + /// Returns an empty vector without a grouping. For each neighbourhood size k, label collection + /// and overlap cost O(k log k) time, including sorting the label lists. + /// + /// # Panics + /// + /// Panics when a nearest list is shorter than a requested neighbourhood or a row lies outside + /// the grouping. fn clump_cells( &self, map_nearest: &[Ranked], @@ -335,8 +354,7 @@ where /// One anchor's sampled-pass output across the space pairs. /// -/// Readings outlive the per-thread scratch arena, so they own plain heap storage; only the ranking -/// intermediates live in the arena. +/// Readings own their storage independently of the reusable ranking scratch. pub(super) struct SampledReading { /// Rank aggregates per space pair, one cell per neighbourhood size. pub cells: SpacePairArray>, @@ -348,10 +366,12 @@ pub(super) struct SampledReading { pub baseline_clumps: Vec, } -/// Shared inputs for ranking every anchor against the comparison rows in all three spaces. +/// Shared inputs for ranking anchors against sampled comparisons in all three spaces. /// -/// The canonical embeddings arrive as the dataset served them: borrowed straight out of a -/// mapped dump, or owned where the source decodes rows. +/// Canonical arrays must align with their respective anchor and comparison rows. Other row-indexed +/// inputs must cover those rows. Templates must match the comparison universe and neighbourhood +/// sizes, with supported totals. Pair indices must address distinct comparison rows. Canonical +/// embeddings may borrow dataset storage or own decoded vectors. pub(super) struct SampledPass<'pass> { /// The representation matrix, in row order. pub representations: &'pass IdSlice>, @@ -363,7 +383,7 @@ pub(super) struct SampledPass<'pass> { pub comparison_canonical: &'pass [Cow<'pass, AlignedVecN>], /// The pass's shared universe of comparison rows. pub comparison_rows: &'pass [NodeRowId], - /// Prevalidated empty aggregates, one per neighbourhood size. + /// Shape-validated empty aggregates, one per neighbourhood size. pub template: &'pass [NeighbourhoodAggregate], /// The neighbourhood sizes, in the template's order. pub neighbourhoods: &'pass [NonZero], @@ -378,6 +398,11 @@ pub(super) struct SampledPass<'pass> { impl SampledPass<'_> { /// Ranks every anchor, yielding per-space-pair and clump-collapsed readings in anchor order. + /// + /// # Panics + /// + /// Evaluating the iterator panics when a row or pair index exceeds its input domain, or a + /// template disagrees with the comparison universe. pub(super) fn run<'call>( &'call self, anchor_rows: &'call [NodeRowId], @@ -392,7 +417,12 @@ impl SampledPass<'_> { /// Ranks one anchor's comparison universe in all three spaces. /// - /// Reads the space pairs, the triplet verdicts, and the clump-collapsed baseline. + /// Produces neighbourhood cells, triplet verdicts and the clump-collapsed baseline. + /// + /// # Panics + /// + /// Panics on an out-of-domain row, anchor ordinal or pair index, or a template universe + /// inconsistent with the comparison count. fn anchor(&self, index: usize, anchor: NodeRowId, scratch: &mut Scratch) -> SampledReading { scratch.reset(); @@ -444,9 +474,7 @@ impl SampledPass<'_> { let representation_canonical = observed(&canonical_order, &representation_order, &mut ranks); - // Triplet verdicts: whether each space orders the pair's two - // points the same way from this anchor, under the shared - // (distance, row) total order. Distinct rows leave no ties. + // distinct rows resolve equal distances, giving each space one order for the pair let mut triplets = SpacePairArray::from_elem(TripletAggregate::default()); for &[first, second] in self.pairs { let nearer_first = |distances: &[NonNegative]| { @@ -482,14 +510,17 @@ impl SampledPass<'_> { } } - /// Reads the clump-collapsed representation-versus-canonical cells from the anchor's orderings. + /// Computes collapsed baseline recall from nearest-first comparison orderings. + /// + /// Returns an empty vector without a grouping. The canonical ordering is the reference. + /// Relabeling measures how much of its neighbourhood the representation keeps with row identity + /// relaxed to the component. Label collection and overlap cost O(k log k) per size k, including + /// sorting. /// - /// Empty without a grouping. + /// # Panics /// - /// Both orderings arrive whole-universe nearest-first, so the collapsed baseline costs two - /// small label sweeps per neighbourhood size. The canonical ordering is the reference: the - /// collapse reads how much of each exact canonical neighbourhood the representation keeps after - /// relabeling rows by their connected-component id. + /// Panics when an ordering is shorter than a requested neighbourhood, an index exceeds the + /// comparison universe, or a row lies outside the grouping. fn clump_cells( &self, canonical_order: &[u32], diff --git a/libs/@local/graph/atlas/src/salt/quality/probe/readings.rs b/libs/@local/graph/atlas/src/salt/quality/probe/readings.rs index c597709dfb6..1f9bf21324e 100644 --- a/libs/@local/graph/atlas/src/salt/quality/probe/readings.rs +++ b/libs/@local/graph/atlas/src/salt/quality/probe/readings.rs @@ -1,8 +1,8 @@ //! The probe's reading grids and their axes. //! -//! Everything here is a value the probe hands back: per-anchor aggregate grids, neighbourhood -//! radii, and the triplet readings with their shared pair sample. Consumers regroup these by -//! merging cells, whole-probe or by subgroup, and nothing here re-ranks. +//! Per-anchor aggregates permit whole-probe and subgroup reductions without re-ranking. The grids +//! share typed anchor and neighbourhood axes. Radius pairs and sampled triplet verdicts retain the +//! measurements needed by the density and agreement reports. use core::{mem, num::NonZero}; @@ -16,30 +16,25 @@ use super::super::{ use crate::{identity::OntologyRowId, math::NonNegative}; hashql_core::id::newtype! { - /// One position on the grids' neighbourhood axis, in the options' reporting order. + /// A position on the grids' neighbourhood axis. /// - /// The probe reads every metric at a ladder of neighbourhood sizes, and a step addresses one of - /// them. The grids address cells by step, and the step's neighbourhood size lives in - /// [`ProbeReadings::neighbourhoods`], so a step index and a neighbourhood size can never - /// stand in for one another. + /// The size at this position is in [`ProbeReadings::neighbourhoods`], in options order. A step indexes that list rather than specifying a neighbourhood size. #[id(const)] pub(crate) struct Step(u32) } hashql_core::id::newtype! { - /// One position on the grids' anchor axis, in sampling order. + /// A position on the grids' anchor axis. /// - /// An ordinal addresses a sampled anchor's readings; the anchor's row id lives at the same - /// position of [`ProbeReadings::anchors`], so an ordinal and a corpus row can never stand in - /// for one another. + /// The anchor's corpus row is at this position in [`ProbeReadings::anchors`], in sampling order. #[id(const)] pub(crate) struct AnchorOrdinal(u32) } -/// The probe's space pairs, in one pinned reporting order. +/// Space-pair indices in reporting order. /// -/// The passes' positional plumbing - pair-indexed arrays - uses this order; the reading structs -/// name their pair fields instead, so the enum is the bridge between the two. +/// Each pair identifies a judged space and its reference. [`ProbeReadings`] exposes the +/// corresponding grids as named fields. #[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Id)] pub(crate) enum SpacePair { /// The 2D map judged against the 512-component representation. @@ -55,14 +50,15 @@ impl SpacePair { pub(crate) const COUNT: usize = mem::variant_count::(); } +/// One value per space pair, indexed in the pinned reporting order. pub(crate) type SpacePairArray = IdArray; /// Per-anchor aggregates for one space pair, anchor-major. /// -/// Every cell reads one anchor at one neighbourhood size; the neighbourhood axis follows the -/// options' reporting order. Roll-ups merge cells, so a consumer groups anchors - whole-probe or by -/// subgroup - without touching orderings again. The cell type is the aggregate the grid holds: rank -/// aggregates for the space-pair grids, clump aggregates for the collapsed corpus grid. +/// Every cell reads one anchor at one neighbourhood size, with sizes in options order. Rank grids +/// describe space pairs and clump grids describe collapsed recall. Merging at a step combines +/// anchors without revisiting orderings. Construction checks rectangular shape alone. Aggregates in +/// a column must share the shape required by their merge operation, with supported combined totals. #[derive(Debug, Clone)] pub(crate) struct ReadingGrid { cells: IdMatrix, @@ -73,8 +69,7 @@ impl ReadingGrid { /// /// # Panics /// - /// This panics when a row's cell count differs from `steps`; every anchor reads the same steps, - /// so a ragged row is a wiring defect. + /// Panics when `steps` is zero or a row's cell count differs from `steps`. pub(crate) fn from_anchor_cells(rows: Vec>, steps: usize) -> Self { Self { cells: IdMatrix::from_rows(rows, steps), @@ -93,14 +88,13 @@ impl ReadingGrid { } } -// Each cell type implements `overall` itself rather than sharing one -// merge trait. impl ReadingGrid { /// Merges every anchor's reading at one step. /// /// # Panics /// - /// This panics when `step` lies outside the grid or the grid holds no anchor. + /// Panics when `step` lies outside the grid, the grid holds no anchor, or the column's + /// aggregates disagree about universe, neighbourhood size or horizon. #[must_use] pub(crate) fn overall(&self, step: Step) -> NeighbourhoodAggregate { let mut column = self.cells.column(step); @@ -116,12 +110,14 @@ impl ReadingGrid { merged } - /// Merges the named anchors' readings at one step: the subset case of - /// [`overall`](Self::overall). + /// Merges the named anchors' readings at one step. + /// + /// A repeated anchor contributes again. Use [`overall`](Self::overall) for every anchor once. /// /// # Panics /// - /// This panics when `anchors` is empty or when an anchor or `step` lies outside the grid. + /// Panics when `anchors` is empty, an index lies outside the grid, or the selected aggregates + /// disagree about universe, neighbourhood size or horizon. #[must_use] pub(crate) fn merged(&self, anchors: &[AnchorOrdinal], step: Step) -> NeighbourhoodAggregate { let (&first, rest) = anchors @@ -140,7 +136,8 @@ impl ReadingGrid { /// /// # Panics /// - /// This panics when `step` lies outside the grid or the grid holds no anchor. + /// Panics when `step` lies outside the grid, the grid holds no anchor, or the column's + /// aggregates disagree about neighbourhood size. #[must_use] pub(crate) fn overall(&self, step: Step) -> ClumpAggregate { let mut column = self.cells.column(step); @@ -151,12 +148,14 @@ impl ReadingGrid { merged } - /// Merges the named anchors' readings at one neighbourhood size: the subset case of - /// [`overall`](Self::overall). + /// Merges the named anchors' readings at one neighbourhood size. + /// + /// A repeated anchor contributes again. Use [`overall`](Self::overall) for every anchor once. /// /// # Panics /// - /// This panics when `anchors` is empty or when an anchor or `step` lies outside the grid. + /// Panics when `anchors` is empty, an index lies outside the grid, or the selected aggregates + /// disagree about neighbourhood size. #[must_use] pub(crate) fn merged(&self, anchors: &[AnchorOrdinal], step: Step) -> ClumpAggregate { let (&first, rest) = anchors @@ -172,9 +171,10 @@ impl ReadingGrid { /// One anchor's neighbourhood radii at one neighbourhood size. /// -/// The radii live in their own metrics - euclidean map distance, cosine representation distance - -/// so a single ratio carries no meaning; the spread of log ratios across anchors does, because a -/// metric change shifts every log ratio by one constant. +/// Map radii use Euclidean distance and representation radii use cosine distance. A log radius +/// ratio includes the scales of both metrics. Uniform multiplicative rescaling of either radius +/// shifts every finite log ratio by one constant, preserving their median absolute deviation in +/// exact arithmetic. Arbitrary metric changes need not preserve the spread. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct RadiusPair { /// Distance to the k-th nearest non-anchor row on the map. @@ -210,9 +210,14 @@ pub(crate) struct ClumpReadings { /// One probe's readings across the three space pairs. /// -/// The corpus grid ranks every non-anchor row, so its universe is `rows - anchors`; the sampled -/// grids share the comparison rows as their universe. Each grid records its own universe in its -/// aggregates, so a reading is never mistaken for a measurement at another scale. +/// The corpus grid ranks every non-anchor row, with universe `rows - anchors`. The sampled grids +/// share the comparison rows as their universe. Each grid records its own universe in its +/// aggregates. +/// +/// Values produced by [`probe`](super::probe) have nonempty, aligned anchor and neighbourhood axes. +/// Direct construction must preserve those axes across grids, radii and triplet columns. The same +/// obligation covers each aggregate's shape and arithmetic capacity, and the fields store what a +/// caller supplies without a mutual-consistency check. #[derive(Debug)] pub(crate) struct ProbeReadings { /// Sampled anchor rows, in sampling order: the grids' anchor axis. @@ -223,10 +228,10 @@ pub(crate) struct ProbeReadings { /// /// The size at each [`Step`] of the grids' neighbourhood axis. pub neighbourhoods: Box>>, - /// Map versus representation, ranking every non-anchor row against each sampled anchor. The - /// comparison universe is every row that is not itself an anchor - exact, and the whole corpus - /// but for the anchors - while the aggregate remains an anchor-sampled statistic rather than a - /// corpus-population estimate. + /// Map versus representation over every non-anchor row. + /// + /// Rankings cover the full non-anchor universe, while aggregate readings retain + /// anchor-sampling uncertainty. pub map_representation: ReadingGrid, /// The corpus reading collapsed onto clump ids. /// @@ -267,8 +272,7 @@ impl ProbeReadings { /// /// # Panics /// - /// This panics when `anchor_types` and the readings disagree about the anchor count; both - /// describe one probe, so a mismatch is a wiring defect. + /// Panics when `anchor_types` and the readings disagree about the anchor count. #[must_use] pub(crate) fn with_anchor_types<'probe>( &'probe self, @@ -288,9 +292,10 @@ impl ProbeReadings { /// One probe's readings beside each anchor's direct types. /// -/// `anchor_types` is parallel to the readings' anchors: each entry lists one anchor's direct -/// types, and an empty entry leaves its anchor in the whole-probe readings only. Construction -/// checks the cover once, so a consumer never re-checks the anchor count. +/// `anchor_types` is parallel to the readings' anchors. Each entry lists one anchor's direct types, +/// and an empty entry leaves its anchor in the whole-probe readings only. Construction checks the +/// anchor count, not type membership or uniqueness. Repeated type entries count the anchor +/// repeatedly in that subgroup. #[derive(Debug)] pub(crate) struct TypedReadings<'probe, N> { /// The probe's readings. @@ -315,7 +320,7 @@ impl<'probe, N> TypedReadings<'probe, N> { } } -// Not derived: the fields are references, so copying needs no `N: Copy` bound. +// copying shared references needs no N: Copy bound impl Copy for TypedReadings<'_, N> {} impl Clone for TypedReadings<'_, N> { @@ -335,7 +340,6 @@ mod tests { tests::{ProbeFixture, irregular_angles}, }; - /// The gathered grid holds exactly the input rows' anchor and step counts. #[test] fn from_anchor_cells_dimensions() { let grid = ReadingGrid::from_anchor_cells(vec![vec![0_u8; 2]; 5], 2); @@ -343,7 +347,6 @@ mod tests { assert_eq!(grid.cells.columns(), 2); } - /// The probe builds every space-pair grid at the requested anchor and step counts. #[tokio::test] async fn probe_grid_dimensions() { let fixture = ProbeFixture::on_circle(&irregular_angles(48)); diff --git a/libs/@local/graph/atlas/src/salt/quality/report/calibration.rs b/libs/@local/graph/atlas/src/salt/quality/report/calibration.rs index edd86af9bb2..b3bdcc114ee 100644 --- a/libs/@local/graph/atlas/src/salt/quality/report/calibration.rs +++ b/libs/@local/graph/atlas/src/salt/quality/report/calibration.rs @@ -1,13 +1,10 @@ //! The clump-threshold calibration over a published k-NN table. //! -//! ε is a calibrated configuration value, so its default must carry measured corpus structure -//! rather than a guess: the calibration opens a published table and reads the grouping's shape - -//! clump count, multi-row group count, covered rows - at every candidate threshold. A candidate -//! qualifies by how well it reproduces the audited corpus shape and by whether the flagged -//! subgroups it should resolve actually restore. -//! -//! The calibration observes a published artifact and never participates in a fit: it reads the -//! stored table, so a reading describes the grouping a run at that threshold would have built. +//! [`calibrate`] measures clump count, multi-row group count and covered rows at each candidate ε. +//! Reusing one published table isolates the effect of the threshold on connected-component +//! grouping. Compare these counts and their trends before choosing ε for that generation, then +//! assess subgroup recall to judge its diagnostic effect. This sweep measures grouping shape alone +//! and selects no threshold. use core::{ error::Error, @@ -24,9 +21,11 @@ use crate::{ }, }; -/// The candidate thresholds swept by default cover one value below the calibrated plateau, both -/// plateau edges, the deployed value, and the percolation boundary above it, so a bare invocation -/// re-derives the evidence behind [`DEFAULT_EPSILON`]. +/// Default sweep thresholds around [`DEFAULT_EPSILON`]. +/// +/// The interval 0.0012 to 0.0028 spans the plateau in some recorded development-corpus fits. The +/// outer points 0.0005 and 0.0045 sample below and above it. Another generation may have no plateau +/// over this interval. pub(crate) const DEFAULT_EPSILONS: &[f32] = &[0.0005, 0.0012, DEFAULT_EPSILON, 0.0028, 0.0045]; /// The grouping's shape at one candidate threshold. @@ -43,7 +42,7 @@ pub(crate) struct Reading { } impl Reading { - /// The share of the corpus that sits inside a multi-row clump. + /// Returns the grouped-row fraction, or zero for an empty corpus. #[expect( clippy::cast_precision_loss, reason = "row counts stay far inside the f64 mantissa" @@ -56,7 +55,7 @@ impl Reading { self.grouped_rows as f64 / rows as f64 } - /// The mean size of a multi-row clump. + /// Returns the mean multi-row clump size, or zero when no such clump exists. #[expect( clippy::cast_precision_loss, reason = "row counts stay far inside the f64 mantissa" @@ -108,14 +107,13 @@ impl Display for Calibration { } } -/// The path does not hold a readable k-NN table. +/// A failure to open a k-NN table for threshold calibration. /// -/// Splices into the chain transparently: the display text and the sources are the wrapped -/// crate-internal fault's, unchanged. +/// Display text and error sources match the underlying artifact error. #[derive(Debug)] pub(crate) struct CalibrationError(CalibrationFault); -/// The ways the table fails to open. +/// The artifact failure exposed by a calibration error. #[derive(Debug)] enum CalibrationFault { /// The sparse file did not open. @@ -142,7 +140,11 @@ impl Error for CalibrationError { } } -/// Reads the grouping shape of the k-NN table at `path` for every candidate threshold. +/// Measures a published table's grouping shape at each candidate threshold. +/// +/// Results preserve `epsilons` order. Threshold comparisons follow [`Clumps::from_knn`], including +/// its NaN and infinity behavior. Backing file bytes must remain immutable throughout the mapping's +/// lifetime. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/quality/report/document.rs b/libs/@local/graph/atlas/src/salt/quality/report/document.rs index 24f633b27a7..a0069539a93 100644 --- a/libs/@local/graph/atlas/src/salt/quality/report/document.rs +++ b/libs/@local/graph/atlas/src/salt/quality/report/document.rs @@ -24,22 +24,18 @@ pub(crate) struct MetricRow { pub trustworthiness: UnitFraction, /// Continuity, in `[0, 1]`. pub continuity: UnitFraction, - /// Fraction of false neighbours past the horizon, in `[0, 1]`. + /// Fraction of map neighbours past the reference-rank horizon, in `[0, 1]`. pub intrusion_rate: UnitFraction, - /// Fraction of banished neighbours past the horizon, in `[0, 1]`. + /// Fraction of reference neighbours past the map-rank horizon, in `[0, 1]`. pub extrusion_rate: UnitFraction, } impl MetricRow { /// Reads one aggregate at the given neighbourhood size. /// - /// Every row a probe produces observes at least one query, from three independent reasons: - /// `ProbeOptions::anchors` is a `NonZero`, the sampled pass observes every step cell once per - /// anchor, and a subgroup row merges at least the anchor that created its membership. A row - /// read from an empty aggregate publishes each reading's own optimum instead - recall one, the - /// rates zero - and [`controls`](QualityReport::controls) folds those into its extremum as - /// observed evidence, where the triplet control keys on its observed count and refuses. A new - /// row source either keeps that invariant or gives the controls `queries` to key on. + /// `neighbourhood` labels the row and must match the aggregate's size. An empty aggregate + /// produces recall, trustworthiness and continuity of one and rates of zero. + /// [`QualityReport::controls`] includes such a row in its extrema without checking `queries`. pub(super) fn read(neighbourhood: NonZero, aggregate: &NeighbourhoodAggregate) -> Self { Self { neighbourhood, @@ -55,19 +51,25 @@ impl MetricRow { /// One neighbourhood size's density-distortion reading. /// -/// The reading is the spread of `ln(map radius / representation radius)` over the anchors: zero -/// when the map rescales every neighbourhood alike, growing as regions compress or dilate unevenly. -/// The median log ratio is the global scale offset - it carries the two metrics' unit difference -/// and is comparable only across probes of the same spaces. +/// For positive finite radii, the reading is the unscaled median absolute deviation of ln(map +/// radius) − ln(representation radius). A constant radius ratio gives zero spread. MAD measures +/// dispersion around the median and can remain zero when some ratios differ. The median log ratio +/// retains the metrics' relative scale. Uniform multiplicative radius rescaling shifts it without +/// changing the spread in exact arithmetic. #[derive(Debug, Copy, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct DensityRow { /// The neighbourhood size both radii come from. pub neighbourhood: NonZero, - /// Anchors contributing a finite log ratio. + /// Anchors with positive radii contributing a log ratio. + /// + /// Finite radii give finite ratios, though positivity alone admits an overflowed infinite + /// radius. pub anchors: usize, /// Anchors excluded for a zero radius. /// - /// At least `neighbourhood` rows coincide with the anchor in one of the spaces. + /// The computed k-th radius is zero in at least one space. Cosine-equivalent directions or + /// floating-point rounding can produce a zero representation radius without equal embedding + /// components. pub degenerate: usize, /// The median log radius ratio, absent without contributing anchors. pub median_log_ratio: Option, @@ -113,7 +115,7 @@ pub(crate) struct ClumpRow { pub queries: usize, /// Mean matched fraction of the collapsed neighbourhoods, in `[0, 1]`. /// - /// Never below the plain recall at the same size. + /// For probe-produced rows, never below plain recall over the same neighbourhood lists. pub recall: UnitFraction, } @@ -167,16 +169,16 @@ pub(crate) struct BaselineRow { pub recall: UnitFraction, /// The same reading collapsed onto clump ids, when clump readings exist. /// - /// Never below the plain recall. + /// For probe-produced rows, never below plain recall over the same neighbourhood lists. pub clump_recall: Option, } /// One subgroup's representation-baseline readings over the sampled universe. /// -/// The stratification separates representation loss from near-tie reshuffling under the triage -/// rule. When a subgroup's plain baseline recall trails the whole-probe reading and its collapsed -/// recall restores to it, the breach lies only on component labels in the representation itself, -/// before any projection. That reading triages the breach and certifies nothing about placement. +/// Plain and collapsed recall show how a subgroup's representation loss changes when component +/// labels replace row identity, before projection. Matching the whole-probe baseline after collapse +/// is diagnostic evidence, not proof that the difference arose from near ties or that the component +/// is compact. These rows contain no placement judgment. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct BaselineSubgroupReport { /// The subgroup's type, as its ontology row. @@ -192,9 +194,9 @@ pub(crate) struct BaselineSubgroupReport { /// One breach of the subgroup degradation rule. /// /// A flag carries its own triage evidence. When clump readings exist, the report re-evaluates the -/// breach on clump ids and marks a breach the collapse restores as resolved, meaning -/// component-label recall no longer breaches. The mark is triage evidence and certifies neither -/// component compactness nor within-component placement, and it never affects admission. +/// breach on clump ids and marks it resolved when collapsed subgroup degradation is at most the +/// configured factor times collapsed whole-probe degradation. The mark certifies neither component +/// compactness nor within-component placement. Flags and resolution never affect admission. #[derive(Debug, Copy, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct SubgroupFlag { /// The flagged subgroup's type, as its ontology row. @@ -220,9 +222,9 @@ pub(crate) struct SubgroupFlag { /// The side of a control's threshold that admits. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum Bound { - /// The reading must reach the threshold. + /// The reading must be at least the threshold. Floor(f64), - /// The reading must stay under the threshold. + /// The reading must be at most the threshold. Ceiling(f64), } @@ -236,19 +238,21 @@ impl Bound { } } -/// One control of the battery, reading one metric against the threshold that admits it. +/// One metric reading paired with its inclusive admission bound. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct Control { /// The metric the control checks. pub metric: QualityMetric, - /// The reading the verdict turns on, absent exactly when the evidence is. + /// The reduced reading, absent when the control's presence check fails. pub reading: Option, /// The applied threshold and the side of it that admits. pub bound: Bound, } impl Control { - /// Returns whether the control admits: evidence present, and inside its bound. + /// Returns whether a reading is present and satisfies its inclusive bound. + /// + /// A NaN reading fails either bound. pub(crate) fn admits(&self) -> bool { self.reading .is_some_and(|reading| self.bound.admits(reading)) @@ -257,8 +261,10 @@ impl Control { /// One probe's rendered evidence and verdict inputs. /// -/// The report carries the probe sizes and the applied thresholds, so a serialized report justifies -/// its own verdict without the configuration that produced it. +/// The report carries probe sizes and applied thresholds, permitting verdict recomputation without +/// the original configuration. These fields record results rather than prove their provenance. +/// Direct construction and deserialization do not check grid alignment, observation counts or +/// consistency between readings. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] pub(crate) struct QualityReport { /// Sampled anchor count. @@ -293,7 +299,7 @@ pub(crate) struct QualityReport { pub subgroups: Vec, /// Per-subgroup representation-baseline readings, ascending by ontology row. /// - /// The audit stratification, report-only. + /// Per-type representation-baseline readings over the sampled universe, report-only. pub baseline_subgroups: Vec, /// Degradation-rule breaches, in subgroup then neighbourhood order. pub flags: Vec, @@ -318,14 +324,16 @@ pub(crate) struct QualityReport { impl QualityReport { /// Returns the battery's controls, each carrying the reading its verdict turns on. /// - /// Each control is a conjunction over the neighbourhood steps, so the reading that decides it - /// is the extremum across them - the lowest step against a floor and the highest against a - /// ceiling. An absent reading is absent evidence - an empty grid, a step whose spread is - /// absent, triplet sampling switched off - and a control refuses that rather than passing - /// vacuously. + /// Neighbourhood floors use the lowest primary-grid reading and the intrusion ceiling uses the + /// highest. An empty primary grid yields absent readings. The density control is absent for an + /// empty density list or any row with no spread, otherwise it uses the maximum spread. Triplet + /// agreement is present only when its recorded triplet count is positive. + /// [`passes`](Self::passes) is the conjunction of these controls. /// - /// [`passes`](Self::passes) is this list's conjunction and an observer reports these same - /// numbers, so the verdict and the observation read one reduction instead of two. + /// These reductions do not check metric-row query counts or alignment between metric and + /// density steps. Density spreads must be finite: [`f64::max`] ignores a NaN operand, + /// permitting a non-finite row to leave a finite maximum or the initial negative infinity. The + /// controls report which readings exist, not whether those readings are sound. #[must_use] pub(crate) fn controls(&self) -> [Control; variant_count::()] { let lowest = |read: fn(&MetricRow) -> UnitFraction| { @@ -341,8 +349,7 @@ impl QualityReport { .reduce(UnitFraction::max) }; - // A step with no spread reading gives the ceiling nothing to check, so the control loses - // its evidence whole rather than reading the steps that do have one. + // every density step must supply a spread; an absent step invalidates the whole control let spread = self .density .iter() @@ -390,13 +397,11 @@ impl QualityReport { ] } - /// Returns whether the full battery admits the generation. + /// Returns whether every reduced metric satisfies its admission bound. /// - /// True exactly when every [control](Self::controls) holds: each reading present and inside its - /// bound. The controls are concrete validated values that stay maximally permissive by default. - /// The verdict therefore turns on evidence and readings, never on configuration shape. Subgroup - /// flags and their clump-resolution triage are report-only fields: they inform the human - /// reading the report and never affect admission. + /// True exactly when every [control](Self::controls) has a present reading inside its inclusive + /// bound. Subgroup flags and clump resolution never affect this verdict. The controls' presence + /// checks do not validate the report as a whole. #[must_use] pub(crate) fn passes(&self) -> bool { self.controls().iter().all(Control::admits) diff --git a/libs/@local/graph/atlas/src/salt/quality/report/live.rs b/libs/@local/graph/atlas/src/salt/quality/report/live.rs index ceb3af38227..db1608bd1c1 100644 --- a/libs/@local/graph/atlas/src/salt/quality/report/live.rs +++ b/libs/@local/graph/atlas/src/salt/quality/report/live.rs @@ -1,9 +1,10 @@ //! One live assessment of a root's active generation. //! -//! The assessment runs the whole suite against the store at the snapshot the generation's metadata -//! records, so artifact rows and store identities describe one corpus, and returns the verdict -//! together with the serialized evidence record. Probe sizing is the instrument's own: a live run -//! over a million rows affords sharper subgroup cells than the suite's own default sample. +//! The assessment reads the root's current-generation id once and returns that generation's report +//! and verdict under default quality thresholds. It queries the store using the recorded temporal +//! axes in a new repeatable-read transaction, whose snapshot can differ from the fit's earlier one +//! even under equal axes. The larger default anchor sample improves expected subgroup coverage, +//! without guaranteeing a count for any type. use core::{ error::Error, @@ -31,14 +32,13 @@ use crate::{ pub(crate) struct Options { /// The probe seed. /// - /// Equal seeds replay the sampling. + /// Uses zero by default. Equal seeds replay sampling when corpus row order and probe settings also match. pub seed: u64 = 0, /// Sampled anchor rows. /// - /// The suite default is 256; a live run over a million rows affords more for sharper subgroup - /// cells. + /// Uses 1,024 by default, increasing expected subgroup sample counts relative to the suite's 256-anchor default. pub anchors: NonZero = const { NonZero::new(1_024).unwrap() }, - /// Sampled comparison rows. + /// Sampled comparison rows, using 4,096 by default. pub comparisons: NonZero = const { NonZero::new(4_096).unwrap() }, } @@ -80,9 +80,9 @@ pub(crate) enum AssessError { Inactive, /// Opening the active generation failed. Generation(OpenError), - /// The generation records no snapshot axes, so no store state reproduces its corpus. + /// The generation records no temporal axes for the store query. Snapshot, - /// The store could not serve the recorded snapshot. + /// Opening the store's read-only repeatable-read transaction failed. Dataset(PostgresDatasetError), /// The quality run failed. Run(QualityRunError), @@ -121,12 +121,21 @@ impl Error for AssessError { } } -/// Assesses the root's active generation against the live store. +/// Assesses the generation selected by the root's current pointer. +/// +/// The assessment uses default quality thresholds and changes no activation state. Concurrent +/// activation can select another generation before the assessment returns. Dataset values must +/// still describe the fitted corpus for the readings to measure that generation's map fidelity. /// /// # Errors /// -/// Returns an [`AssessError`] when the generation fails to open, the store cannot serve its -/// snapshot, the run fails, or serializing the report fails. +/// Returns [`AssessError`] when the selected generation or its temporal axes are unavailable, the +/// dataset or quality run fails, or report serialization fails. +/// +/// # Panics +/// +/// Unchecked design or aggregate arithmetic can panic with integer overflow checks enabled, as +/// described by [`run`]. pub(crate) async fn assess( client: &mut Client, root: &GenerationRoot, diff --git a/libs/@local/graph/atlas/src/salt/quality/report/mod.rs b/libs/@local/graph/atlas/src/salt/quality/report/mod.rs index aad12c1c6aa..c36a7d8da41 100644 --- a/libs/@local/graph/atlas/src/salt/quality/report/mod.rs +++ b/libs/@local/graph/atlas/src/salt/quality/report/mod.rs @@ -1,49 +1,48 @@ //! Rendered probe evidence and the release verdict it supports. //! -//! [`assess`] turns one probe's readings into a flat, serializable record: every grid's whole-probe -//! metrics per neighbourhood size, per-subgroup readings on the primary grid, the subgroup flags -//! the degradation rule raises, and the thresholds the run applied. The report is a rendering of -//! [`ProbeReadings`] - regrouping or re-assessment starts from the readings, never from the report. +//! [`assess`] renders [`ProbeReadings`] as whole-probe metrics, per-type neighbourhood rows, +//! subgroup flags and the applied thresholds. Retain the per-anchor readings for regrouping into +//! different subgroups. The report contains aggregates, not individual anchor cells. //! -//! The primary fidelity surface is the corpus-exact map-versus-representation grid. The pass ranks -//! each sampled anchor against the full corpus, so the comparison universe carries no subsampling, -//! while the anchors themselves come from a sample, so aggregate means retain anchor-sampling -//! uncertainty. The thresholds bind there, against the observed probe statistic rather than a -//! population guarantee or lower confidence bound. The sampled grids provide context - the -//! canonical triangle that supplies the representation baseline for the map's canonical reading - -//! and stay report-only. +//! Neighbourhood thresholds apply to the corpus map-versus-representation grid. Its comparison +//! universe contains every non-anchor row, but the aggregate retains anchor-sampling uncertainty. +//! Thresholds check observed statistics rather than population guarantees or confidence bounds. +//! Sampled neighbourhood grids provide map-versus-canonical context alongside the representation +//! baseline and remain report-only. //! -//! Subgroups are entity types: an anchor contributes to one subgroup per direct type, so -//! multi-typed anchors count in each of their groups. A subgroup flags at a neighbourhood size when -//! its degradation - one minus recall - exceeds the configured factor times the whole-probe -//! degradation - twice, by the normative default. The rule raises a flag per neighbourhood size, so -//! the size trend - recall rising with the neighbourhood suggests near-tie reshuffling rather than -//! placement loss, evidence rather than a classifier - shows up in the flags and in every -//! subgroup's rows. Subgroups below the configured anchor floor never flag - a handful of anchors -//! cannot support a degradation ratio - but the report still carries their rows. +//! Subgroups are entity types. An anchor contributes once per direct-type list entry, including +//! repeated entries. With unique type lists, multi-typed anchors count once in each group. At each +//! neighbourhood size, a sufficiently sampled subgroup flags when dₛ > f · d, where dₛ and d are +//! one minus subgroup and whole-probe recall, and f is a finite configured factor (2 by default). +//! The subtraction and comparison use f64 readings. A NaN factor instead flags every sufficiently +//! sampled subgroup because the implementation tests the negation of dₛ ≤ f · d. Subgroups below +//! the anchor floor (8 by default) never flag, but keep their rows. This limits individual-anchor +//! leverage without establishing statistical significance. Recall increasing with neighbourhood +//! size can suggest boundary reshuffling, but does not classify its cause. //! -//! Density distortion reads the spread of log neighbourhood-radius ratios over the anchors, a -//! unit-free reading of uneven compression; the triplet rows read distance-order preservation over -//! the probe's shared pair sample for all three space pairs. The report renders both from the -//! readings like every other row. +//! Density distortion is the unscaled median absolute deviation of log neighbourhood-radius ratios +//! across anchors. With positive finite map radius rₘ and representation radius rᵣ, each log ratio +//! is ln(rₘ) − ln(rᵣ). Uniform radius rescaling adds a constant to all ratios and leaves the spread +//! unchanged in exact arithmetic. Triplet rows measure distance-order preservation over the shared +//! pair sample for every space pair. Admission checks map-versus-representation triplet agreement +//! and density spread alongside the neighbourhood metrics. //! -//! When the probe carries a clump grouping, the report adds the corpus reading collapsed onto clump -//! ids and re-evaluates every flag on those ids. A flag whose collapsed degradation satisfies the -//! same factor rule comes out clump-resolved, meaning the breach vanishes once recall counts by -//! component label. That is a triage diagnostic and nothing stronger, since chaining lets ε -//! single-linkage components reach arbitrary diameter, so resolution certifies neither component -//! compactness nor within-component placement. Subgroup flags and their resolution are report-only -//! either way, and they steer the human reading the report without affecting admission. +//! When the probe carries [`Clumps`](super::clump::Clumps), the report adds collapsed corpus and +//! representation-baseline recall. Each plain subgroup flag is clump-resolved when its collapsed +//! degradations satisfy dₛ ≤ f · d. This marks agreement after collapsing rows onto their +//! components. Single-linkage components can have diameter much greater than ε, and resolution +//! certifies neither compactness nor within-component placement. Subgroup flags and clump +//! resolution never affect admission. //! -//! The thresholds default to maximally permissive values - floors at zero, ceilings at their domain -//! edge - and the default verdict therefore turns on evidence presence rather than fidelity. An -//! invented floor would rest release verdicts on fiction. Deployments impose measured bounds -//! through the run's validated thresholds document. +//! Default thresholds accept all in-domain fidelity values while requiring readings. Configure +//! measured bounds through [`ThresholdOverrides`] for a stricter assessment. +//! [`QualityReport::controls`] defines the evidence-presence checks, which do not validate report +//! provenance or cross-field consistency. //! -//! The suite's instruments over published artifacts live beside the rendering they read: -//! [`calibration`] sweeps the clump threshold over a published k-NN table for the evidence behind -//! the grouping's default, and [`live`] runs one whole assessment of a root's active generation -//! against the store at the snapshot the generation records. +//! For measurements over published artifacts: +//! [`calibration`] sweeps the clump threshold over a published k-NN table. [`live`] assesses the +//! generation selected by a root's current pointer, querying the store with its recorded temporal +//! axes. use alloc::collections::BTreeMap; @@ -69,7 +68,15 @@ mod thresholds; /// Renders one probe's typed readings into a report under the thresholds. /// -/// Subgroup readings merge the per-anchor cells, so the report costs no ranking work. +/// Subgroup readings merge per-anchor cells without ranking work. `readings` must preserve +/// [`ProbeReadings`]' axis, shape and arithmetic-capacity requirements. Type-list entries become +/// subgroup memberships without deduplication. +/// +/// # Panics +/// +/// Panics on empty primary anchor or neighbourhood axes, out-of-domain grid indices, or +/// incompatible aggregate shapes. Inconsistent radius counts can also panic with integer overflow +/// checks enabled. #[must_use] pub(crate) fn assess( readings: TypedReadings<'_, N>, @@ -95,8 +102,7 @@ pub(crate) fn assess( .collect() }); - // Membership by ontology row; the map iterates ascending, so - // subgroups and flags order deterministically. + // ontology-row ordering fixes subgroup and flag order let mut members: BTreeMap> = BTreeMap::new(); for (anchor, types) in anchor_types.iter().enumerate() { for &ontology in types { @@ -172,8 +178,13 @@ pub(crate) fn assess( /// Merges subgroup memberships into per-type rows and degradation flags. /// -/// When clump readings exist, the same factor rule over the collapsed recalls re-evaluates every -/// breach on clump ids and decides the flag's resolution. +/// When clump readings exist, the same factor rule over collapsed recalls decides each flag's +/// resolution. `overall` and `clump_overall` must align with the neighbourhood axis. +/// +/// # Panics +/// +/// Panics for empty membership lists, invalid grid indices or incompatible aggregate shapes. Clump +/// readings require corresponding whole-probe clump aggregates. fn subgroup_reports( readings: &ProbeReadings, overall: &[MetricRow], @@ -204,8 +215,7 @@ fn subgroup_reports( continue; } - // The triage re-evaluation applies the same rule on clump - // ids, subgroup against the whole probe. + // compare collapsed subgroup degradation with collapsed whole-probe degradation let collapsed = readings.clumps.as_ref().map(|clumps| { let merged = clumps.map_representation.merged(anchors, step); let overall = &clump_overall @@ -240,9 +250,12 @@ fn subgroup_reports( /// Merges subgroup memberships into per-type representation-baseline rows. /// -/// Every row merges the sampled representation-versus-canonical cells of the subgroup's anchors, -/// plain and - when clump readings exist - collapsed, so the audit stratification and its triage -/// evidence travel together. +/// Each row compares plain sampled representation-versus-canonical recall with its collapsed +/// counterpart, when available. +/// +/// # Panics +/// +/// Panics for an empty membership list, an invalid grid index or incompatible aggregate shapes. fn baseline_subgroup_reports( readings: &ProbeReadings, members: &BTreeMap>, @@ -281,7 +294,16 @@ fn baseline_subgroup_reports( .collect() } -/// Reads each neighbourhood size's density distortion from the radii. +/// Computes each neighbourhood size's median log ratio and unscaled MAD. +/// +/// Positive radii contribute, and zero radii count as degenerate. Finite positive inputs are +/// required for finite log ratios. Infinity produced by overflowing distance arithmetic survives +/// the positivity filter. +/// +/// # Panics +/// +/// Panics with integer overflow checks enabled when a step has more contributing radius pairs than +/// anchors. fn density_rows(readings: &ProbeReadings) -> Vec { let steps = readings.neighbourhoods.len(); readings @@ -318,8 +340,6 @@ fn density_rows(readings: &ProbeReadings) -> Vec { .collect() } -// Statistics vocabulary promotes to a shared module on its second -// consumer. Median and MAD have one consumer. /// Returns the median, averaging the middle pair over even lengths. /// /// Empty input yields [`None`]. Sorts `values` in place. diff --git a/libs/@local/graph/atlas/src/salt/quality/report/thresholds.rs b/libs/@local/graph/atlas/src/salt/quality/report/thresholds.rs index 9f55d13ffbb..946a3558508 100644 --- a/libs/@local/graph/atlas/src/salt/quality/report/thresholds.rs +++ b/libs/@local/graph/atlas/src/salt/quality/report/thresholds.rs @@ -1,41 +1,46 @@ -//! Validated quality controls and their override document. +//! Admission bounds, subgroup flag settings and validated absolute-threshold overrides. use core::fmt; use crate::math::{NonNegative, UnitFraction, narrow_f32}; -// The degradation factor is normative: no important subgroup may suffer more than twice the overall -// degradation. The anchor floor bounds single-anchor leverage on a subgroup reading to one eighth, -// which is a sampling-noise floor and says nothing about which subgroups matter. +// the factor marks subgroups above twice the whole-probe degradation. With unique type memberships +// and at least eight anchors, each anchor contributes at most one eighth of subgroup recall. This +// limits leverage, not sampling uncertainty or subgroup importance. +/// The default ceiling on a subgroup's degradation relative to the overall degradation. const DEFAULT_DEGRADATION_FACTOR: f64 = 2.0; +/// The default anchor floor below which a subgroup reading is not judged. const DEFAULT_MINIMUM_SUBGROUP_ANCHORS: usize = 8; /// The maximally permissive density-spread ceiling. /// -/// A few hundred bounds the spread of `ln` radius ratios over f32 radii, so the f32 maximum imposes -/// no practical ceiling while the type keeps the value finite and non-negative by construction. +/// Positive finite f32 radii lie between 2⁻¹⁴⁹ and 2¹²⁸. Their log ratios have magnitude below 277 +/// · ln(2), and deviations from a median are below twice that bound. Therefore `f32::MAX` exceeds +/// every spread obtained from such radii while keeping the ceiling finite. const PERMISSIVE_DENSITY_SPREAD: NonNegative = NonNegative::new(f32::MAX).expect("the f32 maximum is finite and non-negative"); /// A quality-thresholds override document. /// -/// Each of the six absolute controls takes an optional field. A present field overrides its source -/// default after domain validation, an absent field keeps that default, and an unknown field -/// refuses the whole document. +/// Every absolute control takes an optional field, absent by default. A present field replaces the +/// current threshold after [`QualityThresholds::with_overrides`] validates its domain. An absent or +/// null field keeps the current value. Deserialization rejects unknown fields. #[derive(Debug, Copy, Clone, Default, serde::Deserialize)] #[serde(default, deny_unknown_fields)] pub(crate) struct ThresholdOverrides { - /// Overriding recall floor, in `[0, 1]`. + /// Recall floor in `[0, 1]`, absent by default. pub minimum_recall: Option, - /// Overriding trustworthiness floor, in `[0, 1]`. + /// Trustworthiness floor in `[0, 1]`, absent by default. pub minimum_trustworthiness: Option, - /// Overriding continuity floor, in `[0, 1]`. + /// Continuity floor in `[0, 1]`, absent by default. pub minimum_continuity: Option, - /// Overriding intrusion-rate ceiling, in `[0, 1]`. + /// Intrusion-rate ceiling in `[0, 1]`, absent by default. pub maximum_intrusion_rate: Option, - /// Overriding density-spread ceiling, finite and non-negative. + /// Density-spread ceiling in the finite non-negative f32 range, absent by default. + /// + /// The input validates as f64 before rounding to f32. pub maximum_density_spread: Option, - /// Overriding triplet-agreement floor, in `[0, 1]`. + /// Triplet-agreement floor in `[0, 1]`, absent by default. pub minimum_triplet_agreement: Option, } @@ -64,66 +69,73 @@ impl core::error::Error for ThresholdDomainError {} /// The thresholds of one assessment. /// -/// Every control is a concrete validated value. Floors and ceilings apply to the corpus -/// map-versus-representation grid at every neighbourhood size, and every control stays pinned -/// because [`QualityReport::passes`](super::QualityReport::passes) compares all six controls and -/// demands their evidence. +/// Absolute neighbourhood bounds apply to every step of the corpus map-versus-representation grid. +/// The remaining admission controls check density spread and sampled map-versus-representation +/// triplet agreement. [`QualityReport::passes`](super::QualityReport::passes) requires all controls +/// to have readings within their inclusive bounds. /// -/// Every default is the most permissive value in its control's domain, so the default verdict gates -/// evidence presence rather than fidelity. Deployments impose measured bounds through an override -/// document that replaces individual defaults after domain validation ([`ThresholdOverrides`]). -// No serde derives. The derive macros cannot parse default field values, so the report carries the -// applied thresholds as typed fields instead of embedding this struct. +/// Absolute defaults accept every in-domain metric value while requiring readings. +/// [`ThresholdOverrides`] replaces selected bounds after domain validation. Subgroup settings +/// govern report-only flags, and their raw factor and count fields have no validation here. +// serde derives cannot parse default field values; QualityReport serializes applied thresholds as +// individual fields #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct QualityThresholds { /// Minimum recall floor. /// - /// The permissive zero imposes none. + /// Uses zero by default, accepting every in-domain recall. pub minimum_recall: UnitFraction = UnitFraction::ZERO, /// Minimum trustworthiness floor. /// - /// The permissive zero imposes none. + /// Uses zero by default, accepting every in-domain trustworthiness reading. pub minimum_trustworthiness: UnitFraction = UnitFraction::ZERO, /// Minimum continuity floor. /// - /// The permissive zero imposes none. + /// Uses zero by default, accepting every in-domain continuity reading. pub minimum_continuity: UnitFraction = UnitFraction::ZERO, /// Maximum intrusion-rate ceiling. /// - /// The permissive one imposes none. + /// Uses one by default, accepting every in-domain rate. pub maximum_intrusion_rate: UnitFraction = UnitFraction::ONE, /// Maximum density-distortion spread. /// - /// The permissive `f32` maximum imposes no ceiling. The ceiling fails when the reading is - /// absent - a demand for evidence that was never produced is a configuration contradiction, - /// surfaced at the verdict. + /// Uses `f32::MAX` by default, accepting every spread from finite positive f32 radii. The control still fails when any density step has no reading. pub maximum_density_spread: NonNegative = PERMISSIVE_DENSITY_SPREAD, /// Minimum map-versus-representation triplet agreement floor. /// - /// The permissive zero imposes no floor. The control still fails when triplet sampling is off. + /// Uses zero by default, accepting every in-domain agreement. The control still fails when triplet sampling is off. pub minimum_triplet_agreement: UnitFraction = UnitFraction::ZERO, - /// A subgroup flags when its degradation exceeds this factor times the overall degradation. + /// Subgroup degradation multiplier for report-only flags. /// - /// `2` is the normative subgroup rule: no important subgroup may suffer more than twice the - /// overall degradation. + /// Uses 2 by default. A subgroup meeting the anchor floor flags unless its degradation is at most this factor times whole-probe degradation. The factor has no domain validation, and a NaN comparison therefore flags. pub subgroup_degradation_factor: f64 = DEFAULT_DEGRADATION_FACTOR, /// Subgroups with fewer anchors never flag. + /// + /// Uses 8 by default. pub minimum_subgroup_anchors: usize = DEFAULT_MINIMUM_SUBGROUP_ANCHORS, } impl QualityThresholds { /// Applies an override document over these thresholds. /// - /// A present field replaces its default after domain validation, and an absent field keeps the - /// default. + /// A present field replaces the current value after domain validation, and an absent field + /// preserves it. The density ceiling rounds to f32 after validation in f64, including positive + /// underflow to zero. /// /// # Errors /// - /// Returns the first override whose value lies outside its control's domain. + /// Returns [`ThresholdDomainError`] for the first out-of-domain override. Validation order is + /// recall, trustworthiness, continuity, intrusion rate, triplet agreement and density spread. pub(crate) fn with_overrides( mut self, overrides: &ThresholdOverrides, ) -> Result { + /// Overrides `into` with `value` when one is given. + /// + /// # Errors + /// + /// Returns a [`ThresholdDomainError`] attributed to `field` when `value` lies outside the + /// closed unit interval. const fn fraction( field: &'static str, value: Option, @@ -166,9 +178,9 @@ impl QualityThresholds { &mut self.minimum_triplet_agreement, )?; if let Some(value) = overrides.maximum_density_spread { - // The domain check runs on the f64 value before narrowing. A negative underflow narrows - // to -0.0 and a value barely above the f32 maximum rounds down onto it, so both must - // refuse as written rather than as rounded. + // Rounding can map a negative underflow to -0.0 or a value above f32::MAX onto that + // boundary. The check uses the original f64 value before narrowing. Therefore both + // out-of-domain cases refuse even if their rounded values would pass. let admitted = (value.is_finite() && value >= 0.0 && value <= f64::from(f32::MAX)) .then(|| narrow_f32(value)) .flatten() diff --git a/libs/@local/graph/atlas/src/salt/quality/runner.rs b/libs/@local/graph/atlas/src/salt/quality/runner.rs index 7955a177c14..f6acbabb3d2 100644 --- a/libs/@local/graph/atlas/src/salt/quality/runner.rs +++ b/libs/@local/graph/atlas/src/salt/quality/runner.rs @@ -1,14 +1,14 @@ //! One quality probe over a published generation. //! -//! [`run`] wires the suite end to end. It opens the generation's mapped artifacts (the k-NN table -//! for the clump grouping, the representation matrix, the coordinate frame, and the node -//! identities) and probes them against the dataset's canonical space. It resolves the sampled -//! anchors' direct types through the dataset's probe-scoped type stream, then renders the readings -//! into a [`QualityReport`] under the given thresholds. +//! [`run`] opens a generation's k-NN table, representation matrix, coordinate frame and node +//! identities. It groups stored near-duplicate edges and probes neighbourhood fidelity against +//! dataset canonical embeddings. After resolving the anchors' direct types, it returns a +//! [`QualityReport`] under the configured thresholds. //! -//! The dataset must observe the snapshot the fit read (the generation's metadata records the axes), -//! because the runner matches artifact rows to source identities through the identity artifact and -//! a dataset at other axes would resolve types for a different corpus. +//! The dataset must supply canonical embeddings and type memberships consistent with the fitted +//! corpus. Matching source ids and counts leaves those values unverified. Recorded temporal axes +//! select the same query parameters, but do not by themselves restore the fit's database snapshot. +//! The report is returned in memory without changing activation. use hashql_core::id::IdSlice; use rand::Rng; @@ -34,11 +34,11 @@ use crate::{ /// Sampling, grouping, and threshold settings for one quality run. #[derive(Debug, Clone, PartialEq)] pub(crate) struct QualityRunOptions { - /// The probe's sampling and neighbourhood settings. + /// Sampling settings, using [`ProbeOptions::default`] by default. pub probe: ProbeOptions = ProbeOptions::default(), - /// The report's thresholds. + /// Thresholds, using the permissive [`QualityThresholds::default`] by default. pub thresholds: QualityThresholds = QualityThresholds::default(), - /// The clump grouping's distance threshold. + /// Clump cosine-distance threshold, using [`DEFAULT_EPSILON`](super::clump::DEFAULT_EPSILON) (0.002) by default. pub epsilon: f32 = super::clump::DEFAULT_EPSILON, } @@ -50,16 +50,24 @@ const impl Default for QualityRunOptions { /// Probes a published generation and reports its map fidelity. /// -/// The generation's artifacts are read from their whole-file mappings; nothing is copied onto the -/// heap beyond the probe's own bounded scratch. The dataset serves two probe-scoped streams - -/// canonical embeddings for the sampled rows, direct types for the anchors - and must observe the -/// snapshot recorded in the generation's metadata. +/// Artifact arrays borrow whole-file mappings. Clump construction allocates corpus-sized labels and +/// working state, in addition to probe scratch, sampled canonical payloads and report data. Backing +/// files must remain immutable while mapped. The dataset supplies canonical embeddings for sampled +/// rows and direct types for anchors, consistent with the fitted corpus. +/// +/// [`probe`] defines numerical and design-capacity requirements beyond artifact layout and +/// row-count checks. In particular, finite coordinates do not guarantee finite squared distances, +/// and the representation matrix's layout check does not validate its components. /// /// # Errors /// -/// Returns an error when an artifact cannot be opened or does not hold its role's layout, the -/// artifacts disagree about the corpus row count, the probe design cannot run over the corpus, or a -/// dataset stream fails or misdelivers. +/// Returns [`QualityRunError`] for an invalid or unreadable artifact, mismatched row counts, a +/// failed probe or mismatched type delivery. +/// +/// # Panics +/// +/// Unchecked design or aggregate arithmetic can panic when integer overflow checks are enabled, as +/// described by [`probe`]. pub(crate) async fn run( dataset: &D, generation: &Generation, diff --git a/libs/@local/graph/atlas/src/salt/quality/tests.rs b/libs/@local/graph/atlas/src/salt/quality/tests.rs index 8b86a37f8c0..5a04f74574f 100644 --- a/libs/@local/graph/atlas/src/salt/quality/tests.rs +++ b/libs/@local/graph/atlas/src/salt/quality/tests.rs @@ -57,10 +57,10 @@ use crate::{ }, }; -/// A six-row, two-neighbour table. +/// Builds a six-row table with two stored neighbours per row. /// -/// A chained near-duplicate triple {0, 1, 2}, an exact-duplicate pair {3, 4}, and a far singleton -/// 5. +/// Rows {0, 1, 2} connect at ε = 0.05 through a chain, {3, 4} share a zero-distance edge, and row 5 +/// is distant. fn clump_fixture() -> Knn { let indptr: Vec = vec![0, 2, 4, 6, 8, 10, 12]; let indices: Vec = vec![1, 2, 0, 2, 0, 1, 4, 5, 3, 5, 3, 4]; @@ -144,16 +144,12 @@ fn clump_threshold_is_inclusive_and_zero_keeps_exact_duplicates() { exact.clump(NodeRowId::new(1)) ); - // A non-finite threshold admits no edges. + // comparisons against NaN admit no edges let none = Clumps::from_knn(&table.view(), f32::NAN); assert_eq!(none.clumps(), 6); assert_eq!(none.groups(), 0); } -/// The default threshold groups at duplicate scale. -/// -/// The fixture's coincident pair joins while its 0.05-distant chain - twenty-five defaults wide - -/// stays apart. #[test] fn default_epsilon_groups_duplicates_not_neighbours() { let table = clump_fixture(); @@ -184,9 +180,7 @@ fn hand_built_labels_read_like_a_grouping() { assert_eq!(clumps.clump(NodeRowId::new(3)), 2); } -/// Hand-computed multiset overlap. -/// -/// The duplicated label 1 matches twice, 0 once, and the unmatched 2 and 3 earn nothing. +// label 1 matches twice, label 0 once, and unmatched labels 2 and 3 earn no credit #[test] fn clump_aggregate_counts_multiset_overlap() { let mut aggregate = ClumpAggregate::new(NonZero::new(4).expect("nonzero")); @@ -209,7 +203,6 @@ fn clump_aggregate_counts_multiset_overlap() { ); } -/// Both spaces order the universe alike. #[test] fn identical_orderings_are_perfect() { let ordering: Vec = (0..10).collect(); @@ -227,7 +220,6 @@ fn identical_orderings_are_perfect() { assert_eq!(aggregate.extrusion_rate(), 0.0); } -/// A reversed ordering is the worst permutation at every valid k. #[test] fn reversed_ordering_is_worst() { let reference: Vec = (0..8).collect(); @@ -240,8 +232,9 @@ fn reversed_ordering_is_worst() { // Map top-2 = {7, 6}: reference positions 7 and 6, both false. assert_eq!(aggregate.recall(), 0.0); - // Penalties (7-2+1) + (6-2+1) = 11 = the worst case 2*(16-6+1)/2, - // so both normalized readings sit at the floor exactly. + // The penalty is (7 − 2 + 1) + (6 − 2 + 1) = 11, equal to the worst case 2 · (16 − 6 + 1)/2. + // Dividing equal integers gives exactly one. Therefore both normalized readings are exactly + // zero. assert_eq!(aggregate.trustworthiness(), 0.0); assert_eq!(aggregate.continuity(), 0.0); // Both false neighbours lie past the horizon in both directions. @@ -249,7 +242,6 @@ fn reversed_ordering_is_worst() { assert_eq!(aggregate.extrusion_rate(), 1.0); } -/// Hand-computed mixed case: one shared neighbour, one mild swap. #[test] fn hand_computed_partial_agreement() { // Universe of 6. Reference: 0,1,2,3,4,5. Map: 0,2,1,3,4,5. @@ -274,7 +266,6 @@ fn hand_computed_partial_agreement() { assert_eq!(aggregate.extrusion_rate(), 0.0); } -/// The horizon separates near-boundary reshuffles from intruders. #[test] fn horizon_splits_reshuffles_from_intruders() { // Map top-2 = {0, 5}. Point 5 sits at reference position 5, past the horizon 4, and the swap @@ -295,7 +286,6 @@ fn horizon_splits_reshuffles_from_intruders() { assert_eq!(aggregate.continuity(), 1.0 - 4.0 / 7.0); } -/// Aggregation over queries averages penalties, not readings. #[test] fn aggregate_pools_queries() { let reference: Vec = (0..6).collect(); @@ -313,7 +303,6 @@ fn aggregate_pools_queries() { assert_eq!(aggregate.trustworthiness(), 1.0 - 1.0 / 14.0); } -/// The constructor rejects the domains the normalizer excludes. #[test] fn aggregate_rejects_invalid_shapes() { let k = |value: usize| NonZero::new(value).expect("nonzero"); @@ -324,11 +313,10 @@ fn aggregate_rejects_invalid_shapes() { assert!(NeighbourhoodAggregate::new(10, k(5), 10).is_some()); } -/// The clamped constructor caps an overflowing horizon at the universe. #[test] fn clamped_horizon_overflow() { - // A factor-times-k product beyond `usize` exceeds every universe, so - // the mathematical horizon min(factor · k, universe) is the universe. + // the overflowing product 2 · usize::MAX exceeds the universe 4. Saturating before clamping + // preserves the mathematical horizon min(factor · k, universe) = 4. let k = |value: usize| NonZero::new(value).expect("nonzero"); let factor = k(usize::MAX); @@ -338,7 +326,6 @@ fn clamped_horizon_overflow() { ); } -/// The capacity proof refuses a load whose worst-case penalties overflow the carriers. #[test] fn capacity_refusal() { // A valid aggregate whose total worst-case penalty q·k·(2m − 3k + 1)/2 @@ -351,7 +338,6 @@ fn capacity_refusal() { assert!(!aggregate.supports(1_200_000)); } -/// The capacity proof admits exactly the loads whose products fit. #[test] #[expect( clippy::integer_division, @@ -359,9 +345,8 @@ fn capacity_refusal() { reason = "the boundary load is the largest whose product with the worst-case penalty fits" )] fn capacity_boundary() { - // At universe size m = 100 and neighbourhood size k = 50 the worst - // per-query penalty is 50·(200 − 150 + 1)/2 = 1275, so the boundary - // is the largest load q whose normalizer product still fits. + // at m = 100 and k = 50 the worst per-query penalty is 50 · (200 − 150 + 1)/2 = 1,275. + // The largest supported load is ⌊usize::MAX/1,275⌋. let k = |value: usize| NonZero::new(value).expect("nonzero"); let aggregate = NeighbourhoodAggregate::new(100, k(50), 100).expect("50 <= 100 / 2 and 50 <= 100 <= 100"); @@ -371,12 +356,8 @@ fn capacity_boundary() { assert!(!aggregate.supports(heaviest + 1)); } -/// Sampled pairs are distinct and in bounds over every small universe. #[test] fn sampled_pairs_are_distinct_and_in_bounds() { - // A one-point universe holds no pairs, and an empty one holds no - // point to draw first - the second draw's universe is one smaller - // than the first's, so neither may reach for it. assert!(sample_pairs(Xoshiro256PlusPlus::seed_from_u64(3), 1, 64).is_empty()); assert!(sample_pairs(Xoshiro256PlusPlus::seed_from_u64(4), 0, 1).is_empty()); @@ -397,13 +378,12 @@ fn sampled_pairs_are_distinct_and_in_bounds() { seen.insert([first, second]); } - // 512 seeded draws over at most 30 ordered pairs cover the whole - // support, pinning uniformity's reach alongside its bounds. + // these seeded draws cover all possible ordered pairs; coverage alone does not test uniform + // frequencies assert_eq!(seen.len(), comparisons * (comparisons - 1)); } } -/// Rank-vector observation agrees with full-ordering observation. #[test] fn observe_ranks_matches_observe() { // Universe of 8, k = 3, horizon 5, with tangled orderings. @@ -426,7 +406,6 @@ fn observe_ranks_matches_observe() { assert_eq!(through_orderings, through_ranks); } -/// Merging per-query aggregates equals one joint observation. #[test] fn merged_aggregates_match_joint_observation() { let reference: Vec = (0..6).collect(); @@ -448,16 +427,15 @@ fn merged_aggregates_match_joint_observation() { assert_eq!(first, joint); } -/// Rows the probe fixture's aligned backing store can hold. +/// Component capacity for 48 projector rows in the aligned fixture store. const FIXTURE_CAPACITY: usize = 48 * PROJECTOR_DIMENSIONS; -/// A probe corpus whose three spaces share one deterministic geometry. +/// A probe corpus with independently chosen embedding and map angles. /// -/// Row `i` sits at an angle on the unit circle in every space: the representation is `(cos, sin)` -/// in the leading two components, the canonical embedding extends it with zeros, and the -/// coordinates are the circle point itself. Chord length and cosine distance are both monotone in -/// the angular gap, so equal embedding and map angles make all three spaces order every universe -/// identically - a perfect map. +/// Representations and canonical embeddings store `(cos, sin)` in their leading components +/// and zeros elsewhere. Coordinates use the map angles. With identical angles, chord length +/// and cosine distance increase together with the angular gap in `[0, π]` in exact arithmetic. +/// This motivates the shared-angle fixtures without guaranteeing identical floating-point ranks. pub(crate) struct ProbeFixture { node_ids: Vec, storage: BoxedVecN, @@ -467,11 +445,24 @@ pub(crate) struct ProbeFixture { } impl ProbeFixture { + /// Builds a fixture with identical embedding and map angles. + /// + /// Angles must be finite. + /// + /// # Panics + /// + /// Panics when `angles` contains more than 48 rows. pub(crate) fn on_circle(angles: &[f32]) -> Self { Self::new(angles, angles) } /// Places the embeddings at `angles` and the map at `map_angles`. + /// + /// Both angle slices must be finite. + /// + /// # Panics + /// + /// Panics when the slices differ in length or contain more than 48 rows. fn new(angles: &[f32], map_angles: &[f32]) -> Self { assert_eq!(angles.len(), map_angles.len()); assert!(angles.len() * PROJECTOR_DIMENSIONS <= FIXTURE_CAPACITY); @@ -500,6 +491,7 @@ impl ProbeFixture { } } + /// Copies the fixture's canonical embeddings into an in-memory dataset. pub(crate) fn dataset(&self) -> MemoryDataset { MemoryDataset::new( Vec::new(), @@ -510,11 +502,13 @@ impl ProbeFixture { ) } + /// Borrows the populated projector rows from aligned fixture storage. fn representations(&self) -> &[AlignedVecN] { AlignedVecN::from_slice(&self.storage.as_array()[..self.rows * PROJECTOR_DIMENSIONS]) .expect("boxed storage is aligned") } + /// Borrows the row-aligned fixture inputs as a probe corpus. pub(crate) fn corpus(&self) -> ProbeCorpus<'_, MemoryNodeId> { ProbeCorpus::new( IdSlice::from_raw(&self.node_ids), @@ -524,9 +518,10 @@ impl ProbeFixture { } } -/// Irregularly spaced angles inside a quarter circle. +/// Generates mildly warped angles inside a quarter circle. /// -/// No two gaps coincide, so no space carries distance ties. +/// The quadratic spacing reduces symmetry in small fixtures. It does not guarantee distinct gaps or +/// floating-point distances for every row count. pub(crate) fn irregular_angles(rows: usize) -> Vec { #[expect( clippy::cast_precision_loss, @@ -588,7 +583,6 @@ fn indices_of(rows: &[NodeRowId]) -> Vec { rows.iter().map(|row| row.as_usize()).collect() } -/// The corpus pass's counted ranks agree with sorted full orderings. #[tokio::test] async fn corpus_readings_match_a_sorting_reference() { // Embeddings on the circle, coordinates scrambled by reversing the @@ -712,8 +706,11 @@ async fn corpus_readings_match_a_sorting_reference() { /// Asserts the collapsed reading reproduces plain recall for the first `anchors` anchors. /// -/// Under singleton labels the collapse is the identity, so any disagreement is a defect in the -/// collapse itself rather than in the grouping. +/// Singleton labels preserve row identity. +/// +/// # Panics +/// +/// Panics on a recall mismatch, an out-of-domain anchor, or a missing first step. #[track_caller] fn assert_singleton_collapse_matches_plain_recall( clumps: &ClumpReadings, @@ -735,14 +732,10 @@ fn assert_singleton_collapse_matches_plain_recall( } } -/// The probe's clump collapse. -/// -/// Singleton labels reproduce plain recall exactly, and a grouped labelling agrees with a sorting -/// reference collapsed the same way while never reading below plain recall. #[tokio::test] async fn clump_readings_match_a_sorting_reference() { - // The scrambled fixture from the corpus reference test: map and - // representation disagree, so the collapse has work to do. + // reversing and rescaling the map angles introduces disagreement with the representation. + // Separate coincident pairs exercise distance ties in each space. let angles = irregular_angles(40); let mut map_angles: Vec = angles .iter() @@ -762,8 +755,8 @@ async fn clump_readings_match_a_sorting_reference() { ..ProbeOptions::default() }; - // Singleton labels: the multiset overlap is the shared-row count, - // so the collapsed reading equals plain recall anchor by anchor. + // singleton labels preserve row identity: multiset overlap equals the shared-row count for + // each anchor let singletons = Clumps::::from_labels((0..40).collect(), 0.0); let readings = probe( &fixture.dataset(), @@ -848,10 +841,10 @@ async fn clump_readings_match_a_sorting_reference() { } } -/// Hand-built readings. +/// Builds one hit-or-miss cell per supplied anchor for report fixtures. /// -/// Six single-query anchors at k = 1 over a universe of 8, each a plain hit (rank 0) or a horizon -/// miss (rank 7), reused for all four grids. +/// Each anchor observes k = 1 over a universe of 8, with rank 0 for a hit or rank 7 for a horizon +/// miss. All four grids reuse those cells. fn flag_fixture(hits: &[bool]) -> ProbeReadings { let cells: Vec> = hits .iter() @@ -888,21 +881,27 @@ fn flag_fixture(hits: &[bool]) -> ProbeReadings { } } -/// One preserved triplet observation. -/// -/// The verdict demands present triplet evidence, so the fixture carries the minimum. +/// Creates the single preserved triplet needed for present fixture evidence. fn agreed() -> TripletAggregate { let mut aggregate = TripletAggregate::default(); aggregate.observe(true); aggregate } -/// A `[0, 1]` control value. +/// Validates a fixture threshold fraction. +/// +/// # Panics +/// +/// Panics when `value` is outside `[0, 1]` or NaN. fn fraction(value: f64) -> UnitFraction { UnitFraction::new(value).expect("test fractions lie inside [0, 1]") } -/// A density-spread ceiling. +/// Narrows a fixture density ceiling to f32 and validates it. +/// +/// # Panics +/// +/// Panics when the narrowed value is negative or non-finite. #[expect( clippy::cast_possible_truncation, reason = "test ceilings are small round values the f32 range carries exactly enough" @@ -911,15 +910,13 @@ fn ceiling(value: f64) -> NonNegative { NonNegative::new(value as f32).expect("test ceilings are finite and non-negative") } +/// Builds per-anchor type lists from literal ontology row numbers. fn types_of(rows: &[&[u64]]) -> Vec> { rows.iter() .map(|types| types.iter().map(|&row| OntologyRowId::new(row)).collect()) .collect() } -/// Hand-computed subgroup rule. -/// -/// Degradations, the 2x factor, the anchor floor, and multi-typed anchors counting in every group. #[test] fn assess_flags_degraded_subgroups() { // Anchors 0-3 hit, 4-5 miss: overall recall 2/3, degradation 1/3. @@ -995,9 +992,9 @@ fn assess_flags_degraded_subgroups() { assert!(floored.passes()); } -/// A clump readings block over the flag fixture. +/// Builds collapsed fixture cells from per-anchor match flags. /// -/// One k = 1 cell per anchor, its collapsed neighbourhood matched or not. +/// Each flag supplies one k = 1 cell. fn clump_readings_of(matches: &[bool]) -> ClumpReadings { let cells: Vec> = matches .iter() @@ -1018,10 +1015,6 @@ fn clump_readings_of(matches: &[bool]) -> ClumpReadings { } } -/// The clump-resolution triage rule. -/// -/// A flag whose collapsed reading satisfies the factor counts as clump-resolved and stops failing -/// the verdict. One that stays degraded keeps failing. #[test] fn clump_resolution_triages_flags() { let anchor_types = types_of(&[&[100], &[100], &[100], &[100], &[200], &[200]]); @@ -1030,8 +1023,8 @@ fn clump_resolution_triages_flags() { .. }; - // Restored: the misses were clump siblings, so every collapsed - // neighbourhood matches and both degradations read zero. + // the collapsed cells model every miss replaced by a same-clump sibling. All six match, + // giving zero subgroup and whole-probe degradation. let mut readings = flag_fixture(&[true, true, true, true, false, false]); readings.clumps = Some(clump_readings_of(&[true; 6])); let report = assess(readings.with_anchor_types(&anchor_types), &thresholds); @@ -1054,8 +1047,8 @@ fn clump_resolution_triages_flags() { assert_eq!(clumps.map_representation[0].neighbourhood.get(), 1); assert_eq!(clumps.map_representation[0].queries, 6); assert_eq!(clumps.map_representation[0].recall, 1.0); - // The fixture reuses the collapsed cells for the baseline grid, so - // its rendered rows and the subgroup stratification read the same. + // reusing the collapsed cells in the baseline grid gives equal whole-probe and per-type + // readings assert_eq!(clumps.representation_canonical, clumps.map_representation); assert_eq!(report.baseline_subgroups.len(), 2); assert_eq!( @@ -1068,8 +1061,8 @@ fn clump_resolution_triages_flags() { serde_json::from_str(&serialized).expect("the report deserializes"); assert_eq!(roundtrip, report); - // Unresolved: the collapse restores nothing, so the flag keeps - // its breach - 1 against twice the overall 1 - 4/6. + // leaving the collapsed cells unchanged preserves the breach: subgroup degradation 1 + // exceeds 2 · (1 − 4/6) = 2/3 let mut readings = flag_fixture(&[true, true, true, true, false, false]); readings.clumps = Some(clump_readings_of(&[true, true, true, true, false, false])); let report = assess(readings.with_anchor_types(&anchor_types), &thresholds); @@ -1091,9 +1084,6 @@ fn clump_resolution_triages_flags() { assert!(report.passes()); } -/// Hand-computed density rows. -/// -/// The rows cover log ratios, the median/MAD spread, and degenerate-radius exclusion. #[test] fn assess_reads_density_from_radii() { let mut readings = flag_fixture(&[true, true, true]); @@ -1153,7 +1143,6 @@ fn assess_reads_density_from_radii() { assert!(!strict.passes()); } -/// Pinned thresholds on absent evidence fail closed. #[test] fn assess_fails_pinned_thresholds_without_evidence() { // Every radius degenerate: the density reading is absent. @@ -1192,13 +1181,8 @@ fn assess_fails_pinned_thresholds_without_evidence() { assert!(!report.passes()); } -/// The neighbourhood controls demand a nonempty grid. -/// -/// `all` over an empty grid is vacuously true. The verdict must not be. `assess` cannot emit an -/// empty grid (it reads step 0 unconditionally and panics), but the report is a serializable value -/// whose verdict must hold under every construction - persisted reports get read back, and a -/// control over zero steps is the same evidence absence as a density ceiling over absent readings, -/// failing the same way. +// reports may be constructed or deserialized independently of assess, including with an empty +// primary grid #[test] fn neighbourhood_controls_demand_a_nonempty_grid() { let readings = flag_fixture(&[true, true]); @@ -1217,10 +1201,6 @@ fn neighbourhood_controls_demand_a_nonempty_grid() { ); } -/// Override documents validate at the boundary. -/// -/// A present field overrides its default after domain validation, an absent field keeps the -/// default, an out-of-domain value names its field, and an unknown field refuses the document. #[test] fn threshold_overrides_validate_at_the_boundary() { let overrides: ThresholdOverrides = @@ -1297,7 +1277,6 @@ fn threshold_overrides_validate_at_the_boundary() { ); } -/// Floors bind the overall corpus readings. #[test] fn assess_applies_pinned_floors() { let readings = flag_fixture(&[true, true, false]); @@ -1324,7 +1303,6 @@ fn assess_applies_pinned_floors() { assert!(lenient.passes()); } -/// The report wires from a live probe and survives serialization. #[tokio::test] async fn assess_reads_a_probed_fixture() { let fixture = ProbeFixture::on_circle(&irregular_angles(48)); @@ -1364,9 +1342,8 @@ async fn assess_reads_a_probed_fixture() { assert_eq!(report.subgroups.len(), 1); assert_eq!(report.subgroups[0].anchors, 5); assert_eq!(report.subgroups[0].rows[0].recall, 1.0); - // Every space orders the circle identically, so all triplet pairs - // agree; the metric warp between chord and cosine distance keeps - // the density reading present and finite. + // on this sampled circle, chord and cosine distances give equal neighbour orderings. The + // distinct points give positive finite radii, with varying ratios between the two metrics. assert_eq!( report.triplet_map_representation.agreement, UnitFraction::ONE @@ -1453,8 +1430,16 @@ async fn probe_rejects_impossible_designs() { ); } +/// Row count of the runner fixture corpus. const RUNNER_NODES: usize = 48; +/// Returns a per-process fixture path after attempting to remove prior contents. +/// +/// Removal errors are ignored, and the directory is not created here. +/// +/// # Panics +/// +/// Panics when the system temporary directory has no UTF-8 path. fn runner_scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -1466,7 +1451,7 @@ fn runner_scratch(name: &str) -> Utf8PathBuf { dir } -/// A probe-scale corpus for publishing through the real fit. +/// Builds a 48-node corpus for publishing through the real fit. /// /// Unit-norm pseudo-random representations whose canonical embeddings extend them with zeros, one /// node type alternating between two ontology rows, and one link type. @@ -1577,12 +1562,15 @@ impl CardEmbedder for HashEmbedder { } } -/// A deterministic classifier fitted from a synthetic corpus. +/// Fits a supplied classifier from a synthetic training corpus. +/// +/// # Panics /// -/// The supplied model input of the fixture fit. +/// Panics if the training set or classifier fit fails. fn runner_classifier() -> ClassifierInput { const ROWS: usize = 4; - // Coprime to the dimension, so no two corpus rows repeat. + // The 13-component pattern and 3,072-component row width are coprime. Successive rows begin at + // distinct pattern offsets for these four rows. Therefore no two fixture rows repeat. const PATTERN: [f32; 13] = [ -0.75, -0.625, -0.5, -0.375, -0.25, -0.125, 0.0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, ]; @@ -1630,9 +1618,7 @@ fn runner_classifier() -> ClassifierInput { } } -/// The runner fixture's probe design. -/// -/// A handful of anchors and comparisons sized to the 48-row corpus. +/// Chooses probe counts and neighbourhoods that fit the 48-row runner corpus. fn runner_probe_options() -> QualityRunOptions { QualityRunOptions { probe: ProbeOptions { @@ -1649,10 +1635,6 @@ fn runner_probe_options() -> QualityRunOptions { } } -/// The runner end to end. -/// -/// The real fit publishes a generation. The runner reopens its artifacts and probes them against -/// the same dataset, then resolves anchor types and reports. #[tokio::test] async fn runner_reports_a_published_generation() { let path = runner_scratch("runner"); @@ -1702,7 +1684,7 @@ async fn runner_reports_a_published_generation() { .await .expect("the run should produce a report"); - // The probe design landed as configured. + // the report records the requested probe design assert_eq!(report.anchors, 8); assert_eq!(report.corpus_universe, RUNNER_NODES - 8); assert_eq!(report.comparisons, 16); @@ -1726,8 +1708,8 @@ async fn runner_reports_a_published_generation() { "only the two node types carry anchors", ); - // Zero-extended canonical embeddings rank identically to the - // representation, so the baseline reads as perfect. + // appending zero components preserves the dot products and norms for these vectors, giving + // equal canonical and representation distance orderings for row in &report.sampled_representation_canonical { assert_eq!(row.recall, 1.0); } @@ -1764,9 +1746,8 @@ async fn runner_reports_a_published_generation() { } } - // Subgroups below the default anchor floor never flag, and the verdict still refuses. Step 2 of - // this landmark-baseline fixture reads all-degenerate radii, so the density evidence is absent - // there and the control fails closed on absence, permissive ceilings included. + // the k=2 step has no density spread when every anchor has a zero radius in at least one space. + // Missing density evidence rejects even under the permissive ceiling. assert!(report.flags.is_empty()); assert!( report.density[0].spread.is_none(), @@ -1777,9 +1758,6 @@ async fn runner_reports_a_published_generation() { #[test] fn every_metric_is_listed_once_under_the_noun_its_threshold_is_keyed_by() { - // `ALL` transmutes the discriminants of a `repr(u8)` enum, so this test has to catch a variant - // added anywhere but the end, or a representation that stops being `u8`, before a renderer - // draws one metric twice. assert_eq!(QualityMetric::ALL.first(), Some(&QualityMetric::Recall)); assert_eq!( QualityMetric::ALL.last(), @@ -1794,9 +1772,7 @@ fn every_metric_is_listed_once_under_the_noun_its_threshold_is_keyed_by() { labels.dedup(); assert_eq!(labels.len(), QualityMetric::ALL.len(), "{labels:?}"); - // The report checks each metric under a threshold key whose noun is the label, so an operator - // reading a rendered reading knows which key moves it. The keys are this test's own copy of - // that wire vocabulary: renaming one is a deliberate edit here as well as there. + // these literal threshold keys check the labels' relationship to the override vocabulary for (metric, key) in QualityMetric::ALL.into_iter().zip([ "minimum_recall", "minimum_trustworthiness", @@ -1828,9 +1804,7 @@ async fn a_delivery_stream_must_cover_every_request_exactly_once() { .expect("both requested rows were delivered"); assert_eq!(matched, ['a', 'z']); - // A repeat refuses instead of replacing the payload a reading would - // have used: nothing here can tell an echo of the same bytes from a - // second, different answer under one id. + // duplicate ids fail the stream even when a later delivery would complete coverage let repeated = match_deliveries( node_ids, &rows, diff --git a/libs/@local/graph/atlas/src/salt/relation/artifact.rs b/libs/@local/graph/atlas/src/salt/relation/artifact.rs index 147193722d7..adaf73e10bd 100644 --- a/libs/@local/graph/atlas/src/salt/relation/artifact.rs +++ b/libs/@local/graph/atlas/src/salt/relation/artifact.rs @@ -1,15 +1,14 @@ //! The relation indexes' published forms and their mapped readers. //! -//! A [`ProtectionIndex`] publishes as one [`crate::file::sprs`] file holding its -//! [`ProtectionMatrix`](super::protection::ProtectionMatrix) verbatim; the evidence pair travels as -//! an opaque 8-byte value. [`ProtectionArchive`] reopens the file over a whole-file mapping and -//! validates the index invariants once, so hard-negative mining reads the evidence from the page -//! cache without holding it on the heap. +//! [`ProtectionIndex`] writes a [`crate::file::sprs`] matrix with opaque 8-byte evidence values. +//! [`AttractionIndex`] writes [`crate::file::attraction`] group records and a flat edge array. Both +//! archive types validate local index invariants over mapped bytes without copying those regions +//! onto the heap. //! -//! An [`AttractionIndex`] publishes as one [`crate::file::attraction`] file: group records over a -//! flat edge array, the same factorization the resident index stores. [`AttractionArchive`] reopens -//! it the same way and validates the index invariants once, so relation-edge sampling reads groups -//! and edges from the page cache. +//! Mapped borrowing requires the backing bytes to remain immutable. The archives verify no shared +//! provenance between the files and reconstruct no evidence from links and policies. Sparse-view +//! creation and attraction-region access repeat lower-level validation, with the costs documented +//! on the accessors. #![cfg_attr( not(test), expect( @@ -51,15 +50,6 @@ where { type Error = WriteSprsError; - /// Writes the index as a sparse matrix file. - /// - /// Returns the SHA-256 of the written bytes, which is the identity the repository records for - /// the published file. - /// - /// # Errors - /// - /// Returns an error when the underlying writer fails, or the index spans zero rows, which the - /// format cannot represent: a generation without node rows publishes no artifacts. fn write_into(&self, write: impl io::Write) -> Result { let mut writer = Writer { accumulator: Sha256::new(), @@ -68,7 +58,7 @@ where write_matrix(&self.matrix(), &mut writer).map_err(|error| match error { error @ (WriteSprsError::Io(_) | WriteSprsError::ZeroDimension { .. }) => error, - // A validated index is row-compressed and unsliced. + // index validation accepts offset pointers, while the writer requires an initial zero. WriteSprsError::Sliced => { unreachable!("a validated index's pointers begin at zero") } @@ -78,8 +68,7 @@ where } } -// Only the corpus-domain index publishes; the distinct-domain twin feeds the trainer and -// never stages. +// publication is restricted to the corpus row domain. impl WriteAs for ProtectionIndex {} /// An opened sparse matrix file does not hold a valid protection index. @@ -119,10 +108,9 @@ impl Error for InvalidProtectionFile { /// A published protection index opened over its mapped file. /// -/// Construction checks the index invariants once, so an open index only serves valid views; the -/// matrix regions stay in the page cache under memory pressure and off the heap. Each -/// [`view`](Self::view) re-checks the compressed-row structure ([`SprsFile::matrix`]'s contract), -/// so stages call it once and hold the view. +/// Construction checks the index invariants once. Matrix regions borrow from the file mapping +/// rather than a heap copy. [`Self::view`] repeats [`SprsFile::matrix`]'s value-pattern and +/// sparse-structure checks. Reuse a view for repeated lookups. #[derive(Debug)] pub(crate) struct ProtectionArchive { file: SprsFile, @@ -137,8 +125,8 @@ where /// /// # Errors /// - /// Returns an error when the file does not hold the index's matrix layout or the matrix - /// violates a [`ProtectionIndex`] invariant. + /// Returns [`InvalidProtectionFile`] for an incompatible matrix layout or an index-invariant + /// violation. #[tracing::instrument(skip_all)] pub(crate) fn new(file: SprsFile) -> Result { let matrix = file.matrix().map_err(InvalidProtectionFile::Matrix)?; @@ -150,7 +138,12 @@ where }) } - /// Borrows the validated index. + /// Borrows the index after rechecking value patterns and sparse structure. + /// + /// # Complexity + /// + /// Takes `O(N + M)` time for `N` rows and `M` stored entries. It borrows the regions without + /// copying them. #[must_use] pub(crate) fn view(&self) -> ProtectionView<'_, N> { let matrix = self @@ -169,10 +162,9 @@ where { /// Writes the index as an attraction file. /// - /// The file persists the index's row domains in its header, so it reopens only under the - /// same types. `rows` is the row count of the endpoint domain the edges index into; the - /// index does not carry it, the caller's generation does. Returns the SHA-256 of the written - /// bytes: the identity the repository records for the published file. + /// The header records `N` and `E`'s row-kind tags and the writer's byte order. `rows` must + /// cover every endpoint, but writing does not check this bound. Returns the SHA-256 of the + /// written bytes. A [`io::BufWriter`] can combine the small per-record writes. /// /// # Errors /// @@ -226,9 +218,9 @@ pub(crate) enum InvalidAttractionIndex { BrokenEdgeRanges { group: usize }, /// An edge references a node row outside the corpus domain. RowOutOfDomain { edge: usize }, - /// An edge carries score-provenance bits this module does not speak. + /// An edge sets score-provenance bits outside [`Scored`]'s flags. UnknownScoredBits { edge: usize }, - /// The edges within one group break the ascending `(source, target, edge)` order. + /// The edges within one group break the strictly ascending `(source, target, edge)` order. UnorderedEdges { edge: usize }, } @@ -266,15 +258,14 @@ impl Error for InvalidAttractionIndex {} /// A published attraction index opened over its mapped file. /// -/// Construction checks every index invariant once, so an open index only serves valid groups -/// and consumers re-validate nothing. The invariants: +/// Construction checks strict relation and in-group edge order, nonempty ranges partitioning the +/// edge region, endpoint bounds and score-presence flags. [`AttractionFile`] checks each scalar +/// field's domain. These checks do not reject self-edges or verify that factors derive from a +/// policy and instance set. /// -/// - relations ascend strictly -/// - edge ranges partition the edge region into non-empty spans -/// - weights and scores stay in their domains -/// - edges ascend within their group -/// -/// The regions stay in the page cache under memory pressure and off the heap. +/// Regions borrow from the mapping rather than a heap copy. Access to either raw region repeats its +/// scalar validation through [`AttractionFile::groups`] or [`AttractionFile::edges`]. Even count +/// accessors can scan an entire region. #[derive(Debug)] pub(crate) struct AttractionArchive { file: AttractionFile, @@ -289,7 +280,7 @@ where /// /// # Errors /// - /// Returns an error when the file violates an attraction-index invariant. + /// Returns [`InvalidAttractionIndex`] for a range, ordering, endpoint or score-flag violation. #[tracing::instrument(skip_all)] pub(crate) fn new(file: AttractionFile) -> Result { let groups = file.groups(); @@ -325,7 +316,7 @@ where .map(GroupRecord::edge_offset) .peekable(); for (index, edge) in edges.iter().enumerate() { - // Group boundaries reset the in-group order comparison. + // compare edge order only within a group. if boundaries.next_if_eq(&(index as u64)).is_some() { previous = None; } @@ -356,14 +347,14 @@ where self.file.rows() } - /// Returns the relation group count. + /// Returns the relation group count, revalidating the group region in `O(G)` time. #[inline] #[must_use] pub(crate) fn group_count(&self) -> usize { self.file.groups().len() } - /// Returns the retained instance count over all groups. + /// Returns the retained instance count, revalidating the edge region in `O(E)` time. #[inline] #[must_use] pub(crate) fn edge_count(&self) -> usize { @@ -372,6 +363,11 @@ where /// Borrows one relation group. /// + /// # Complexity + /// + /// Takes `O(G + E)` time for `G` groups and `E` edges, revalidating both complete record + /// regions before borrowing the selected span. + /// /// # Panics /// /// This panics when `index` is not below [`group_count`](Self::group_count). @@ -380,8 +376,7 @@ where let groups = self.file.groups(); let record = &groups[index]; - // Construction validated the ranges against the edge region, so - // the narrowing repeats accepted in-bounds values. + // construction bounded these offsets by the resident edge region's length. let start = usize::try_from(record.edge_offset()) .expect("a validated edge offset fits the address space"); let end = groups.get(index + 1).map_or_else( @@ -429,13 +424,17 @@ where } } - /// Iterates the instances, ascending by `(source, target, edge)`. + /// Iterates the instances in strictly ascending `(source, target, edge)` order. pub(crate) fn edges(&self) -> impl ExactSizeIterator> + '_ { self.records.iter().map(decode) } } -/// Decodes one validated edge record into the resident edge type. +/// Decodes one edge record with validated score-presence flags. +/// +/// # Panics +/// +/// Panics when `record` sets a bit outside [`Scored`]'s flags. fn decode(record: &EdgeRecord) -> AttractionEdge where N: NodeRow, diff --git a/libs/@local/graph/atlas/src/salt/relation/attraction.rs b/libs/@local/graph/atlas/src/salt/relation/attraction.rs index 094c359efa1..32785d59f7d 100644 --- a/libs/@local/graph/atlas/src/salt/relation/attraction.rs +++ b/libs/@local/graph/atlas/src/salt/relation/attraction.rs @@ -1,11 +1,9 @@ -//! Force-bearing instances grouped by relation. +//! Retained link instances grouped by relation. //! -//! [`AttractionIndex`] stores every admitted instance that survives force pruning, contiguously per -//! relation type. A group carries the factors shared by its relation (class weights, frozen -//! strength); its edges carry the per-instance factors (effective confidence, degree -//! normalization). Training weights one edge by multiplying the group and edge factors into its -//! class energies, so every factor of the relation-attraction objective enters exactly once by -//! construction. +//! [`AttractionIndex`] groups retained non-self instances by relation type for per-type sampling. A +//! group supplies class weights and frozen strength. Each edge supplies effective confidence and +//! share-weighted degree normalization. The [relation weight model](super#weights) defines how +//! these factors combine. use super::EffectiveConfidence; use crate::{ @@ -13,11 +11,11 @@ use crate::{ math::{NonNegative, PositiveUnitFraction}, }; -/// Shared attraction settings of one generation, valid by construction. +/// Shared class scaling and attraction-pruning settings of one generation. /// -/// The Coincident coefficient `κ_C` scales the Coincident energy relative to Proximal's unit scale. -/// It stays 0 until the generation meets its Coincident release criterion; after that, tuning grids -/// ratios in `2..=8` (the composite-objective tuning protocol), so enabling runs start there. +/// The Coincident coefficient `κ_C` scales Coincident relative to Proximal's unit scale. It is zero +/// by default. A nonzero coefficient is accepted without checking any release criterion. The +/// calibration starting grid is `2..=8`, to be judged against the generation's quality evidence. /// /// The pruning threshold `η_F` drops instances whose force mass `c · s · s+` cannot move the /// layout, and 0 retains every instance. The omitted-mass fraction a threshold produces @@ -38,9 +36,8 @@ const impl Default for AttractionOptions { impl AttractionOptions { /// Creates settings from a Coincident coefficient and a pruning threshold. /// - /// Both values carry their domain in the type, so construction validates nothing. The - /// default is `κ_C = 0` (the Coincident class exerts no pull until the generation meets its - /// release criterion) and `η_F = 0` (every admitted instance survives). + /// Both values must be finite and non-negative. The defaults are `κ_C = 0` and `η_F = 0`, + /// disabling Coincident weighting and attraction pruning. #[must_use] pub(crate) const fn new( coincident_coefficient: NonNegative, @@ -67,10 +64,10 @@ impl AttractionOptions { } } -/// One force-bearing link instance under its group's relation. +/// One retained link instance under its group's relation. /// -/// The stored factors are the ones that vary per instance; the class weights and strength -/// multiplier live on the owning [`AttractionGroup`]. +/// These factors vary per instance. [`AttractionGroup`] supplies shared class weights and strength. +/// Retention alone does not imply positive force. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct AttractionEdge { /// The edge row that produced the instance. @@ -81,9 +78,9 @@ pub(crate) struct AttractionEdge { pub target: N, /// The instance's effective confidence `c` with score provenance. pub confidence: EffectiveConfidence, - /// The degree normalization `ν`. + /// The combined degree normalization and reading share, `ν · s`. /// - /// Computed over the complete admitted instance set of the group's relation. + /// Degrees cover the group's complete non-self instance set before pruning. pub normalization: PositiveUnitFraction, } @@ -105,8 +102,8 @@ pub(crate) struct AttractionWeights { impl AttractionWeights { /// Returns the positive force scale `s+`, the sum of the class weights. /// - /// An instance's force mass is its confidence times its reading share times this scale; the - /// pruning predicate compares that mass against the threshold. + /// The pruning mass is confidence times reading share times this scale. Strength remains + /// separate. Arbitrary weights can overflow the `f32` sum to infinity. #[inline] #[must_use] pub(crate) const fn scale(self) -> NonNegative { @@ -115,6 +112,8 @@ impl AttractionWeights { } /// One relation type's retained instances and shared weights. +/// +/// Edges are strictly ascending by `(source, target, edge)`. #[derive(Debug, Clone)] pub(crate) struct AttractionGroup { relation: OntologyRowId, @@ -123,9 +122,9 @@ pub(crate) struct AttractionGroup { } impl AttractionGroup { - /// Assembles a group. + /// Assembles one relation's retained instances and shared weights. /// - /// The builder upholds the documented edge order. + /// `edges` must be strictly ascending by `(source, target, edge)`. pub(super) const fn new( relation: OntologyRowId, weights: AttractionWeights, @@ -152,7 +151,7 @@ impl AttractionGroup { self.weights } - /// Borrows the retained instances, ascending by `(source, target, edge)`. + /// Borrows the retained instances in strictly ascending `(source, target, edge)` order. #[inline] #[must_use] pub(crate) const fn edges(&self) -> &[AttractionEdge] { @@ -160,28 +159,29 @@ impl AttractionGroup { } } -/// Force-bearing instances of one generation, grouped by relation type. +/// Retained link instances of one generation, grouped by relation type. /// -/// Groups ascend by relation row; a relation none of whose instances survived pruning stores no -/// group. Within a group, edges ascend by `(source, target, edge)`. Both orders are total, so the -/// index is identical for any input order of the same instances. +/// Groups ascend strictly by relation row, omitting empty groups. Within a group, edges ascend +/// strictly by `(source, target, edge)`. Under the [instance uniqueness +/// contract](super#input-contract), [`super::RelationIndexes::build`] produces the same index for +/// any input order at the same floating-point implementation. #[derive(Debug, Clone)] pub(crate) struct AttractionIndex { groups: Vec>, } impl AttractionIndex { - /// Assembles the index. + /// Assembles nonempty groups in strictly ascending relation order. /// - /// The builder upholds the documented group order. + /// Each group must satisfy [`AttractionGroup`]'s edge-order contract. pub(super) const fn new(groups: Vec>) -> Self { Self { groups } } - /// Returns the index carrying no force at all. + /// Returns an empty attraction index. /// - /// The trainer's vacuous run consumes it. A placement configured to withhold the relation - /// evidence trains against every other term while the published relation artifacts stay real. + /// Use this to disable relation attraction while retaining the other objective terms. It does + /// not alter protection evidence. #[must_use] pub(crate) const fn vacuous() -> Self { Self { groups: Vec::new() } diff --git a/libs/@local/graph/atlas/src/salt/relation/bench/fixture.rs b/libs/@local/graph/atlas/src/salt/relation/bench/fixture.rs index 53cd35c6cf9..0737c007d1d 100644 --- a/libs/@local/graph/atlas/src/salt/relation/bench/fixture.rs +++ b/libs/@local/graph/atlas/src/salt/relation/bench/fixture.rs @@ -1,4 +1,4 @@ -//! Synthetic relation corpora at the live store's measured shape. +//! Synthetic relation corpora for measuring volume concentration and hubbed endpoints. use core::num::NonZero; use std::sync::OnceLock; @@ -21,9 +21,10 @@ use crate::{ /// Cumulative specific-type link volumes measured in the live store. /// -/// Sixteen specific relation types over 2,196,563 links, spanning five orders of magnitude; the -/// largest owns 34% of links, the smallest 4 links. Sampling a uniform position below the total and -/// bucketing by these boundaries reproduces the measured volume distribution at any corpus scale. +/// The recorded histogram covers 2,196,563 links across sixteen specific types. The largest has +/// 739,374 links (about 34%), the smallest four. Uniform positions below the total select types +/// with probabilities proportional to these volumes. Finite synthesized counts fluctuate rather +/// than reproducing exact proportions. const MEASURED_SPECIFIC_CUMULATIVE: [u64; 16] = [ 739_374, 1_405_028, 1_861_990, 1_971_015, 2_041_671, 2_096_752, 2_143_950, 2_165_211, 2_185_797, 2_190_391, 2_193_025, 2_195_240, 2_196_479, 2_196_553, 2_196_559, 2_196_563, @@ -38,24 +39,27 @@ const RELATION_TYPES: usize = 1 + MEASURED_SPECIFIC_CUMULATIVE.len(); /// Odd multiplier scattering hub ranks over the power-of-two row domain. /// -/// Odd times anything is invertible modulo a power of two, so distinct ranks map to distinct rows. +/// An odd integer is invertible modulo any power of two. Multiplication by this constant followed +/// by the row mask permutes that domain. Therefore distinct in-domain ranks map to distinct rows. const HUB_SCATTER: u64 = 0x9E37_79B9_7F4A_7C15; /// How a synthesized corpus distributes volume over relation types. /// -/// All three profiles share the same endpoint generator, instance volume, and policy table; they -/// differ only in volume concentration, so a timing difference between them is attributable to skew -/// alone. +/// Profiles share an endpoint-generation law, instance count and policy table. Live type draws +/// consume additional randomness, changing the realized endpoints even at the same seed. Relation +/// assignments also change policy-weight mixtures and degrees. Timing differences do not isolate +/// relation skew alone. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum Profile { - /// The measured live shape. + /// A shared base type plus a specific type drawn from the recorded histogram. /// - /// Every link carries the shared base type plus one specific type drawn from the measured - /// histogram, so the base relation owns exactly half of all instances. + /// Exactly half of all instances have the base relation. Both readings use multiplicity one. Live, - /// The same instance volume spread evenly over the same type count. + /// Round-robin readings whose type counts differ by at most one. + /// + /// Each synthetic link has two distinct relation readings, both at multiplicity one. Uniform, - /// One relation owns every instance. + /// One relation owns every instance, as pairs of distinct single-reading edges. Mega, } @@ -73,10 +77,13 @@ impl Profile { /// A synthesized relation corpus with sorted stage inputs on demand. /// -/// Holds the raw instance set; the group-sorted and pair-sorted copies each build stage starts from -/// materialize on first use and stay cached, so a corpus that only runs full builds keeps one copy -/// resident (about 250 MB at the live scale of 2.2M links) and one that isolates stages keeps -/// three. +/// Synthesis retains the instance buffer. Group-sorted instances, emission-order protection records +/// and the assembled protection index initialize lazily and remain cached. Full builds additionally +/// require a mutable instance copy. Stage isolation can retain all of these buffers at once. +/// +/// `N` must represent the full row domain and its end fencepost, and `E` must represent the +/// generated edge ids. Stage assembly requires the row count to fit `u32`, a bound synthesis does +/// not check. pub struct Corpus { rows: usize, links: usize, @@ -90,17 +97,31 @@ pub struct Corpus { impl Corpus { /// Synthesizes a corpus of `links` links under `profile`. /// - /// The row domain is the largest power of two at most half the link count (the live ratio: 2.2M - /// links over 1M rows), floored at 64. Sources are uniform over the rows. Targets follow a - /// truncated Zipf tail over one eighth of the rows (the measured hub shape has 124K distinct - /// targets, the largest gathering 9% of all links). Confidence stays unscored throughout, the - /// live corpus's only shape. Every draw comes from `rng` seeded with `seed`, so equal arguments - /// synthesize equal corpora. + /// The row count is the largest power of two not exceeding `max(links / 2, 64)`. This + /// approximates the recorded 2.2M-link, 1M-row scale. Sources are uniform over rows. The + /// recorded hub profile has about 124K distinct targets, motivating `H = rows / 8`. Its largest + /// target holds 9% of links. + /// + /// For a uniform `U ∈ [0, 1)`, target rank is `floor(Hᵁ) − 1`, scattered through an odd modular + /// multiplier. In ideal arithmetic, rank `k` has probability `ln((k + 2)/(k + 1)) / ln(H)` for + /// `0 ≤ k ≤ H − 2`. This is a Zipf-like hub distribution and no exact replay of the measured + /// target counts. Floating-point power evaluation and discrete random draws approximate this + /// law. + /// + /// Each input link produces two instances with unscored confidence. Live and Uniform repeat the + /// edge id under distinct relations but leave multiplicity at one. They measure full-share + /// emission, not the production two-reading share of one half. Mega instead assigns distinct + /// edge ids under one relation. Self-references remain in synthesis output for the build to + /// drop. + /// + /// Equal arguments repeat at the same RNG and floating-point implementation. Power evaluation + /// does not promise cross-target bit identity. The `links` parameter counts endpoint draws, not + /// distinct edge ids in every profile. /// /// # Panics /// - /// This panics when the instance set does not fit the address space. Construction satisfies - /// every internal expectation. + /// Panics when generated ids exceed `N` or `E`'s range, or the instance allocation exceeds + /// capacity. `2 · links` must fit `usize`. #[expect( clippy::integer_division, clippy::integer_division_remainder_used, @@ -148,8 +169,7 @@ impl Corpus { let edge = link as u64; match profile { Profile::Live => { - // One link, two readings sharing the edge row: the - // base type and a histogram-drawn specific type. + // a base reading and a histogram-drawn specific reading share the edge id. let at = endpoints(&mut rng); let position = uniform_below(&mut rng, MEASURED_LINKS); let specific = 1 + MEASURED_SPECIFIC_CUMULATIVE @@ -160,16 +180,15 @@ impl Corpus { instances.push(instance(edge, specific, at)); } Profile::Uniform => { - // Round-robin over an odd type count: the pair is - // always distinct and every type's volume is even. + // consecutive positions modulo seventeen give distinct relations within each + // pair. Across all readings, type counts differ by at most one. let at = endpoints(&mut rng); instances.push(instance(edge, (link * 2) % RELATION_TYPES, at)); instances.push(instance(edge, (link * 2 + 1) % RELATION_TYPES, at)); } Profile::Mega => { - // A parallel pair of single-reading links between one endpoint pair gives the - // same instance, pair, and protection-entry volume as the other profiles while - // one relation owns everything. + // distinct edge ids keep each (edge, relation) reading unique while one + // relation holds both instances. let at = endpoints(&mut rng); instances.push(instance(edge * 2, 0, at)); instances.push(instance(edge * 2 + 1, 0, at)); @@ -179,9 +198,7 @@ impl Corpus { let policies = (0..RELATION_TYPES) .map(|relation| { - // Masses and applicabilities spread across the table so - // pruning-threshold and floor sweeps separate relations - // instead of dropping all or nothing. + // varying mass and applicability let threshold and floor sweeps separate relations. let step = f64::from(u8::try_from(relation).expect("the table holds 17 types")); let distribution = ClassProbabilities { coincident: UnitFraction::ZERO, @@ -218,7 +235,7 @@ impl Corpus { self.rows } - /// Returns the synthesized link count. + /// Returns the requested number of endpoint draws. #[inline] #[must_use] pub const fn links(&self) -> usize { @@ -252,6 +269,10 @@ impl Corpus { } /// Borrows the emitted protection records in emission order, emitting on first use. + /// + /// # Panics + /// + /// Panics when `N` cannot represent zero for scratch initialization. pub(super) fn records(&self) -> &[ProtectionRecord] where N: Id, @@ -280,6 +301,11 @@ impl Corpus { } /// Borrows the assembled protection index, assembling on first use. + /// + /// # Panics + /// + /// Panics when the row domain exceeds the matrix encoding or required row positions cannot be + /// represented by `N`. pub(super) fn protection(&self) -> &ProtectionIndex where N: Id, diff --git a/libs/@local/graph/atlas/src/salt/relation/bench/judge.rs b/libs/@local/graph/atlas/src/salt/relation/bench/judge.rs index e5b38511e5c..e8d56641a91 100644 --- a/libs/@local/graph/atlas/src/salt/relation/bench/judge.rs +++ b/libs/@local/graph/atlas/src/salt/relation/bench/judge.rs @@ -1,15 +1,12 @@ //! Judge-layout runners for pointwise probes and row-batched merges. //! -//! Hard-negative mining vets every mined candidate pair against the protection index in one of two -//! shapes. The first shape is a `judge` probe per pair (a row resolution plus a binary search). The -//! second is one `row` walk per query point, merged against that point's sorted candidate list. The -//! runners here execute both shapes over identical probe sets so numbers decide the miner's access -//! layout. The pointwise runner calls the production probe. The row-merge runner is the candidate -//! layout under audition, written here once so a decision for it promotes this merge into the -//! protection view. +//! Pointwise judgment uses one row resolution and binary search per pair. Row-merge judgment walks +//! each query row's partners against its sorted candidates. These runners compare both access +//! patterns over identical probe sets. The pointwise path uses the production lookup, while the +//! row-merge path implements the alternative for measurement. //! -//! The runners judge probes under the default protection configuration, whose zero thresholds -//! protect exactly the linked pairs, the conservative baseline every calibration starts from. +//! Both use the default protection configuration, whose zero thresholds protect every stored pair. +//! The comparison does not evaluate non-default threshold calibration. use core::num::NonZero; @@ -24,9 +21,9 @@ use crate::{ /// A full mining sweep's candidate pairs, one chunk per node row. /// -/// Row `i`'s candidates occupy the `i`-th fixed-width chunk, ascending within the chunk: the shape -/// a mined neighbour list takes after canonical ordering, and the order the row-merge runner -/// requires. +/// Row `i`'s candidates occupy the `i`-th fixed-width chunk in ascending order, as the row-merge +/// runner requires. Draws have replacement and may include duplicates and self-candidates. This +/// type carries no corpus identity. pub struct JudgeProbes { per_row: NonZero, candidates: Vec, @@ -46,14 +43,23 @@ impl Corpus { /// /// Every node row queries `per_row` candidates. Each candidate is one of the row's linked /// partners with probability `partner_fraction` (falling back to a uniform row when the row has - /// no partners) and a uniform row otherwise. The fraction dials the sweep's protected-hit rate: - /// attraction pulls linked pairs together in 2D, so a real mining sweep skews hit-rich, and the - /// layout question needs a hit-poor reading beside it. + /// no partners) and a uniform row otherwise. Draws have replacement. + /// + /// For a row with `d > 0` partners among `N` rows and fraction `f ∈ [0, 1]`, the ideal + /// protected-hit probability is: + /// + /// `f + (1 − f) · d/N`, since a uniform draw can also hit a partner. + /// + /// Rows with no partners have zero hits. Sweep the fraction to compare hit-poor and hit-rich + /// access, without assuming a real miner's hit rate. + /// + /// The fraction is not validated. Values at or below zero and NaN select only uniform draws. + /// Values at or above one always select a partner when one exists. /// /// # Panics /// - /// This panics when the probe set does not fit the address space. Construction satisfies every - /// internal expectation. + /// Panics when the cached protection index cannot be assembled, row ids cannot be represented, + /// or the probe allocation exceeds capacity. `rows · per_row` must fit `usize`. #[must_use] pub fn judge_probes( &self, @@ -108,8 +114,18 @@ impl Corpus { /// Judges every probe pair through pointwise probes. /// - /// One production `judge` call per pair. Returns the hard-protected count, which doubles as the - /// cross-layout agreement check. + /// Returns the hard-protected count, including repeated candidates. Use probes from this corpus + /// to compare with [`Self::judge_by_row`]. Probes carry no checked corpus association. + /// + /// # Complexity + /// + /// After cached index assembly, `P` probes take `O(P log(d + 2))` time for maximum stored row + /// length `d`. + /// + /// # Panics + /// + /// Panics if cached protection assembly fails or a probe row position cannot be represented by + /// `N`. #[must_use] pub fn judge_pointwise(&self, probes: &JudgeProbes) -> usize where @@ -138,9 +154,19 @@ impl Corpus { /// Judges every probe pair through one row merge per query row. /// - /// Walks each row's protected partners once, merged against the row's ascending candidate - /// chunk. Returns the hard-protected count; equal probes yield the pointwise runner's count - /// exactly. + /// Returns the hard-protected count, including repeated candidates. With probes from this + /// corpus, the count equals [`Self::judge_pointwise`]'s. Candidate duplicates leave the + /// matching partner available for each repeated count. + /// + /// # Complexity + /// + /// After cached index assembly, takes `O(N + P + M)` time for `N` queried rows, `P` probes and + /// `M` stored entries in those rows. Each row's partners are traversed at most once. + /// + /// # Panics + /// + /// Panics if cached protection assembly fails, `N` cannot represent a required row, or the + /// probes contain more row chunks than this corpus has rows. #[must_use] pub fn judge_by_row(&self, probes: &JudgeProbes) -> usize where diff --git a/libs/@local/graph/atlas/src/salt/relation/bench/mod.rs b/libs/@local/graph/atlas/src/salt/relation/bench/mod.rs index fae49eb3b73..98cfa93c4f3 100644 --- a/libs/@local/graph/atlas/src/salt/relation/bench/mod.rs +++ b/libs/@local/graph/atlas/src/salt/relation/bench/mod.rs @@ -1,21 +1,35 @@ -//! Benchmark seams over the relation-index build. +//! Synthetic corpora and stage runners for measuring relation-index construction. //! -//! Wall-time claims about the build - the mega relation no longer serializes emission, assembly is -//! sort-dominated, the emission chunk is a batch size rather than a tuned number - are claims about -//! parallel composition, and only hold or fail at realistic scale and skew. This module gives the -//! bench target (an external crate) exactly the levers those claims need while every internal type -//! stays private: corpus synthesis at measured live shapes ([`Corpus`], [`Profile`]), each -//! production stage runnable from its own input state, and plain-number summaries -//! ([`BuildSummary`]). +//! Use [`production_corpus`] to construct inputs with the production row types. [`Profile`] varies +//! relation concentration, with the fixture's statistical and multiplicity limitations documented +//! on [`Corpus::synthesize`]. The runners expose the production stages of +//! [`RelationIndexes::build`] to measure group-emission scaling, assembly cost and the effect of +//! the emission-chunk size. //! -//! The stage runners call the production functions that [`RelationIndexes::build`] composes, never -//! mirrors of them, so the benchmarks measure a change to the build rather than diverging from it -//! unnoticed. Stages that reorder their input take a [`Scratch`] buffer the caller clones outside -//! the timed region. Stages that only read consume the corpus's pre-sorted copies directly. +//! Clone [`Scratch`] and [`Records`] outside the timed region. A corpus lazily caches grouped +//! instances, emitted records and a protection index. Warm the relevant accessor or stage before +//! timing if those initializations should be excluded. Stage runners still include the allocations +//! performed by the stage itself. //! -//! Beside the build seams, the judge runners (`judge`) compare the two access layouts hard-negative -//! mining could vet candidates through: pointwise pair probes against per-row partner merges, over -//! one synthesized mining sweep ([`JudgeProbes`]). +//! [`JudgeProbes`] supplies one candidate sweep to compare pointwise lookups with a row-merge +//! implementation. The comparison evaluates an alternative access pattern under the default +//! protection settings, not a complete mining algorithm. +//! +//! # Example +//! +//! With the `bench` feature enabled, a small corpus can exercise the build without timing it: +//! +//! ```rust +//! use hash_graph_atlas::bench::relation::{Profile, production_corpus}; +//! +//! let corpus = production_corpus(Profile::Live, 128, 42); +//! assert_eq!(corpus.instance_count(), 256); +//! let mut scratch = corpus.scratch(); +//! let proper = scratch.sort_by_group(); +//! let summary = corpus.build_in(&mut scratch, 0.0, 0.0); +//! assert_eq!(summary.retained_edges, proper); +//! assert_eq!(summary.pruned_edges, 0); +//! ``` use hashql_core::id::Id; use rand_xoshiro::Xoshiro256PlusPlus; @@ -40,17 +54,20 @@ mod judge; #[cfg(test)] mod tests; -/// The production emission chunk size, for sweeping around it. +/// Returns the production emission chunk size for comparison with nearby sizes. #[must_use] pub const fn production_chunk() -> usize { EMISSION_CHUNK } -/// Synthesizes a corpus at the production row-id instantiation, under the fixture generator. +/// Synthesizes a corpus with the production node and edge row types. /// -/// The benchmark targets measure the build over exactly the id types production indexes, and this -/// constructor is what fixes them, so a target never names an id type itself. Equal arguments -/// synthesize equal corpora ([`Corpus::synthesize`]). +/// Uses [`Xoshiro256PlusPlus`] with [`Corpus::synthesize`]'s fixture model and replay limits. See +/// the [module example](self). +/// +/// # Panics +/// +/// Panics when synthesis exceeds representable instance counts or allocation capacity. #[must_use] pub fn production_corpus( profile: Profile, @@ -67,7 +84,7 @@ pub struct BuildSummary { pub retained_edges: usize, /// Attraction edges dropped by the pruning predicate. pub pruned_edges: usize, - /// The fraction of total force mass the pruning dropped, read out at the hook boundary. + /// The fraction of `c · s · s+` mass dropped by pruning. pub omitted_mass_fraction: f64, /// Stored protection entries. /// @@ -77,8 +94,8 @@ pub struct BuildSummary { /// An owned instance buffer for stages that reorder their input. /// -/// Cloning one costs a large memcpy at bench scales; do it in the benchmark harness's setup phase, -/// outside the timed region. +/// Cloning copies the full instance buffer. Do it in the benchmark harness's setup phase, outside +/// the timed region. #[derive(Clone)] pub struct Scratch(Vec>); @@ -86,7 +103,7 @@ impl Scratch { /// Runs the group sort alone, returning the proper instance count. /// /// The buffer should hold instances in synthesis order ([`Corpus::scratch`]): the sort has a - /// sortedness fast path, so only an unsorted buffer measures the production pass. + /// sortedness fast path. Reusing the sorted result would measure a different input state. pub fn sort_by_group(&mut self) -> usize where N: Id, @@ -98,8 +115,7 @@ impl Scratch { /// An owned protection-record buffer for the assembly stage, in emission order. /// -/// The assembly reorders its input, so each timed run takes a fresh clone; clone in the benchmark -/// harness's setup phase, outside the timed region. +/// Assembly reorders the records. Clone a fresh buffer outside the timed region for each run. #[derive(Clone)] pub struct Records(Vec>); @@ -115,6 +131,12 @@ where } /// Clones the emitted protection records in emission order, the assembly's input state. + /// + /// The first call initializes the grouped-instance and emitted-record caches. + /// + /// # Panics + /// + /// Panics when `N` cannot represent zero. #[must_use] pub fn records_scratch(&self) -> Records { Records(self.records().to_vec()) @@ -122,10 +144,14 @@ where /// Runs the full production build over `scratch`. /// + /// `scratch` must come from this corpus. Supplying another corpus's instances does not validate + /// their provenance. + /// /// # Panics /// - /// This panics when the settings are not finite and non-negative, or the build rejects the - /// corpus, which the synthesis contract excludes. + /// Panics for non-finite or negative settings, a row domain exceeding `u32`, uncovered + /// relations, or endpoints outside this corpus's row domain. `N` must represent every row, + /// including zero. #[must_use] pub fn build_in( &self, @@ -133,8 +159,8 @@ where coincident: f32, pruning: f32, ) -> BuildSummary { - // The hook synthesizes its typed inputs from the sweep's plain settings: a bench target - // is its own crate and cannot name the crate-internal scalar types. + // the external benchmark target supplies plain settings because the scalar types are + // crate-private. let attraction = AttractionOptions::new( NonNegative::new(coincident).expect("the sweep passes a finite non-negative setting"), NonNegative::new(pruning).expect("the sweep passes a finite non-negative setting"), @@ -153,13 +179,13 @@ where /// Runs the group emission alone with `chunk` as the emission chunk size. /// - /// Reads the corpus's group-sorted instances and allocates the protection record buffer it - /// fills, exactly as the production build does. + /// Includes group-range resolution and allocation of the protection records. The first call + /// also initializes the cached group-sorted instances. /// /// # Panics /// - /// This panics when the corpus references an uncovered relation, which the synthesis contract - /// excludes. + /// Panics when `N` cannot represent zero, or when `chunk` is zero and the corpus has a non-self + /// group. pub fn emit_groups(&self, chunk: usize) { let ranges = build::resolve_groups(self.grouped(), self.policies()) .expect("the synthesized corpus covers every relation"); @@ -175,21 +201,27 @@ where /// Runs the protection assembly alone. /// - /// The assembly covers the record sort, the aggregation, and the scatter. + /// Includes record sorting, aggregation, scatter and index validation. `records` must come from + /// this corpus. + /// + /// # Panics + /// + /// Panics when record endpoints are outside this corpus's row domain, required row positions + /// are unrepresentable, or the assembled matrix fails validation. pub fn assemble_protection(&self, records: &mut Records) { drop(build::assemble_protection(self.rows(), &mut records.0)); } /// Runs the protection index's validation alone, over the corpus's assembled index. /// - /// Assembly constructs every invariant the validation re-checks; timing the check against - /// [`assemble_protection`](Self::assemble_protection) attributes the assembly stage's cost - /// between the scatter and the re-validation. + /// The first call also assembles and caches the index. Warm it before timing validation + /// separately. Comparing with [`Self::assemble_protection`] estimates the validation share of + /// that stage, with cache and measurement effects. /// /// # Panics /// - /// This panics when the assembled matrix fails its own validation, which the scatter contract - /// excludes. + /// Panics if the initial assembly fails for an unrepresentable domain or the matrix violates an + /// index invariant. pub fn validate_protection(&self) { let matrix = self.protection().matrix(); super::protection::validate(matrix).expect("the assembled matrix is valid"); diff --git a/libs/@local/graph/atlas/src/salt/relation/bench/tests.rs b/libs/@local/graph/atlas/src/salt/relation/bench/tests.rs index edbd634f2e0..e78d981ed8b 100644 --- a/libs/@local/graph/atlas/src/salt/relation/bench/tests.rs +++ b/libs/@local/graph/atlas/src/salt/relation/bench/tests.rs @@ -10,9 +10,12 @@ use rand_xoshiro::Xoshiro256PlusPlus; use super::{Corpus, Profile}; use crate::identity::{EdgeRowId, NodeRowId}; +/// Link count of the synthesised bench corpora. const LINKS: usize = 4_096; +/// Seed every synthesised corpus and probe draw in this file uses. const SEED: u64 = 42; +/// Synthesizes a corpus from the fixed [`LINKS`] and [`SEED`] settings. fn corpus(profile: Profile) -> Corpus { Corpus::synthesize::(profile, LINKS, SEED) } @@ -65,7 +68,7 @@ fn mega_concentrates_and_uniform_spreads() { } let smallest = volumes.iter().min().expect("the table is non-empty"); let largest = volumes.iter().max().expect("the table is non-empty"); - // Round-robin assignment leaves at most one extra pair per type. + // round-robin assignment leaves at most one extra reading per type. assert!(largest - smallest <= 2, "{volumes:?}"); } @@ -82,10 +85,10 @@ fn targets_are_hubbed_and_sources_are_not() { } let top = target_volume.values().max().expect("links exist"); - // The Zipf head gathers a few percent of all instances; a uniform - // target draw over 2048 rows would put ~4 instances on each. + // a uniform target draw over 2,048 rows would give 8,192 / 2,048 = 4 instances per row in + // expectation. The hub model concentrates more mass near its lowest ranks. assert!(*top > live.instance_count() / 50, "top hub owns {top}"); - // Sources are uniform over the domain: most rows appear. + // uniform source draws cover more than half the rows in this seeded fixture. assert!(sources.len() > live.rows() / 2, "{} sources", sources.len()); } @@ -98,8 +101,6 @@ fn full_build_matches_composed_stages() { assert_eq!(summary.pruned_edges, 0); assert_eq!(summary.omitted_mass_fraction.to_bits(), 0.0_f64.to_bits()); - // The isolated stages run over the same corpus without panicking - // and the sorts agree with the full build's proper split. let mut sorting = live.scratch(); let proper = sorting.sort_by_group(); assert_eq!(proper, live.grouped().len()); @@ -129,7 +130,8 @@ fn pruning_sweep_is_monotone() { previous_omitted = summary.omitted_mass_fraction; } - // Above every policy mass the sweep prunes everything. + // confidence and shares are one, and every policy scale is at most one. Threshold 1.5 prunes + // all non-self instances. let mut scratch = live.scratch(); let ceiling = live.build_in(&mut scratch, 0.0, 1.5); assert_eq!(ceiling.retained_edges, 0); @@ -150,8 +152,7 @@ fn judge_layouts_agree() { let live = corpus(Profile::Live); let per_row = core::num::NonZero::new(24).expect("the candidate width is positive"); - // Hit-poor and hit-rich sweeps: both layouts must count the same - // protected pairs at every hit rate. + // include both hit-poor and hit-rich probe sets in the access-layout comparison. for fraction in [0.0, 0.25, 0.9] { let probes = live.judge_probes::(per_row, fraction, SEED); assert_eq!(probes.pairs(), live.rows() * per_row.get()); @@ -167,8 +168,7 @@ fn judge_hit_rate_follows_partner_fraction() { let live = corpus(Profile::Live); let per_row = core::num::NonZero::new(24).expect("the candidate width is positive"); - // Zero thresholds protect every linked pair, so drawing candidates from the partner lists must - // raise the protected count. + // partner draws increase the expected hit rate under zero thresholds. let uniform = live.judge_probes::(per_row, 0.0, SEED); let linked = live.judge_probes::(per_row, 0.9, SEED); assert!(live.judge_pointwise(&linked) > live.judge_pointwise(&uniform) * 4); diff --git a/libs/@local/graph/atlas/src/salt/relation/build.rs b/libs/@local/graph/atlas/src/salt/relation/build.rs index ff8a105dd8c..cca84b56260 100644 --- a/libs/@local/graph/atlas/src/salt/relation/build.rs +++ b/libs/@local/graph/atlas/src/salt/relation/build.rs @@ -1,4 +1,4 @@ -//! Construction of both relation indexes from one admitted instance set. +//! Joint attraction and protection construction with shared instance ordering. use core::ops::Range; @@ -20,21 +20,27 @@ use crate::math::{DNonNegative, NonNegative, PositiveUnitFraction, narrow_f32}; /// Instances per parallel emission chunk within one relation group. /// -/// Relation volume is heavily skewed - a handful of types own most links - so the group pass cannot -/// lean on group-level parallelism alone: one dominant relation would serialize it. Within a group, -/// instances therefore emit over chunks of this size. Boundaries are fixed positions of the sorted -/// slice and the chunk partials combine in chunk order, so the double-precision mass sums associate -/// identically on every run: the build stays a function of the instance set, whatever the thread -/// count or scheduling. The size is a granularity, not a tuned number - large enough that per-chunk -/// task and buffer overhead vanishes behind tens of thousands of column searches, small enough that -/// a million-instance relation splits into dozens of stealable pieces; any nearby power of two -/// serves equally. +/// Chunking exposes parallel work within a dominant relation instead of relying on group-level +/// parallelism alone. With a unique sorted instance order, fixed chunk boundaries and ordered +/// combination of partials fix the association of the double-precision mass sums independently of +/// thread scheduling. Changing the chunk size can change the final bits. +/// +/// The 16,384-instance chunk size is an unvalidated choice. Sweep nearby powers of two on +/// representative skewed corpora to compare task and buffer overhead against available parallelism. pub(super) const EMISSION_CHUNK: usize = 1 << 14; /// Builds the attraction and protection indexes together. /// -/// See [`RelationIndexes::build`] for the contract; this is its implementation, composed from the -/// named stages below so each stage stays measurable on its own. +/// Reorders instances and derives both indexes under [`RelationIndexes::build`]'s input contract. +/// +/// # Errors +/// +/// Returns [`RelationIndexError`] for an oversized row domain or an uncovered non-self relation, in +/// that order. +/// +/// # Panics +/// +/// Panics for out-of-domain non-self endpoints or row positions that `N` cannot represent. pub(super) fn build( rows: usize, policies: Policies<'_>, @@ -45,7 +51,7 @@ where N: Id, E: Id, { - // The check precedes every allocation sized by `rows`. + // check before every allocation sized by `rows`. if u32::try_from(rows).is_err() { return Err(RelationIndexError::TooManyRows { rows }); } @@ -87,9 +93,8 @@ where retained_mass, pruned_mass, self_references, - // Starts empty: the histogram counts readings per edge, which only the fit's relation stage - // sees while draining the edge stream. That stage writes the counts here after the build - // returns. + // the histogram counts source edges, including readings outside this build's non-self + // partition. The edge drain supplies it separately. multi_typed_edges: Vec::new(), }; @@ -104,9 +109,10 @@ where /// Sorts instances by relation group and returns the proper count. /// -/// Self-references sort behind every proper instance, so the returned partition point drops them -/// without moving memory. The remainder of the key is total under the edge stream's uniqueness -/// contract, making the unstable parallel sort deterministic. +/// Self-references follow every proper instance. The returned boundary selects the non-self prefix +/// without a second compaction pass. Unique `(edge, relation)` readings make the complete key +/// distinct, fixing the sorted order. Duplicate keys with different scores or multiplicities do not +/// have that guarantee. pub(super) fn sort_by_group(instances: &mut [RelationInstance]) -> usize where N: Id, @@ -127,12 +133,12 @@ where /// Resolves every group's range and policy over group-sorted instances. /// -/// The resolution precedes any parallel work: the emission pass is infallible, and the first -/// uncovered relation in ascending order is the deterministic error. +/// `proper` must be sorted by relation with self-references removed. Resolution precedes group +/// emission, returning the first uncovered relation in ascending order. /// /// # Errors /// -/// Returns an error when an instance references a relation the policy table does not cover. +/// Returns [`RelationIndexError::MissingPolicy`] for an uncovered relation. pub(super) fn resolve_groups<'policy, N, E>( proper: &[RelationInstance], policies: Policies<'policy>, @@ -152,11 +158,10 @@ pub(super) fn resolve_groups<'policy, N, E>( Ok(group_ranges) } -/// One proper instance's protection contribution: its canonical pair and class evidence. +/// A canonical endpoint pair and the evidence from one non-self instance. /// -/// The group emission writes one record per instance - pruning-exempt, since protection evidence -/// covers the complete admitted set - and the protection assembly orders and aggregates the records -/// without revisiting instances or policies. +/// Group emission writes one record per non-self instance regardless of attraction pruning. +/// Assembly can then aggregate pair evidence without revisiting policies. #[derive(Debug, Copy, Clone)] pub(super) struct ProtectionRecord { pair: NodePair, @@ -165,7 +170,11 @@ pub(super) struct ProtectionRecord { } impl ProtectionRecord { - /// The zero record the build's scratch starts from; emission overwrites every slot. + /// Returns a zero scratch record for emission to overwrite. + /// + /// # Panics + /// + /// Panics when `N` cannot represent zero. pub(crate) const fn empty() -> Self where N: [const] Id, @@ -190,8 +199,15 @@ pub(super) struct GroupMeasurements { /// /// Each group also emits its instances' protection records into its slice of `records`, /// positionally: the record at a group-relative offset describes the instance at that offset. -/// `chunk` is the emission batch size. The index build passes [`EMISSION_CHUNK`], and benchmarking -/// other values through this parameter verifies the batch-size claim on that constant. +/// `group_ranges` must partition `proper` contiguously from zero into nonempty relation runs in +/// correct order, each paired with its policy. `records` must have the same length as `proper`. +/// `chunk` is the positive emission batch size. Use [`EMISSION_CHUNK`] to retain the production +/// summation order. +/// +/// # Panics +/// +/// Panics for an out-of-bounds or empty group range, insufficient record storage, or zero `chunk` +/// when a group emits. pub(super) fn build_groups( proper: &[RelationInstance], group_ranges: Vec<(Range, &RelationPolicy)>, @@ -203,8 +219,7 @@ where N: Id, E: Id, { - // The resolved ranges are contiguous and ascending from zero, so the - // record buffer carves into the groups' disjoint slices by length. + // contiguous ranges partition the records into disjoint group slices. let mut slices = Vec::with_capacity(group_ranges.len()); let mut rest = records; for (range, _) in &group_ranges { @@ -224,20 +239,21 @@ where /// Assembles the symmetric evidence matrix from the emitted protection records. /// -/// The records order by canonical pair first: one pair's records may have emitted at any positions, -/// and the aggregation's per-component maximum is order-independent, so the assembled index is a -/// function of the instance set. Two passes over the pair runs then build the matrix: counting -/// fills the row pointers, the scatter writes each pair's aggregated evidence into both of its -/// rows. Canonical pair order makes the scatter emit every row's partners ascending without a sort: -/// a row's smaller partners arrive while the row is some pair's second endpoint (ascending by the -/// pairs' first components), its larger partners afterwards while it is the first (ascending by -/// second components). The scatter is sequential, the assembly's serial floor; the index validation -/// behind it re-checks the constructed invariants in parallel. +/// Records must have distinct endpoints in the `rows` domain, finite non-negative components and +/// `discounted ≤ undiscounted`. `rows` must fit the column encoding, and `N` must represent the row +/// positions and end fencepost. +/// +/// Sorting groups equal pairs for component-wise maximum aggregation. Counting each pair into both +/// rows determines the row pointers, then a sequential scatter copies the aggregate into both +/// directions. For a fixed row, lexicographic pair order lists smaller partners first, ordered by +/// the pair's first component, followed by larger partners ordered by the second. Therefore the +/// scatter emits strictly ascending partners without a per-row sort. Validation rechecks the +/// resulting matrix in parallel. /// /// # Panics /// -/// This panics when a record endpoint lies outside the `rows` domain, which the dataset row -/// contract excludes. +/// Panics for out-of-domain endpoints, unrepresentable row positions or entry counts, or a matrix +/// that fails [`ProtectionIndex::new`]'s validation. `rows + 1` must fit `usize`. pub(super) fn assemble_protection( rows: usize, records: &mut [ProtectionRecord], @@ -268,9 +284,7 @@ where let pair = run[0].pair; let value = pair_evidence(run); for (row, partner) in [(pair.lhs(), pair.rhs()), (pair.rhs(), pair.lhs())] { - // `columns` and `evidence` are the CSR entry arrays: their - // positions are storage offsets in the matrix encoding, not - // ids of any domain, so they stay raw. + // positions in the CSR entry arrays are storage offsets, not node ids. let slot = usize::try_from(cursor[row]).expect("resident entries fit the address space"); #[expect( @@ -320,11 +334,14 @@ struct GroupFactors { /// Builds one relation's attraction group from its contiguous instances. /// -/// The slice is one relation's run of the `(source, target, edge)` sort. Degrees count over two -/// compact endpoint columns in one scratch allocation: the source column inherits the run's order, -/// the target column sorts here, and a row's degree is the share sum over its run in each column, -/// read as a prefix difference. Emission then parallelizes over `chunk`-sized chunks reading those -/// shared columns. +/// `instances` must be a nonempty `(source, target, edge)`-sorted run for `policy.relation`, with +/// equally long `records`. Source and target columns each sort by `(row, share)` and build separate +/// share prefixes. Their prefix differences give each endpoint's share-weighted degree over the +/// complete run. Emission reads those columns in parallel. +/// +/// # Panics +/// +/// Panics when `instances` is empty or `chunk` is zero. fn build_group( instances: &[RelationInstance], policy: &RelationPolicy, @@ -338,8 +355,7 @@ where { let relation = instances[0].relation; let weights = AttractionWeights { - // A fraction of a widened f32 stays at or below that f32, so the narrowing cannot - // overflow. + // A fraction of a widened finite f32 is at most that f32. Narrowing cannot overflow. coincident: (policy.attraction.coincident * attraction.coincident_coefficient().widen()) .narrow_lossy(), proximal: NonNegative::new( @@ -374,8 +390,7 @@ where .map(|(chunk, records)| emit_chunk(chunk, records, &sources, &targets, factors, attraction)) .collect(); - // Combined in chunk order; see EMISSION_CHUNK for why that keeps - // the sums deterministic. + // combine in chunk order to preserve the association fixed by EMISSION_CHUNK. let (edges, measurements) = if yielded.len() == 1 { yielded.pop().expect("one chunk was just checked") } else { @@ -397,7 +412,8 @@ where /// One group column of endpoint rows. /// /// The rows ascend, with the running share total ahead of every position. A row's degree is the -/// share sum over its run, read as a prefix difference. Lookups stay binary searches. +/// share sum over its run, read as a prefix difference. Floating-point prefix subtraction can lose +/// small shares after a large prefix. Lookups use binary searches. struct DegreeColumn { rows: Vec, prefix: Vec, @@ -406,8 +422,8 @@ struct DegreeColumn { impl DegreeColumn { /// Sorts the entries and accumulates the share prefix. /// - /// The sort key includes the share bits, so equal rows order their shares deterministically and - /// the prefix sums are reproducible. + /// Entries must carry finite positive shares. Total ordering by `(row, share)` fixes their + /// summation order, including when equal rows have different shares. fn new(mut entries: Vec<(u64, f64)>) -> Self { entries.par_sort_unstable_by(|left, right| { left.0.cmp(&right.0).then(left.1.total_cmp(&right.1)) @@ -437,8 +453,15 @@ impl DegreeColumn { /// Emits one fixed chunk of a group's instances against its columns. /// -/// Every instance writes its protection record - pruning-exempt - and the instances the pruning -/// predicate retains emit attraction edges. +/// `records` must have the same length as `chunk`, and both degree columns must describe the +/// complete relation run. `factors` must contain the policy's in-domain scale, selected class sum +/// and applicability. Every instance writes a protection record regardless of pruning. Instances at +/// or above the mass threshold emit attraction edges. +/// +/// # Panics +/// +/// Panics if invalid factors produce non-finite or negative masses, or the computed normalization +/// leaves `(0, 1]`. fn emit_chunk( chunk: &[RelationInstance], records: &mut [ProtectionRecord], @@ -481,8 +504,7 @@ where } let normalization = { - // A row's degree spans both columns: it may source some - // edges and receive others. + // a row can source some instances and receive others. let degree = |row: u64| sources.degree(row) + targets.degree(row); let source = degree(instance.source.as_u64()); let target = degree(instance.target.as_u64()); diff --git a/libs/@local/graph/atlas/src/salt/relation/confidence.rs b/libs/@local/graph/atlas/src/salt/relation/confidence.rs index 3ef9d0a279b..32907a53a96 100644 --- a/libs/@local/graph/atlas/src/salt/relation/confidence.rs +++ b/libs/@local/graph/atlas/src/salt/relation/confidence.rs @@ -1,9 +1,8 @@ //! Link-confidence algebra: scores, provenance bits, and the effective confidence. //! -//! The dataset stream attaches up to three scores to one link instance ([`RelationConfidence`]); -//! [`RelationConfidence::effective`] combines them into the per-instance factor `c = c_link · -//! √(c_source · c_target)` with missing scores contributing the neutral factor 1, and [`Scored`] -//! retains which scores were present, down to its artifact wire encoding. +//! [`RelationConfidence::effective`] combines a link's scores into `c = c_link · √(c_source · +//! c_target)`, with a missing score contributing the neutral factor 1. [`Scored`] preserves which +//! scores were present when the original components are no longer retained. use crate::math::UnitFraction; @@ -12,7 +11,7 @@ use crate::math::UnitFraction; pub(crate) struct Scored(u8); impl Scored { - /// Every speakable presence bit. + /// All defined score-presence bits. const ALL: Self = Self::LINK | Self::SOURCE | Self::TARGET; /// No score present. pub(crate) const EMPTY: Self = Self(0); @@ -47,7 +46,6 @@ impl Scored { const impl core::ops::BitOr for Scored { type Output = Self; - /// Unions the presence bits. #[inline] fn bitor(self, rhs: Self) -> Self { Self(self.0 | rhs.0) @@ -72,8 +70,9 @@ pub(crate) struct RelationConfidence { impl RelationConfidence { /// Combines the three scores into one effective confidence. /// - /// The value is `link · √(source · target)` with missing scores contributing the neutral factor - /// 1. The provenance bits record which scores were present. + /// The value is `link · √(source · target)`, substituting `1` for each missing score. The + /// provenance bits record which scores were present. Operations round in `f64`, and the + /// source-target product can underflow to zero before its square root. #[must_use] pub(crate) fn effective(self) -> EffectiveConfidence { let scored = self.link.map_or(Scored::EMPTY, |_| Scored::LINK) @@ -99,9 +98,10 @@ pub(crate) struct EffectiveConfidence { } impl EffectiveConfidence { - /// Reassembles a confidence from its combined value and provenance bits. + /// Assembles a combined value and its score-presence bits. /// - /// The domain rides in the fraction, so construction validates nothing. + /// The fields constrain their individual domains, not whether any source scores produce this + /// combination. #[inline] #[must_use] pub(crate) const fn new(value: UnitFraction, scored: Scored) -> Self { @@ -130,7 +130,8 @@ mod tests { #[test] fn effective_confidence_combines_scores_exactly() { - // 0.5 · √(0.25 · 0.25): every factor is a power of two, so the product 0.125 is exact. + // 0.5 · √(0.25 · 0.25) = 0.125: these products and the square root are exactly + // representable. let confidence = RelationConfidence { link: Some(unit_fraction!(0.5)), source: Some(unit_fraction!(0.25)), diff --git a/libs/@local/graph/atlas/src/salt/relation/mod.rs b/libs/@local/graph/atlas/src/salt/relation/mod.rs index 56483d2bb4c..7e038ec7356 100644 --- a/libs/@local/graph/atlas/src/salt/relation/mod.rs +++ b/libs/@local/graph/atlas/src/salt/relation/mod.rs @@ -1,28 +1,28 @@ //! Relation indexes: factorized attraction edges and no-repel protection. //! -//! The deliverable is [`RelationIndexes`]: the two link-derived structures projector training -//! consumes, built together from one pass over the generation's admitted link instances so both -//! always describe the same edge set. +//! [`RelationIndexes::build`] derives attraction and protection from the same admitted link +//! instances. Attraction assigns geometric weights to typed instances. Protection retains pair +//! evidence independently of attraction pruning, for deciding which negative pairs to exclude. //! -//! - [`attraction::AttractionIndex`] holds every force-bearing instance, grouped by relation type, -//! carrying the weight factors of the relation-attraction objective. Factors live where they -//! vary: the per-relation values (class weights, frozen strength) on the group, the per-instance -//! values (effective confidence, degree normalization) on the edge. Each factor therefore enters -//! the objective exactly once, and the grouped layout is the shape minibatch sampling limits -//! per-relation representation over. -//! - [`protection::ProtectionIndex`] holds the per-pair evidence masses that veto targeted -//! repulsion between linked endpoint rows. Masses aggregate before attraction admission and -//! before force pruning, so an edge too weak to pull still vetoes a false-neighbour repulsion. +//! - [`attraction::AttractionIndex`] groups retained instances by relation type. Class weights and +//! frozen strength belong to the group, effective confidence and share-weighted degree +//! normalization to the edge. Multiplying these factors into the class energies applies each +//! exactly once. The grouped layout supports per-relation sampling caps. +//! - [`protection::ProtectionIndex`] aggregates evidence over each endpoint pair. A pruned +//! attraction instance still contributes protection evidence. Whether that evidence vetoes +//! repulsion depends on the channel's floor and threshold. //! //! # Input contract //! -//! Instances are the caller's admission decision over the dataset's edge stream: one -//! [`RelationInstance`] per admitted `(edge, relation)` reading, sharing the edge row's endpoints -//! and confidence scores. Row references and score ranges are the dataset stream's contracts -//! (`crate::dataset`), and each edge row appears at most once per relation because the stream -//! assigns edge rows by position; the build consumes them under those contracts. Every invariant -//! the build itself requires is carried by a validating type: [`Policies`] certifies the policy -//! table once at construction, and the option types are valid by construction. +//! Supply one [`RelationInstance`] per admitted `(edge, relation)` reading, with endpoints in the +//! node-row domain. Readings of one edge must share endpoints and scores, and each must carry the +//! edge's total admitted reading count. Each `(edge, relation)` must occur at most once. The build +//! assumes these relationships rather than validating them. It treats a zero multiplicity as one +//! and drops self-references before resolving policies. +//! +//! [`Policies`] certifies strictly ascending policy rows. Confidence and option types constrain +//! individual scalar domains. They do not certify instance uniqueness, reading counts or that a +//! policy's class components sum to at most one. //! //! # Weights //! @@ -33,28 +33,29 @@ //! c = c_link · √(c_source · c_target), //! ``` //! -//! where a missing score contributes the neutral factor 1 and sets a retained provenance bit, the -//! reading share +//! where each score lies in `[0, 1]`. A missing score contributes the neutral factor 1, and a +//! presence bit records each supplied score. The reading share is //! //! ```text //! s = 1 / multiplicity, //! ``` //! -//! which distributes one link's worth of force over the edge's relation readings - a multi-typed -//! link is a mixture of its types' geometric opinions, never a sum, while parallel links remain -//! independent assertions at full strength - and the degree normalization +//! which makes a complete set of one edge's readings a mixture of its types' geometric weights +//! before degree normalization. Parallel edges remain independent assertions. This conservation is +//! a real-arithmetic law for the shares and makes no claim about equal realized force after degree +//! normalization. The degree normalization is //! //! ```text //! ν = 1 / √((1 + degree_r(i)) · (1 + degree_r(j))), //! ``` //! -//! where `degree_r` sums the shares of the relation's admitted instances at a row, so an edge -//! contributes one unit of degree across its readings at each endpoint. Degrees always cover the -//! complete admitted instance set: force pruning drops an edge from sampling without reweighting -//! its neighbours. The persisted per-instance factor is the combined normalization `ν · s`. +//! where `degree_r` sums the shares of every non-self instance of relation `r` incident to the row. +//! A complete set of readings contributes one unit across relations at each endpoint in real +//! arithmetic. Degrees cover all non-self instances before pruning: removing an edge from sampling +//! does not reweight its neighbours. The persisted per-instance factor is `ν · s`. //! -//! Protection is exempt from the share on purpose: evidence aggregates by maximum, and a fractional -//! reading still fully asserts its relation - conservation for geometry, conjunction for safety. +//! Protection uses undivided evidence. Aggregation by maximum lets one reading assert its full +//! evidence even when its attraction share is fractional. //! //! The per-relation group carries the class weights //! @@ -62,23 +63,29 @@ //! coincident = κ_C · p*_C, proximal = p*_P, //! ``` //! -//! the shared Coincident coefficient applied to the effective attraction distribution `p*`; -//! Proximal's unit coefficient is the scale convention of normalized distance. The group's frozen -//! strength multiplier completes the factors. +//! where `p*` is the policy's effective attraction distribution over the [geometry +//! classes](crate::salt::policy::GeometryClass), and `κ_C` is the non-negative Coincident +//! coefficient. Proximal's unit coefficient fixes the scale convention. The group's frozen +//! non-negative strength multiplier completes the factors. +//! +//! Confidence, shares, degree prefixes and combined normalization compute in `f64`. Class weights +//! narrow to `f32`. These operations round, including prefix subtraction for a row's degree, and +//! confidence products can underflow to zero. Fixed sorted order and fixed emission +//! chunks make build results independent of input order and thread scheduling under the uniqueness +//! contract, at the same floating-point implementation. //! //! # Protection //! -//! Protection evidence derives from the selected class distribution `p` and the calibrated -//! applicability `a`: per instance, the applicability-discounted evidence `c · (p_C + p_P) · a` and -//! the undiscounted evidence `c · (p_C + p_P)`, each aggregated by maximum over every instance of -//! an endpoint pair, including instances of different relations and parallel links. A channel's -//! mass under an applicability floor `F` is then exactly `max(discounted, F · undiscounted)`, -//! because the maximum distributes over the per-instance `max(a, F)` - so floors and admission -//! thresholds are both query-time parameters ([`protection::ProtectionView::judge`]), and one built -//! index serves every floor and threshold calibration, including the floor-ablation matrix, -//! unchanged. The index is a symmetric sparse matrix over the node-row domain -//! ([`protection::ProtectionIndex`]): row `i` lists every protected partner of node row `i`, the -//! shape hard-negative mining vets one projected point's candidates against. +//! Protection derives from the selected class distribution `p` and calibrated applicability `a ∈ +//! [0, 1]`. It aggregates the discounted evidence `c · (p_C + p_P) · a` and undiscounted evidence +//! `c · (p_C + p_P)` by independent maxima over all non-self instances of each endpoint pair, +//! across relations and parallel links. +//! +//! A channel's mass under applicability floor `F ∈ [0, 1]` is `max(discounted, F · undiscounted)`. +//! [`protection`] gives the factorization and its rounding convention. Floors and admission +//! thresholds are query-time parameters of [`protection::ProtectionView::judge`]. One built index +//! supports the full floor/threshold calibration grid. Its symmetric sparse matrix exposes each +//! row's partners for both pair lookups and row-wise candidate checks. use hashql_core::id::Id; @@ -87,8 +94,6 @@ pub(crate) use self::{ confidence::{EffectiveConfidence, RelationConfidence, Scored}, error::RelationIndexError, }; -// The policy row vocabulary is `salt::policy`'s deliverable; the -// certified `Policies` view over it stays here with its consumer. #[cfg(test)] pub(crate) use crate::salt::policy::ClassProbabilities; pub(crate) use crate::salt::policy::RelationPolicy; @@ -100,8 +105,7 @@ use crate::{ pub(crate) mod artifact; pub(crate) mod attraction; -// Fully public: the root `bench` facade re-exports it; the private -// module chain above keeps it unreachable except through the facade. +// public for the root bench facade's re-export. #[cfg(feature = "bench")] pub mod bench; mod build; @@ -112,13 +116,13 @@ pub(crate) mod protection; #[cfg(test)] mod tests; -/// One admitted link instance: an edge row read under one of its relation types. +/// One admitted reading of an edge row under a relation type. /// -/// A link entity carrying several relation types yields one instance per type, all referencing the -/// same edge row and confidence scores. Each instance carries the share `1 / multiplicity` of the -/// edge's force, so the edge's total force mass is one link's worth regardless of how many types it -/// carries. The caller admits instances (security mode and conflict quarantine are upstream -/// concerns); every instance handed to the build participates. +/// A link entity carrying multiple relation types yields one instance per type, all referencing the +/// same edge row and confidence scores. With `multiplicity` equal to the admitted reading count, +/// the shares `1 / multiplicity` sum to one before degree normalization, up to rounding. Admission +/// belongs to the producer. The build drops self-references and applies attraction pruning to the +/// remaining instances. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct RelationInstance { /// The edge row that produced the instance. @@ -131,7 +135,10 @@ pub(crate) struct RelationInstance { pub target: N, /// The link's confidence scores. pub confidence: RelationConfidence, - /// The edge's total reading count across its relation types, at least 1. + /// The edge's total admitted reading count across its relation types, at least 1. + /// + /// The build clamps zero to one without checking that this count matches the supplied + /// readings. pub multiplicity: u32, } @@ -151,9 +158,8 @@ impl RelationInstance { /// A certified relation policy table. /// /// Construction checks the strictly ascending relation order once, or adopts it from an owned -/// [`CertifiedPolicies`] whose construction already proved it. Every value's domain rides in -/// the policy's field types. Lookups and the build consume the table without further -/// validation. +/// [`CertifiedPolicies`] with the same ordering invariant. The policy's field types constrain +/// individual values. Certification does not check relationships between class probabilities. #[derive(Debug, Copy, Clone)] pub(crate) struct Policies<'policy>(&'policy [RelationPolicy]); @@ -162,7 +168,7 @@ impl<'policy> Policies<'policy> { /// /// # Errors /// - /// Returns an error when the policies are not strictly ascending by relation row. + /// Returns [`RelationIndexError`] for a policy-order violation. #[cfg(any(test, feature = "bench"))] pub(crate) fn new( policies: &'policy [RelationPolicy], @@ -181,8 +187,8 @@ impl<'policy> Policies<'policy> { /// Looks up a relation's policy. /// - /// Returns [`None`] when the table does not cover the relation. Time is `O(log R)` in the table - /// length. + /// Returns [`None`] when the table does not cover the relation. Time is `O(log(R + 2))` for + /// table length `R`. #[must_use] pub(crate) fn get(self, relation: OntologyRowId) -> Option<&'policy RelationPolicy> { self.0 @@ -193,7 +199,6 @@ impl<'policy> Policies<'policy> { } impl<'policy> From<&'policy CertifiedPolicies> for Policies<'policy> { - /// Adopts the owned table's construction fact without re-checking it. fn from(certified: &'policy CertifiedPolicies) -> Self { Self(certified.as_slice()) } @@ -220,15 +225,18 @@ pub(crate) struct BuildMeasurements { pub self_references: usize, /// The edge multiplicity histogram. /// - /// Entry `i` counts edges carrying `i + 1` relation readings. + /// Entry `i` counts edges carrying `i + 1` relation readings. [`RelationIndexes::build`] + /// leaves this empty. The edge drain supplies the histogram separately. pub multi_typed_edges: Vec, } impl BuildMeasurements { /// Returns the fraction of total force mass the pruning dropped. /// - /// This quantity audits the pruning threshold: a threshold is admissible while the omitted - /// fraction stays numerically negligible. An instance set without positive mass omits nothing. + /// Compare this fraction with an acceptable omitted-mass budget when choosing a pruning + /// threshold. It measures `c · s · s+`, excluding degree normalization, frozen strength and + /// class-energy derivatives, and does not bound layout movement. An instance set without + /// positive mass returns zero. #[must_use] #[cfg(any(test, feature = "bench"))] pub(crate) fn omitted_mass_fraction(&self) -> crate::math::UnitFraction { @@ -240,8 +248,9 @@ impl BuildMeasurements { return UnitFraction::ZERO; } - // Adding the non-negative retained mass never rounds the sum below the pruned mass, so - // the quotient lies in [0, 1]; an overflowed total gives a zero quotient, still in domain. + // A rounded sum of finite non-negative masses is at least each operand. Both stored masses + // are finite, even if their sum overflows to infinity. Therefore the quotient lies in [0, + // 1], with an infinite total yielding zero. UnitFraction::new(self.pruned_mass.get() / total) .expect("a non-negative share of a total at least as large lies in [0, 1]") } @@ -249,12 +258,12 @@ impl BuildMeasurements { /// The relation-force and no-repel structures of one generation. /// -/// Both indexes derive from the same admitted instance set in one build, so the edge an attraction -/// group weights and the pair a protection evidence entry covers can never disagree about the -/// underlying link. +/// [`Self::build`] derives both indexes from the same admitted instance set. Protection includes +/// every non-self pair, including pairs whose attraction instances all prune. The public fields do +/// not validate shared provenance when assembled separately. #[derive(Debug, Clone)] pub(crate) struct RelationIndexes { - /// Force-bearing instances grouped by relation type. + /// Retained link instances grouped by relation type. pub attraction: attraction::AttractionIndex, /// The symmetric per-row no-repel evidence matrix. pub protection: protection::ProtectionIndex, @@ -265,29 +274,35 @@ pub(crate) struct RelationIndexes { impl RelationIndexes { /// Builds both indexes from the generation's admitted link instances. /// - /// `rows` is the node-row domain the protection matrix spans, and every instance endpoint lies - /// in it under the dataset row contract. The build reorders the instances in place, and both - /// indexes are functions of the instance set alone, identical for any input order. The build - /// drops instances whose endpoints are one row and counts them in the measurements, because - /// they exert no force between distinct points and protect nothing. Degrees and protection - /// evidence cover the complete remaining instance set regardless of pruning. + /// `rows` is the node-row domain the protection matrix spans. Endpoints must lie in this + /// domain, and `N` must represent the domain's row positions and end fencepost. Instances must + /// satisfy the module's uniqueness and multiplicity contract. The build reorders them in place + /// and drops self-references, counting each dropped instance. Degrees and protection evidence + /// cover the complete remaining instance set regardless of pruning. + /// + /// Groups build concurrently. Within a group, emission uses fixed-position chunks with partial + /// masses combined in chunk order. Together with the unique sort keys, this gives input-order + /// and scheduling independence at the same floating-point implementation. + /// + /// # Complexity + /// + /// For `E` instances, `R` policy rows and `N` node rows, time is: + /// + /// `O(E log(E + 1) + E log(R + 1) + N)`. /// - /// Sorting and emission are parallel at two levels. Groups build concurrently, and a group's - /// instances emit over fixed-position chunks, so one high-volume relation cannot serialize the - /// pass. The fixed boundaries keep the double-precision mass sums a function of the instance - /// set alone. Time is `O(E log E)` in the instance count. Beyond the returned indexes the build - /// allocates one two-column endpoint scratch per relation group and one per-instance protection - /// record buffer. + /// Working storage is `O(E + N)` beyond the returned indexes: endpoint columns and their share + /// prefixes, per-chunk edge buffers, group ranges, protection records and row cursors. The + /// final protection scatter is sequential. /// /// # Errors /// - /// Returns an error when an instance references a relation the policy table does not cover, or - /// `rows` exceeds the protection matrix's `u32` column encoding. + /// Returns [`RelationIndexError`] for an oversized row domain or a non-self instance without a + /// policy. The row bound is checked first, then missing policies in ascending relation order. /// /// # Panics /// - /// This panics when an instance endpoint lies outside the `rows` domain, which the dataset row - /// contract excludes. + /// Panics when a non-self endpoint is outside `rows` or `N` cannot represent a required row + /// position or fencepost. pub(crate) fn build( rows: usize, policies: Policies<'_>, diff --git a/libs/@local/graph/atlas/src/salt/relation/protection.rs b/libs/@local/graph/atlas/src/salt/relation/protection.rs index 3b5a564709a..9c403a5906c 100644 --- a/libs/@local/graph/atlas/src/salt/relation/protection.rs +++ b/libs/@local/graph/atlas/src/salt/relation/protection.rs @@ -5,27 +5,29 @@ //! repulsion at specific pairs while ordinary sampled negatives spread it across many pairs. Each //! channel carries its own applicability floor and admission threshold. //! -//! The index stores evidence and computes judgement. It holds one [`PairEvidence`] per linked pair -//! (the maxima of the applicability-discounted and undiscounted class evidence over the pair's -//! instances), and every channel's mass under a floor `F` is +//! One [`PairEvidence`] holds the independently aggregated discounted and undiscounted maxima over +//! a pair's instances. With `uᵢ` the non-negative undiscounted evidence, applicability `aᵢ ∈ [0, +//! 1]` and floor `F ∈ [0, 1]`, the governing identity is //! //! ```text -//! m_F = max(discounted, F · undiscounted), +//! maxᵢ(uᵢ · max(aᵢ, F)) = max(maxᵢ(uᵢ · aᵢ), F · maxᵢ(uᵢ)). //! ``` //! -//! exactly, because the maximum distributes over the per-instance `max(a, F)`: `max_i(c_i p_i -//! max(a_i, F)) = max(max_i(c_i p_i a_i), F max_i(c_i p_i))`. Floors and thresholds are therefore -//! both query-time parameters ([`ProtectionConfig`]), and one built index serves every floor and -//! threshold calibration unchanged. +//! Non-negative multiplication is monotone and distributes over a finite maximum. The build narrows +//! `cᵢ · (p_C + p_P)` once to `f32` as `uᵢ` and narrows `aᵢ` once. It computes discounted evidence +//! from those shared values. Rounded non-negative multiplication remains monotone, preserving the +//! identity numerically for these rounded inputs. Therefore `max(discounted, F · undiscounted)` +//! reproduces per-instance flooring without rebuilding the index. It does not recover unrounded +//! real-valued evidence. Floors and thresholds remain query-time parameters of +//! [`ProtectionConfig`]. //! -//! The index is a symmetric compressed sparse row matrix over the node-row domain, with each linked -//! pair stored in both of its rows with bit-equal evidence. Row `i` lists every partner whose link -//! protects the pair, which is the shape hard-negative mining consumes when it vets the candidates -//! of one projected point. +//! The index is a symmetric compressed sparse row matrix over the node-row domain. Construction +//! copies each pair's aggregate into both directions. Validation requires numerically equal +//! evidence, admitting opposite signs of zero. A stored pair need not pass a channel's threshold. +//! Row-wise access supports checking a point's candidate list. //! -//! Protection is blind to attraction strength. Class coefficients, degree normalization, strength, -//! and force pruning answer how strongly an admitted force pulls, while protection answers whether -//! repulsion is safe, so none of those factors enters the evidence. +//! Protection uses class evidence independently of attraction strength. Class coefficients, reading +//! shares, degree normalization, frozen strength and force pruning never enter its evidence. #![expect(clippy::empty_enums, reason = "zerocopy uses them in the derive")] use core::{ @@ -42,7 +44,7 @@ use crate::{ math::NonNegative, }; -/// The index's matrix layout of evidence values, `u32` partner columns, and `u64` row pointers. +/// A sparse evidence matrix with `u32` partner columns and `u64` row pointers. pub(crate) type ProtectionMatrix = CsMatI; /// A borrowed [`ProtectionMatrix`]. @@ -51,14 +53,12 @@ pub(crate) type ProtectionMatrixView<'view> = CsMatViewI<'view, PairEvidence, u3 /// One linked pair's aggregated class evidence. /// /// Both components take the maximum over every admitted instance between the pair's rows, parallel -/// links and distinct relations alike. One strong link suffices to veto repulsion, however many -/// weak ones accompany it. Per instance, the class evidence is the effective confidence times the -/// selected Coincident and Proximal probability. `discounted` additionally multiplies the -/// relation's calibrated applicability. The index validates both components finite, non-negative, -/// and ordered `discounted ≤ undiscounted`. -// FromBytes on purpose: the components carry no construction invariant -// of their own - the index validates its entries as a whole, exactly -// like the semantic graph's mapped weights. +/// links and distinct relations alike. One instance suffices to veto repulsion when its floored +/// evidence reaches the channel threshold, however many weaker instances accompany it. The +/// [module's rounding convention](super::protection) defines the components. [`ProtectionIndex`] +/// validates finiteness, non-negativity and `discounted ≤ undiscounted`, but does not verify their +/// derivation from link instances. +// raw f32 fields admit every bit pattern. ProtectionIndex validates their joint evidence contract. #[derive( Debug, Copy, @@ -81,8 +81,8 @@ pub(crate) struct PairEvidence { impl PairEvidence { /// Returns the pair's evidence mass under an applicability floor. /// - /// This is the exact per-channel mass: the floor's `max(a, F)` distributes through the - /// per-instance maximum into `max(discounted, floor · undiscounted)`. + /// Returns `max(discounted, floor · undiscounted)` without validating either input. For the + /// module's floor identity, `self` must contain valid aggregated evidence. #[inline] #[must_use] pub(crate) fn mass(self, floor: f32) -> f32 { @@ -91,8 +91,7 @@ impl PairEvidence { } impl SprsValue for PairEvidence { - // Opaque on purpose: the pair is this stage's vocabulary, not a scalar the format vocabulary - // pins. Width is the wire identity. + // opaque values identify their layout by width, leaving evidence semantics to this index. const TAG: ValueTag = ValueTag::Opaque; } @@ -111,7 +110,7 @@ pub(crate) struct ChannelConfig { } impl ChannelConfig { - /// Creates a channel configuration. + /// Returns whether the floored mass reaches the channel's threshold. /// /// Returns [`None`] unless the floor lies in `0.0..=1.0` and the threshold is finite and /// non-negative. The default is floor 0, threshold 0. @@ -176,9 +175,9 @@ impl ProtectionConfig { /// Creates a protection configuration from the two channels. /// /// Returns [`None`] unless `ordinary.floor ≤ hard.floor` and `hard.threshold ≤ - /// ordinary.threshold`. `protect_ordinary` disables the ordinary channel outright, so every - /// ordinary negative passes while hard-negative protection stands. The default is both channels - /// at floor 0 and threshold 0 with both active. + /// ordinary.threshold`. Setting `protect_ordinary` to `false` disables only the ordinary + /// protection veto. Other negative-sampling exclusions remain independent. By default both + /// channels use floor zero and threshold zero, with ordinary protection enabled. #[must_use] pub(crate) const fn new( hard: ChannelConfig, @@ -220,8 +219,12 @@ impl ProtectionConfig { /// An unordered pair of node rows in canonical order. /// -/// The smaller row becomes [`lhs`](Self::lhs) and the larger becomes [`rhs`](Self::rhs), so a pair -/// equals itself however its rows arrive, and the derived order is total over pairs. +/// [`Self::lhs`] is the smaller row and [`Self::rhs`] the larger. Equal endpoints are allowed. +/// Pairs order lexicographically by these canonical endpoints. +/// +/// # Properties +/// +/// For all row ids `a` and `b`, `NodePair::new(a, b) == NodePair::new(b, a)`. #[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)] pub(crate) struct NodePair { lhs: N, @@ -281,7 +284,7 @@ impl PairVerdict { }; } -/// One protected partner of a row. +/// One stored partner and its evidence, before channel judgment. #[derive(Debug, Copy, Clone, PartialEq)] #[cfg(any(test, feature = "bench"))] pub(crate) struct ProtectedPartner { @@ -348,10 +351,18 @@ impl core::error::Error for ProtectionValidationError {} /// Checks every index invariant over a borrowed matrix. /// -/// Rows check in parallel; the reported violation is the first in row order regardless of -/// scheduling, so failures are deterministic. -// The symmetry check compares the two directions bit-exactly (derived PartialEq over the f32 -// components): one aggregated value produces both, so bit equality is the constructed contract. +/// Checks compression and shape before checking rows in parallel. The reported row violation is the +/// first in row order regardless of scheduling. Symmetry uses numeric `f32` equality, treating +/// positive and negative zero as equal. +/// +/// # Errors +/// +/// Returns [`ProtectionValidationError`] for a compression, shape or row-evidence violation. +/// +/// # Complexity +/// +/// For `N` rows, `M` entries and maximum row length `d`, time is `O(N + M log(d + 1))`. +/// Reverse-entry lookups account for the logarithmic factor. pub(super) fn validate(matrix: ProtectionMatrixView<'_>) -> Result<(), ProtectionValidationError> { if !matrix.is_csr() { return Err(ProtectionValidationError::ColumnCompressed); @@ -368,7 +379,16 @@ pub(super) fn validate(matrix: ProtectionMatrixView<'_>) -> Result<(), Protectio .map_or(Ok(()), Err) } -/// Checks one row's entries against the index invariants. +/// Checks one row's values and reverse entries in a square CSR matrix. +/// +/// # Errors +/// +/// Returns [`ProtectionValidationError`] for an invalid self-edge, evidence value, ordering or +/// reverse entry. +/// +/// # Panics +/// +/// Panics when `row` is outside the matrix's outer dimension. fn validate_row( matrix: ProtectionMatrixView<'_>, row: usize, @@ -410,10 +430,17 @@ fn validate_row( /// The symmetric no-repel evidence matrix of one generation. /// -/// Row `i` stores the evidence of every protected pair at node row `i`, keyed by the other endpoint -/// in ascending row order. Every pair occupies both of its rows with bit-equal evidence, and no row -/// references itself. A pair absent from the matrix has no admitted link between its rows and stays -/// unprotected under every configuration. +/// Each row lists stored partners in strictly ascending order. Every pair occupies both directions +/// with numerically equal evidence, and no row references itself. A pair absent from the matrix is +/// unprotected under every configuration. Validation establishes these matrix properties, not +/// completeness or provenance relative to a dataset. +/// +/// Writing through [`crate::file::WriteInto`] returns an error for zero rows. +/// +/// # Panics +/// +/// Writing a nonempty matrix with a nonzero initial row pointer panics, although [`Self::new`] +/// accepts such offset pointers. #[derive(Debug, Clone)] pub(crate) struct ProtectionIndex(ProtectionMatrix, PhantomData); @@ -425,9 +452,8 @@ where /// /// # Errors /// - /// Returns an error when the matrix is not row-compressed, not square, self-referencing, stores - /// a non-finite, negative, or misordered evidence pair, or stores an edge whose two directions - /// are missing or unequal. + /// Returns [`ProtectionValidationError`] for a matrix-invariant violation. See [`validate`] for + /// check order and cost. pub(crate) fn new(matrix: ProtectionMatrix) -> Result { validate(matrix.view())?; Ok(Self(matrix, PhantomData)) @@ -456,10 +482,10 @@ impl<'view, N> ProtectionView<'view, N> where N: Id, { - /// Wraps a matrix whose invariants already hold. + /// Borrows an evidence matrix satisfying [`validate`]'s invariants. /// - /// The caller promises the matrix passed [`validate`]; the wrapper performs no checks of its - /// own. + /// The matrix must satisfy those invariants throughout the borrow. + // the unchecked promise concerns index semantics, not memory safety. #[inline] #[must_use] pub(super) const fn new_unchecked(matrix: ProtectionMatrixView<'view>) -> Self { @@ -481,11 +507,12 @@ where self.0.nnz() } - /// Returns row `row`'s protected partners in ascending row order. + /// Returns row `row`'s stored partners in ascending row order. /// /// # Panics /// - /// This panics when `row` is outside the matrix's row domain. + /// Panics when `row` is outside the matrix's row domain or a stored partner cannot be + /// represented by `N`. #[cfg(any(test, feature = "bench"))] pub(crate) fn row(&self, row: N) -> impl Iterator> + '_ { let (columns, evidence) = self @@ -505,8 +532,9 @@ where /// Looks up a pair's evidence. /// - /// Returns [`None`] when no admitted link connects the pair's rows, or either row lies outside - /// the row domain. Time is one row resolution plus a binary search of that row's partners. + /// Returns [`None`] when the pair is absent or either row lies outside the matrix domain. Both + /// ids must be representable as `usize`. Time is a binary search of the smaller endpoint's + /// stored partners. #[must_use] pub(crate) fn get(&self, pair: NodePair) -> Option { self.0 @@ -517,7 +545,8 @@ where /// Judges a pair's protection under the given configuration. /// /// A channel protects when the pair's evidence mass under the channel's floor reaches the - /// channel's threshold. A pair with no link evidence stays unprotected in both channels. + /// channel's threshold. An absent pair is unprotected in both channels. Both ids must be + /// representable as `usize`. #[must_use] pub(crate) fn judge(&self, pair: NodePair, config: ProtectionConfig) -> PairVerdict { let Some(evidence) = self.get(pair) else { diff --git a/libs/@local/graph/atlas/src/salt/relation/tests.rs b/libs/@local/graph/atlas/src/salt/relation/tests.rs index b8eb0ce6e3c..fef32132d30 100644 --- a/libs/@local/graph/atlas/src/salt/relation/tests.rs +++ b/libs/@local/graph/atlas/src/salt/relation/tests.rs @@ -48,7 +48,7 @@ fn proximal_policy(relation: u64) -> RelationPolicy { } } -/// An unscored instance of `relation` between `source` and `target`. +/// Creates an unscored instance of `relation` between `source` and `target`. fn instance( edge: u64, relation: u64, @@ -65,7 +65,7 @@ fn instance( } } -/// The instance with its link score set. +/// Sets the instance's link score. fn scored( mut base: RelationInstance, link: UnitFraction, @@ -74,6 +74,11 @@ fn scored( base } +/// Builds fixture indexes after certifying policy order. +/// +/// # Panics +/// +/// Panics when policies are unordered or the build rejects the fixture's row domain or instances. fn build( rows: usize, policies: &[RelationPolicy], @@ -89,6 +94,11 @@ fn build( .expect("the fixture instances satisfy the input contract") } +/// Builds fixture indexes over [`ROWS`] with default attraction options. +/// +/// # Panics +/// +/// Panics when [`build`] rejects the fixture. fn build_default( policies: &[RelationPolicy], instances: Vec>, @@ -96,11 +106,16 @@ fn build_default( build(ROWS, policies, instances, AttractionOptions::default()) } +/// Creates a canonical node pair from literal row numbers. fn pair(one: u64, other: u64) -> NodePair { NodePair::new(NodeRowId::new(one), NodeRowId::new(other)) } -/// Builds a protection configuration from each channel's floor and threshold pair. +/// Builds a protection configuration with both channels enabled. +/// +/// # Panics +/// +/// Panics for out-of-domain settings or incorrectly ordered channels. fn config(hard: (f32, f32), ordinary: (f32, f32)) -> ProtectionConfig { ProtectionConfig::new( ChannelConfig::new(hard.0, hard.1).expect("the fixture channel is in domain"), @@ -112,8 +127,8 @@ fn config(hard: (f32, f32), ordinary: (f32, f32)) -> ProtectionConfig { #[test] fn degree_normalization_counts_the_relations_complete_instance_set() { - // Node 0 sources three instances and node 1 receives three, so the - // 0 → 1 edge sees (1 + 3)(1 + 3) = 16 and ν = 0.25 exactly. + // node 0 sources three instances and node 1 receives three. For 0 → 1, (1 + 3) · (1 + 3) = 16 + // and ν = 0.25 exactly. let indexes = build_default( &[proximal_policy(0)], vec![ @@ -138,9 +153,7 @@ fn degree_normalization_counts_the_relations_complete_instance_set() { #[test] fn degrees_are_per_relation() { - // The same endpoints under a second relation contribute nothing to - // the first relation's degrees: each relation's 0 → 1 edge sees - // (1 + 1)(1 + 1) = 4. + // degrees are relation-local: each relation's 0 → 1 edge sees (1 + 1) · (1 + 1) = 4. let indexes = build_default( &[proximal_policy(0), proximal_policy(1)], vec![instance(0, 0, 0, 1), instance(1, 1, 0, 1)], @@ -174,6 +187,9 @@ fn group_weights_carry_the_policy_and_coefficient() { AttractionOptions::new(non_negative!(2.0), non_negative!(0.0)), ); + // the Coincident coefficient 2 multiplies attraction probability 0.25 to give 0.5. + // The Proximal weight is attraction probability 0.5. Their sum gives scale 1, while + // strength remains the separate multiplier 2. let weights = indexes.attraction.groups()[0].weights(); assert_eq!(weights.coincident, non_negative!(0.5)); assert_eq!(weights.proximal, non_negative!(0.5)); @@ -262,8 +278,8 @@ fn pruning_splits_mass_at_the_threshold_inclusively() { #[test] fn pruned_instances_keep_their_degree_contributions() { - // Every instance but the first prunes at zero confidence, yet the - // retained 0 → 1 edge still sees both endpoints at degree 3. + // every instance but the first prunes at zero confidence. The retained 0 → 1 edge still sees + // both endpoints at degree 3. let indexes = build( ROWS, &[proximal_policy(0)], @@ -285,7 +301,7 @@ fn pruned_instances_keep_their_degree_contributions() { #[test] fn pruning_never_reaches_protection() { - // An instance pruned from attraction still protects its pair. + // the pruning predicate excludes attraction without discounting the pair evidence. let indexes = build( ROWS, &[proximal_policy(0)], @@ -468,9 +484,6 @@ fn empty_instances_build_empty_indexes() { #[test] fn policy_tables_certify_order() { - // Every value domain rides in the policy's field types, so ordering is the one contract - // left for certification to check; the domain assertion died when the last raw field - // (strength) took its type. assert_eq!( Policies::new(&[proximal_policy(1), proximal_policy(0)]) .expect_err("descending policies violate the order contract"), @@ -529,9 +542,8 @@ fn option_constructors_reject_out_of_domain_settings() { #[test] fn group_spanning_several_emission_chunks_matches_the_chain_reference() { - // A chain 0 → 1 → ... → n under one relation forces the group - // through multiple fixed emission chunks: source runs cross chunk - // boundaries, and every degree must still count the whole group. + // this chain spans four emission chunks. Interior degrees require both the source and target + // columns, including where adjacent edges occupy different chunks. let nodes = 3 * build::EMISSION_CHUNK + 7; let instances: Vec> = (0..nodes - 1) .map(|link| instance(link as u64, 0, link as u64, link as u64 + 1)) @@ -559,8 +571,8 @@ fn group_spanning_several_emission_chunks_matches_the_chain_reference() { assert_eq!(edge.normalization.get(), expected, "edge {position}"); } - // Every unscored instance carries mass exactly 1.0, so the chunked - // double-precision partial sums are exact whatever the chunking. + // every unscored instance carries mass exactly 1.0. At this fixture size every partial sum is + // an exactly representable integer, independent of chunking. #[expect( clippy::cast_precision_loss, reason = "the fixture size sits far below f64 integer precision" @@ -570,7 +582,11 @@ fn group_spanning_several_emission_chunks_matches_the_chain_reference() { assert_eq!(indexes.measurements.retained_edges, nodes - 1); } -/// Asserts two builds produced identical indexes, component by component. +/// Compares every stored index component and build measurement. +/// +/// # Panics +/// +/// Panics on unequal measurements, sparse storage, groups, weights or edges. #[track_caller] fn assert_indexes_equal( one: &RelationIndexes, @@ -601,7 +617,7 @@ fn assert_indexes_equal( } prop_compose! { - /// Instances over three relations and eight rows. + /// Generates instances over three relations and eight rows. /// /// Edge rows are unique, and optional scores are arbitrary. fn arbitrary_instances()( @@ -637,9 +653,6 @@ prop_compose! { } } -/// The build is a function of the instance set, not its order. -/// -/// The output orders are the documented invariants. #[property_test] fn build_is_order_independent_and_sorted( #[strategy = arbitrary_instances()] instances: Vec>, @@ -730,6 +743,7 @@ fn build_is_order_independent_and_sorted( NodePair::new(NodeRowId::new(row), entry.partner), floor, ); + prop_assert_eq!(entry.evidence.mass(floor), expected); } } @@ -805,8 +819,8 @@ fn published_attraction_index_reopens_mapped() { let _: Result<(), std::io::Error> = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).expect("the temp directory is writable"); - // The fixture uses two relations with distinct weights and one scored instance, so the - // provenance bits round-trip a non-default value. + // distinct relation weights and a scored instance exercise non-default factors and provenance + // bits. let policies = [ RelationPolicy { attraction: ClassProbabilities { @@ -936,10 +950,9 @@ fn corrupted_attraction_file_names_its_broken_invariant() { InvalidAttractionIndex::RowOutOfDomain { edge: 0 }, ); - // A confidence above one refuses at the file parse itself: the - // record's field type admits only [0, 1], so the corruption never - // reaches the index validation. The confidence field sits 24 - // bytes into the record. + // the typed file parse rejects confidence above one before index validation. The field starts + // 24 bytes into the record. Its native-f64 encoding makes this little-endian fixture + // host-specific. let mut confident = bytes.clone(); confident[8192 + 24..8192 + 32].copy_from_slice(&2.0_f64.to_le_bytes()); let refused = dir.join("confidence.atrc"); @@ -986,9 +999,13 @@ fn corrupted_attraction_file_names_its_broken_invariant() { let _: Result<(), std::io::Error> = std::fs::remove_dir_all(&dir); } -/// The floored pair mass computed instance by instance. +/// Computes floored pair mass before maximum factorization. +/// +/// Uses the build's f32 evidence and applicability rounding boundaries. +/// +/// # Panics /// -/// The pre-factorization form of the protection evidence. +/// Panics when a non-self instance matching `pair` has no policy. fn forward_reference_mass( instances: &[RelationInstance], policies: &[RelationPolicy], @@ -1008,8 +1025,7 @@ fn forward_reference_mass( .expect("the fixture policies cover every relation"); let confidence = instance.confidence.effective().value(); let positive = f64::from(policy.selected.coincident) + f64::from(policy.selected.proximal); - // The mirror narrows where the build narrows: one narrow derives the undiscounted - // evidence, and the floored discount scales that shared f32 value. + // share the build's narrowed undiscounted value before applying the floor. let undiscounted = narrow_f32(confidence * positive) .expect("a fraction of a finite f32 factor narrows finitely"); let applicability = @@ -1019,7 +1035,7 @@ fn forward_reference_mass( mass } -/// The instance carrying `multiplicity` readings of its edge. +/// Sets the instance's reading count. fn multi( mut base: RelationInstance, multiplicity: u32, @@ -1030,9 +1046,8 @@ fn multi( #[test] fn two_typed_edge_carries_the_mean_of_its_readings_not_the_sum() { - // One edge read under two relations at multiplicity 2 versus the same two readings as - // independent single-typed edges. The mixture halves each reading's mass, so the total is the - // mean. Every factor is a power of two, so the arithmetic is exact. + // one edge read twice at multiplicity 2 has shares 0.5. Against two independent single-typed + // edges, the pre-normalization mass is exactly half for these unit weights and confidences. let policies = [proximal_policy(0), proximal_policy(1)]; let mixed = build_default( &policies, @@ -1048,9 +1063,8 @@ fn two_typed_edge_carries_the_mean_of_its_readings_not_the_sum() { separate.measurements.retained_mass.get() / 2.0, ); - // Each group holds the reading at half a link's force: share 0.5 - // on the mass and share-weighted degrees 0.5 at both endpoints, - // so the persisted factor is 0.5 / √(1.5 · 1.5). + // each endpoint has share-weighted degree 0.5. The persisted factor is the rounded value of 0.5 + // / √(1.5 · 1.5), not half the single-reading normalization. let expected = PositiveUnitFraction::new(0.5 / (1.5_f64 * 1.5).sqrt()) .expect("the reference factor lies in (0, 1]"); for group in mixed.attraction.groups() { @@ -1062,13 +1076,13 @@ fn two_typed_edge_carries_the_mean_of_its_readings_not_the_sum() { #[test] fn two_typed_realized_coefficients_sum_between_the_mean_and_its_double() { - // Under the narrowed conservation law, shares conserve the pre-ν mass - // exactly, while the realized coefficients (ν · s) of an unpruned - // k-typed edge sum to T with M ≤ T < 2M against the mean M of its - // single-typed counterfactuals. The isolated 2-typed edge realizes - // exactly 4/3 · M: each reading's ν is 1/(1 + 1/2) = 2/3 against - // the counterfactual 1/2, so T = 2 · (1/2 · 2/3) = 2/3 over - // M = 1/2. + // For fixed non-negative background endpoint degrees B and share s ∈ (0, 1], each ratio (2 + + // B)/(1 + B + s) lies in [1, 2). Taking the geometric mean at the two endpoints gives the ratio + // of shared to single-reading degree normalizations. Therefore, in real arithmetic and without + // pruning, the sum T of a k-typed edge's ν · s coefficients satisfies M ≤ T < 2M, where M is + // the mean of its single-reading counterfactual coefficients with all other degrees fixed. + // This isolated two-typed fixture has s = 0.5 and B = 0: T = 2/3 versus M = 1/2, a ratio of + // 4/3. The stored thirds round in f64. let policies = [proximal_policy(0), proximal_policy(1)]; let mixed = build_default( &policies, @@ -1095,16 +1109,15 @@ fn two_typed_realized_coefficients_sum_between_the_mean_and_its_double() { assert!(total >= mean); assert!(total < 2.0 * mean); - // The exact fixture ratio, at the double precision the build keeps per reading. + // compare the same rounded per-reading expression used by the build. let per_reading = 0.5 / (1.5_f64 * 1.5).sqrt(); assert_eq!(total, 2.0 * per_reading); } #[test] fn protection_evidence_ignores_multiplicity() { - // The same pair under one relation, single-typed versus 4-typed: - // protection aggregates by maximum over undivided evidence, so a - // fractional reading still fully vetoes. + // varying multiplicity on one supplied reading isolates the evidence computation from the + // share. This fixture does not supply a complete four-type edge. let policies = [proximal_policy(0)]; let single = build_default(&policies, vec![instance(0, 0, 1, 2)]); let quartered = build_default(&policies, vec![multi(instance(0, 0, 1, 2), 4)]); @@ -1121,10 +1134,9 @@ fn protection_evidence_ignores_multiplicity() { #[test] fn single_typed_builds_are_unchanged_by_the_share_machinery() { - // Shares of 1.0 sum to exact integer degrees and multiply masses - // by exactly 1: the k = 1 path is bit-identical to the pre-share - // arithmetic. Both edges meet at row 1, so the shared endpoint's - // degree is 2 and the far endpoints' degrees are 1. + // unit shares give exact integer degrees at this fixture size. Row 1 has degree 2 and each far + // endpoint has degree 1, giving the rounded normalization 1 / √(3 · 2) and total mass exactly + // 2. let policies = [proximal_policy(0)]; let indexes = build_default(&policies, vec![instance(0, 0, 1, 2), instance(1, 0, 1, 3)]); diff --git a/libs/@local/graph/atlas/src/salt/runner/error.rs b/libs/@local/graph/atlas/src/salt/runner/error.rs index d593c9eaaa8..9b3f77a204c 100644 --- a/libs/@local/graph/atlas/src/salt/runner/error.rs +++ b/libs/@local/graph/atlas/src/salt/runner/error.rs @@ -1,5 +1,3 @@ -//! Generation-run errors. - use core::{error::Error, fmt}; use crate::{ @@ -7,18 +5,19 @@ use crate::{ salt::{fit::FitError, quality::error::QualityRunError}, }; -/// The run could not reach a verdict. +/// Failure to complete a generation run. /// -/// Variants after the fit carry the published generation's identity: the artifacts are complete on -/// disk, and the remedy - reopening, re-probing, or correcting and re-running - starts from that -/// id. +/// Errors after a successful fit carry the published generation's identity for diagnosis or a +/// repeated probe. An activation failure can follow a passing quality verdict. #[derive(Debug)] pub(crate) enum RunnerError { /// The run could not read the current-generation pointer. Current(CurrentError), /// The run could not open the active generation as the prior. Prior(OpenError), - /// The fit could not publish, so nothing is on disk. + /// The fit did not return a successful publication. + /// + /// A publication error after the directory rename can leave a generation on disk. Fit(FitError), /// The run could not reopen the published generation. Reopen { id: GenerationId, source: OpenError }, @@ -27,7 +26,9 @@ pub(crate) enum RunnerError { id: GenerationId, source: QualityRunError, }, - /// The run could not activate the admitted generation. + /// Activation of the admitted generation did not complete successfully. + /// + /// An error after the pointer rename can leave the generation selected. Activate { id: GenerationId, source: ActivateError, diff --git a/libs/@local/graph/atlas/src/salt/runner/mod.rs b/libs/@local/graph/atlas/src/salt/runner/mod.rs index 63985ba8b33..928f3452374 100644 --- a/libs/@local/graph/atlas/src/salt/runner/mod.rs +++ b/libs/@local/graph/atlas/src/salt/runner/mod.rs @@ -1,16 +1,14 @@ //! One production run from snapshot to active generation. //! -//! [`run`] composes the pipeline's separate decisions into the one sequence production takes: the -//! prior comes from the root's active generation, [`fit`] publishes a complete verified generation, -//! the quality suite probes the published artifacts against the same snapshot, and a passing -//! verdict activates the generation by the atomic pointer flip. A failing verdict returns an -//! [`Outcome`] whose generation stays published as a candidate beside its report; the remedy is to -//! diagnose the report, correct data, configuration, or metric, and run again - a candidate never -//! activates by hand. +//! [`run`] resolves the prior from the root's active generation by default, publishes a generation +//! through [`fit`], then probes its artifacts against the same dataset snapshot. A passing quality +//! verdict activates the generation by atomically replacing the current pointer. A failing verdict +//! returns [`Admission::Candidate`] without activating it. The published artifacts remain available +//! for diagnosis, and [`Outcome`] returns the report in memory. //! -//! The whole run replays from the one fit seed. The admission probe's generator derives from that -//! seed under a pinned name, exactly as the fit stages derive theirs, so equal configurations -//! sample equal anchors. +//! Admission describes this run's decision. Another root operation can change the current pointer, +//! including while a run fits or probes. The runner's quality check supplies no restriction on +//! direct [`GenerationRoot::activate`] calls. //! //! Retiring old generations is offline tooling over published directories. @@ -41,49 +39,49 @@ pub(crate) mod operator; #[cfg(test)] mod tests; -/// Where one run's prior generation comes from. +/// Prior-generation selection for embedding reuse and landmark retention. #[derive(Debug, Copy, Clone, PartialEq, Eq, Default)] pub(crate) enum PriorMode { /// The root's active generation is the prior. /// - /// Card rows reuse embeddings by text hash and landmarks compete for the retained share. A root - /// without an activation runs fresh. + /// This is the default. Card rows reuse embeddings by text hash under a matching embedder + /// fingerprint, and prior landmarks compete for the retained share. A root without an + /// activation runs fresh. #[default] FromActive, - /// No prior: every card row embeds anew and the landmark selection starts cold. + /// Ignore the active generation for embedding reuse and landmark retention. /// - /// The reset for a changed embedding contract. + /// Every unique card text requires an embedding. Use this mode to run under a changed embedding + /// contract. Fresh, } -/// Every setting of one generation run. +/// Fit and admission settings for one generation run. #[derive(Debug, Clone, PartialEq)] pub(crate) struct RunnerOptions { /// The fit's settings, whose seed also derives the admission probe's sampling. pub fit: FitConfig, - /// Where the prior generation comes from. + /// Prior-generation source, [`PriorMode::FromActive`] by default. pub prior: PriorMode = PriorMode::FromActive, - /// The admission probe's sampling, grouping, and thresholds. + /// Admission sampling, grouping and thresholds, using [`QualityRunOptions::default`] by default. pub quality: QualityRunOptions = QualityRunOptions::default(), - /// The device the fit's tensor stages run on. The host-derived family by default. + /// The explicitly selected device for the fit's tensor stages. pub device: PhysicalDevice, } -/// How one published generation left the runner. +/// The runner's activation decision for a published generation. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub(crate) enum Admission { - /// The report's thresholds held and the root's pointer names the generation. + /// The report passed and this run successfully activated the generation. Active, - /// The report refused admission, so the generation stays published and unactivated. - /// - /// Serving it anyway is a decision recorded outside the runner. + /// The report refused admission, and this run did not activate the generation. Candidate, } /// The published generation of one finished run and its admission evidence. #[derive(Debug, Clone)] pub(crate) struct Outcome { - /// The published generation, reopened and verified against its identity. + /// The published generation reopened from its root. pub generation: Generation, /// The admission probe's full evidence record. pub report: QualityReport, @@ -93,21 +91,20 @@ pub(crate) struct Outcome { /// Runs one generation end to end and activates it on admission. /// -/// The dataset serves both halves of the run (the fit's ingest streams and the admission probe's -/// sampled lookups), so the probed corpus is the fitted corpus by construction. The classifier -/// input is a supplied fitted artifact or an annotation corpus the fit assembles and fits inside -/// the run ([`ClassifierInput`]). The reviewed verdicts are a supplied input the fit stages for the -/// trainer, and [`None`] runs without a review file. +/// The same [`Dataset`] supplies the fit's ingest streams and the admission probe's sampled +/// lookups. Its implementation must preserve one frozen snapshot across both phases. +/// [`ClassifierInput`] supplies either a fitted classifier or an annotation corpus to assemble and +/// fit. Supplied verdicts stage as a reviewed-verdicts artifact, and [`None`] runs without that +/// artifact. /// -/// A report that refuses admission returns [`Admission::Candidate`] with the generation published -/// and unactivated. Only failures that prevent a verdict are errors. +/// A report that refuses admission returns [`Admission::Candidate`] with the generation still +/// published. The runner returns the report rather than persisting it. An admitted generation +/// activates before the successful outcome returns. /// /// # Errors /// -/// Returns an error when the run cannot resolve the prior ([`RunnerError::Current`], -/// [`RunnerError::Prior`]), when the fit cannot publish ([`RunnerError::Fit`]), or, with the -/// published generation's identity attached, when the run cannot reopen the generation, when the -/// probe cannot produce a report, or when the run cannot activate the admitted generation. +/// Returns [`RunnerError`] when prior resolution, fitting, reopening, probing or activation fails. +/// Publication and activation errors can occur after their respective filesystem renames. pub(crate) async fn run( dataset: &D, embedder: &E, @@ -162,11 +159,9 @@ where .await .map_err(|source| RunnerError::Quality { id, source })?; - // The run reports the battery's readings here rather than from inside the probe, because the - // reading a control turns on is an extremum over the probe's steps and only exists once the - // report reduces them - and it is the same reduction the verdict reads. A control whose - // evidence is absent reports nothing: there is no measurement to observe, and the refusal is - // the report's to carry. + // controls reduce the probe's steps to the extrema that decide admission. Reporting that same + // reduction keeps the observer's readings consistent with the verdict. Missing evidence emits + // no reading and still causes the corresponding control to refuse admission. for control in report.controls() { if let Some(reading) = control.reading { progress.quality_probe(control.metric, reading); @@ -204,8 +199,14 @@ where /// Derives the admission probe's generator from the fit seed. /// -/// The pinned name keeps the derivation disjoint from every fit stage's, so the probe samples -/// independently of the fit's draws while the whole run replays from the one seed. +/// The generator seed is SHA-256 over the fit seed's eight little-endian bytes followed by +/// `admission-probe`. Keeping a separate generator makes the probe's draw sequence independent of +/// how many draws fitting consumes. +/// +/// # Properties +/// +/// For every fit seed, repeated calls produce the same generator state. Equal subsequent sampling +/// operations therefore produce equal draws. pub(crate) fn probe_rng(seed: u64) -> Xoshiro256PlusPlus { let mut hasher = Sha256::new(); #[expect( diff --git a/libs/@local/graph/atlas/src/salt/runner/operator/live.rs b/libs/@local/graph/atlas/src/salt/runner/operator/live.rs index 2e0710428ef..7b0c8b21553 100644 --- a/libs/@local/graph/atlas/src/salt/runner/operator/live.rs +++ b/libs/@local/graph/atlas/src/salt/runner/operator/live.rs @@ -14,9 +14,9 @@ use crate::{ /// Runs one production generation over the store's snapshot at `axes`. /// -/// The run publishes the generation under the generation root at `root`; the caller names the -/// snapshot explicitly, so equal inputs describe the same run. Cards embed through `embedder`, the -/// provider the shell constructed with its credentials. +/// The dataset holds one repeatable-read transaction across fitting and admission at the requested +/// temporal axes. Cards embed through `embedder`, and the generation publishes under `root`. +/// Repeating the axes alone does not freeze a later transaction to the same database snapshot. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/runner/operator/mod.rs b/libs/@local/graph/atlas/src/salt/runner/operator/mod.rs index 9e313ad9e1a..1def84613a7 100644 --- a/libs/@local/graph/atlas/src/salt/runner/operator/mod.rs +++ b/libs/@local/graph/atlas/src/salt/runner/operator/mod.rs @@ -1,24 +1,18 @@ -//! The operator entry points for one production run. +//! Operator configuration for live and offline generation runs. //! -//! [`live()`] drives the generation runner end to end over a pinned store snapshot, and -//! [`offline()`] drives the same runner over a dump directory, so a fit runs where the store does -//! not. Both -//! cover prior resolution, fit, admission probe, and the activation decision, configured by -//! [`Options`] and read back as a plain-number [`Summary`]. Failures return a [`RunError`] naming -//! the failing step, the step's concrete fault chained beneath. +//! Use [`live()`] for a pinned store snapshot or [`offline()`] for a dump directory when the store +//! is unavailable. Both resolve [`Options`] and supplied documents before fitting and admission, +//! and return statistics plus the admission report in [`Summary`]. [`RunError`] identifies a failed +//! step and retains its concrete error as a source. //! -//! Types carry the option vocabulary: [`ClassifierSource`] names the classifier supply every run -//! carries, and [`Placement`] carries exactly the controls its placer consumes, so option -//! combinations the pipeline cannot honor are unrepresentable. +//! [`ClassifierSource`] selects a supplied model or an annotation corpus. [`Placement`] keeps +//! projector controls on the trained-placement variant. Corpus-dependent constraints, including +//! whether the probe's sample fits, remain runtime checks. //! -//! A live run embeds cards through the external embedding provider the shell constructs and -//! supplies; the embedder fingerprint recorded in the published artifacts names the provider -//! contract, and fingerprint equality guards prior-generation reuse. An offline run embeds out of -//! the dump's own embedding stream under the fingerprint the dump recorded, so the published -//! artifacts name the provider whose vectors they carry either way. -//! -//! Nothing here is API for consumers of the crate; the module exists for the -//! [`cli`](crate::cli) operator commands, which re-export its vocabulary. +//! A live run embeds through its supplied external provider, while an offline run looks up +//! embeddings in the dump's stream under its recorded fingerprint. In either case the fingerprint +//! declares the embedding contract and guards prior-generation reuse. It does not verify which +//! provider produced the vectors. use core::num::NonZero; use std::io; @@ -61,10 +55,9 @@ const DEFAULT_ANCHORS: NonZero = const { NonZero::new(1_024).unwrap() }; /// The default comparison sample of the admission probe. const DEFAULT_COMPARISONS: NonZero = const { NonZero::new(4_096).unwrap() }; -/// The relation classifier's supply, the one input every run names. +/// A supplied classifier model or annotation corpus for a generation run. /// -/// A run fits the classifier from an annotation corpus or adopts an already-fitted artifact; the -/// variant carries the document's path. +/// Select an annotation corpus to fit a classifier in-run, or an artifact to reuse a fitted model. #[derive(Debug, Clone, PartialEq, Eq)] pub enum ClassifierSource { /// Fit the classifier in-run from the annotation-corpus document at the path. @@ -76,9 +69,10 @@ pub enum ClassifierSource { Artifact(Utf8PathBuf), } -/// How one run places rows on the map. +/// Map placement by landmark assignment or a trained projector. /// -/// Each variant carries exactly the controls its placer consumes. +/// The default [`Options::placement`] selects the projector with its reference schedule and +/// relation attraction enabled. #[derive(Debug, Copy, Clone, PartialEq, Eq)] pub enum Placement { /// Place at the landmark baseline: the fallback placer, without a training stage. @@ -87,13 +81,15 @@ pub enum Placement { Projector { /// Override the trained placement's step count. /// - /// Keeps the ratified options and the midpoint boundary. Absent, the configuration - /// default trains. + /// A supplied count uses the reference projector settings with the phase boundary at + /// `floor(steps / 2)`. This is [`None`] by default, retaining the reference 20,000-step + /// schedule with its boundary at step 5,000. steps: Option>, - /// Withhold the relation evidence from the trained placement. + /// Disable relation attraction in the trained placement. /// - /// Every other objective term trains, and the run needs no reviewed verdicts. For corpora - /// without reviewed-Proximal coverage that still want the full trained placement. + /// This is `false` by default. Enabling it supplies an empty attraction index while + /// retaining semantic, protection and landmark-support inputs. It permits trained + /// placement without reviewed Proximal pairs. vacuous: bool, }, } @@ -101,45 +97,44 @@ pub enum Placement { /// Options of one production run. #[derive(Debug, Clone)] pub struct Options

{ - /// The fit seed; equal seeds replay every draw of the run, the admission probe's included. + /// Fit seed, also used to derive the admission probe's generator. + /// + /// This is `0` by default. Repeating a draw sequence also requires equal sampling inputs and algorithms. pub seed: u64 = 0, - /// The landmark capacity `M`. + /// Maximum landmark count `M`, `4,096` by default. pub landmarks: NonZero = DEFAULT_LANDMARKS, /// Run without a prior even when the root holds an active generation. + /// + /// This is `false` by default. pub fresh: bool = false, - /// Sampled anchor rows of the admission probe. + /// Sampled anchor rows of the admission probe, `1,024` by default. pub anchors: NonZero = DEFAULT_ANCHORS, - /// Sampled comparison rows of the admission probe. + /// Sampled comparison rows of the admission probe, `4,096` by default. pub comparisons: NonZero = DEFAULT_COMPARISONS, /// Path of a reviewed-verdicts document to supply to the run. /// - /// The trained placement's phase boundary freezes its Proximal radius from the reviewed pairs, - /// so a corpus whose relations carry Proximal force needs one to train. + /// This is [`None`] by default. At the trained placement's phase boundary, a non-vacuous Proximal attraction requires reviewed pairs to establish its radius. pub verdicts: Option = None, /// Path of a quality-thresholds document overriding the source defaults. /// - /// The optional fields are `minimum_recall`, `minimum_trustworthiness`, `minimum_continuity`, - /// `maximum_intrusion_rate`, `maximum_density_spread`, and `minimum_triplet_agreement`. A - /// present field overrides its default after domain validation, an absent field keeps it, and - /// an unknown field refuses the document. The source defaults are maximally permissive, gating - /// evidence presence rather than fidelity. + /// This is [`None`] by default. The optional fields are `minimum_recall`, `minimum_trustworthiness`, `minimum_continuity`, `maximum_intrusion_rate`, `maximum_density_spread`, and `minimum_triplet_agreement`. A present field overrides its default after domain validation, an absent field keeps it, and an unknown field refuses the document. The source defaults are maximally permissive: admission requires evidence without imposing a measured fidelity threshold. pub quality_thresholds: Option = None, /// The relation classifier's supply. pub classifier: ClassifierSource, - /// How the run places rows on the map. + /// Placement strategy, [`Placement::Projector`] with no overrides by default. pub placement: Placement = Placement::Projector { steps: None, vacuous: false, }, /// Construct the k-NN lists by NN-Descent instead of the HNSW backend. /// - /// Either construction answers to the same recall admission. + /// This is `false` by default. Either construction answers to the same k-NN recall spot check. pub nn_descent: bool = false, /// The observer the run reports its progress to. pub progress: P, } -/// Plain-number summary of one production run. +/// Generation identity, fit statistics and admission evidence from one run. #[derive(Debug, Clone)] pub struct Summary { /// The published generation's identity, in directory-name form. @@ -150,14 +145,14 @@ pub struct Summary { pub edges: u64, /// The neighbour backend's recall evidence, admission reading included. /// - /// A published generation carries either an admitted reading or an unresolved one; the - /// difference is what the sample demonstrated, not whether the probe measured a number. + /// This is the fit's k-NN spot check, distinct from the map-quality report's recall control. + /// An unresolved admission interval still records a measured point estimate. pub recall: RecallSpotCheck, /// Unique card texts copied from the prior generation. pub reused: usize, - /// Unique card texts submitted to the provider. + /// Unique card texts supplied to the embedder rather than copied from the prior. pub embedded: usize, - /// Whether the admission report's gates held. + /// Whether every admission control had evidence within its bound. pub passes: bool, /// Whether the run activated the generation. pub activated: bool, @@ -165,7 +160,7 @@ pub struct Summary { pub report: String, } -/// The refusal grounds of a supplied quality-thresholds document. +/// Failure to read or validate a quality-thresholds override document. #[derive(Debug)] pub enum ThresholdSupplyError { /// The run could not read the document. @@ -225,9 +220,9 @@ pub enum RunError { Annotations(AnnotationSupplyError), /// The run refused the supplied classifier artifact. Classifier(ClassifierSupplyError), - /// The live run could not reach a verdict. + /// The live generation run did not complete successfully. Run(RunnerError), - /// The offline run could not reach a verdict. + /// The offline generation run did not complete successfully. OfflineRun(RunnerError), } @@ -274,8 +269,8 @@ impl core::error::Error for RunError { /// /// # Errors /// -/// Returns [`RunError::Annotations`] or [`RunError::Classifier`] when the run refuses the named -/// document. +/// Returns [`RunError`] when the selected annotation corpus or classifier artifact cannot be +/// admitted. fn classifier_input(source: &ClassifierSource) -> Result { match source { ClassifierSource::Annotations(path) => Ok(ClassifierInput::Annotations( @@ -291,8 +286,7 @@ fn classifier_input(source: &ClassifierSource) -> Result, @@ -310,8 +304,10 @@ fn quality_thresholds( /// Resolves the run's placement options over the configuration default. /// -/// A step-count override rebuilds the ratified options around the shortened schedule; otherwise -/// the projector controls apply to the configuration default's options. +/// A step-count override starts from [`ProjectorOptions::live`] and replaces its schedule with +/// [`TrainingSchedule::shortened`]. Without that override, an initial projector configuration keeps +/// its settings. An initial baseline uses the reference projector settings. Both projector paths +/// apply the requested `vacuous` flag. fn placement_options(placement: Placement, initial: PlacementOptions) -> PlacementOptions { let Placement::Projector { steps, vacuous } = placement else { return PlacementOptions::LandmarkBaseline; @@ -332,13 +328,12 @@ fn placement_options(placement: Placement, initial: PlacementOptions) -> Placeme PlacementOptions::Projector(projector) } -/// The dataset-independent half of one run, resolved from its options. +/// Runner settings and admitted documents independent of dataset contents. /// -/// Everything here is decided by the operator options, the pinned device, and the documents the -/// options name, before any dataset exists, so the live and offline entry points resolve it -/// identically. +/// Resolution uses operator options, the pinned device and supplied documents without reading +/// dataset rows. The live and offline paths share this resolution. struct ResolvedRun { - /// The runner options the entry point hands to the run. + /// Fit, probe, prior and device settings. runner: RunnerOptions, /// The admitted reviewed-verdicts document, when one was supplied. verdicts: Option, @@ -350,8 +345,8 @@ struct ResolvedRun { /// /// # Errors /// -/// Returns a [`RunError`] naming the refused document: the supplied quality-thresholds, -/// verdicts, annotation-corpus, or classifier document, in that order. +/// Returns [`RunError`] when a supplied document cannot be admitted. Quality thresholds resolve +/// first, then verdicts, then the selected annotation corpus or classifier artifact. fn resolve

(options: &Options

, device: PinnedDevice) -> Result { let mut runner_options = RunnerOptions { fit: FitConfig { diff --git a/libs/@local/graph/atlas/src/salt/runner/operator/offline.rs b/libs/@local/graph/atlas/src/salt/runner/operator/offline.rs index 3029b6884cf..34ce0cfa545 100644 --- a/libs/@local/graph/atlas/src/salt/runner/operator/offline.rs +++ b/libs/@local/graph/atlas/src/salt/runner/operator/offline.rs @@ -10,13 +10,13 @@ use crate::{ /// Runs one production generation over the dump directory at `dump`. /// -/// The dump carries the snapshot, its temporal axes, and every embedding the run requests, so the -/// run reaches neither the store nor the embedding provider. The generation publishes under the -/// generation root at `root` exactly as a live run's does, and equal dumps under equal options -/// describe the same run. +/// The dump supplies the snapshot and its temporal axes. All embedding requests resolve locally, +/// and the generation publishes under `root`. The dump must cover the requested canonical sample +/// and every requested card text, including texts from a supplied annotation corpus. A missing +/// embedding fails the run instead of making a provider request. /// -/// The supplied documents resolve before the dump opens, because admitting them costs file reads -/// while opening the dump hashes every stream, so each step fails ahead of everything costlier. +/// Resolving supplied documents before opening the dump rejects invalid supplies without hashing +/// the dump's streams. /// /// # Errors /// diff --git a/libs/@local/graph/atlas/src/salt/runner/tests.rs b/libs/@local/graph/atlas/src/salt/runner/tests.rs index 77e4454ef6a..3f96c58de24 100644 --- a/libs/@local/graph/atlas/src/salt/runner/tests.rs +++ b/libs/@local/graph/atlas/src/salt/runner/tests.rs @@ -35,8 +35,14 @@ use crate::{ }, }; +/// Row count of the runner fixture corpus. const NODES: usize = 48; +/// Returns a per-process scratch path after attempting to remove its previous directory. +/// +/// # Panics +/// +/// This panics if the system temporary directory's path is not UTF-8. fn scratch(name: &str) -> Utf8PathBuf { let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) .expect("the temp directory is UTF-8") @@ -48,10 +54,11 @@ fn scratch(name: &str) -> Utf8PathBuf { dir } -/// A probe-scale corpus for the real fit. +/// Builds a small corpus with seeded representations and typed rows. /// -/// Unit-norm pseudo-random representations whose canonical embeddings extend them with zeros, one -/// node type alternating between two ontology rows, and one link type. +/// Canonical embeddings zero-extend the normalized representations. Normalization computes in +/// double precision before rounding the components to `f32`. Nodes alternate between two direct +/// ontology types, and the link uses a third type. fn dataset() -> MemoryDataset { let mut rng = Xoshiro256PlusPlus::seed_from_u64(0x27A); let mut canonical = HashMap::new(); @@ -159,12 +166,16 @@ impl CardEmbedder for HashEmbedder { } } -/// A deterministic classifier fitted from a synthetic corpus. +/// Fits the fixture's supplied classifier from a fixed synthetic corpus. /// -/// The supplied model input of the fixture runs. +/// # Panics +/// +/// This panics if the synthetic training fixture fails validation or fitting. fn classifier() -> ClassifierInput { const ROWS: usize = 4; - // Coprime to the dimension, so no two corpus rows repeat. + // A period coprime to the row width visits every pattern offset before repeating. This + // 13-element pattern spans four 3,072-component rows. Therefore each row starts at a distinct + // pattern offset and has a distinct embedding. const PATTERN: [f32; 13] = [ -0.75, -0.625, -0.5, -0.375, -0.25, -0.125, 0.0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, ]; @@ -212,7 +223,7 @@ fn classifier() -> ClassifierInput { } } -/// Fixture-sized runner options over the given thresholds. +/// Configures a small landmark-baseline run with the given seed and thresholds. fn options(seed: u64, thresholds: QualityThresholds) -> RunnerOptions { RunnerOptions { fit: FitConfig { @@ -249,15 +260,18 @@ fn options(seed: u64, thresholds: QualityThresholds) -> RunnerOptions { } } -/// An observer keeping every admission reading the battery reported, in arrival order. +/// A shared log of admission readings in reporting order. /// -/// Cloneable and shareable because a detached half records into the same log: the readings arrive -/// exactly as the run reported them. +/// Detached observers append to the same log. Reading or appending panics if its mutex is poisoned. #[derive(Debug, Clone, Default)] struct RecordingBattery(Arc>>); impl RecordingBattery { - /// Every reading so far, in arrival order. + /// Copies the recorded readings in reporting order. + /// + /// # Panics + /// + /// This panics if the log's mutex is poisoned. fn readings(&self) -> Vec<(QualityMetric, f64)> { self.0 .lock() @@ -267,7 +281,6 @@ impl RecordingBattery { } impl Progress for RecordingBattery { - /// Both halves share the log, so a detached half records into the same fixture. type Detached = Self; fn detach(&self) -> Self { @@ -282,7 +295,6 @@ impl Progress for RecordingBattery { } } -/// A run whose report passes activates what it publishes. #[tokio::test] async fn passing_run_activates_the_generation() { let root = GenerationRoot::new(scratch("activates")).expect("the root should open"); @@ -304,9 +316,6 @@ async fn passing_run_activates_the_generation() { .expect("the run should reach a verdict"); assert_eq!(outcome.admission, Admission::Active); - // The battery reports the readings its own verdict turns on, one per control. They are the same - // numbers the report reduces, so an observer and the verdict can never disagree about the - // measurement. assert_eq!( outcome .report @@ -341,16 +350,13 @@ async fn passing_run_activates_the_generation() { ); } -/// A run whose report refuses admission publishes a candidate and leaves the pointer alone. #[tokio::test] async fn refused_run_leaves_a_candidate() { let root = GenerationRoot::new(scratch("candidate")).expect("the root should open"); let dataset = dataset(); let classifier = classifier(); - // A 2D projection of 48 pseudo-random unit vectors cannot carry - // near-perfect neighbourhoods: the floor refuses admission on a - // real reading, not on a rigged fixture. + // raise the recall floor to exercise refusal on this fixture's measured neighbourhood loss. let outcome = run( &dataset, &HashEmbedder, @@ -382,7 +388,6 @@ async fn refused_run_leaves_a_candidate() { ); } -/// The second run reuses the active generation as its prior, while a fresh run ignores it. #[tokio::test] async fn prior_modes_route_reuse() { let root = GenerationRoot::new(scratch("prior")).expect("the root should open"); @@ -451,10 +456,6 @@ async fn prior_modes_route_reuse() { ); } -/// Witnesses the replay half of [`probe_rng`]'s contract. -/// -/// Equal seeds deriving equal draws is what lets a whole run replay from the one fit seed; the -/// inequality is the complement that a constant generator would otherwise satisfy. #[test] fn the_admission_probe_derives_its_draws_from_the_fit_seed() { let draws = |seed| { diff --git a/libs/@local/graph/atlas/src/salt/semantic/artifact.rs b/libs/@local/graph/atlas/src/salt/semantic/artifact.rs index 867c19a30a3..1bdda2ca9c2 100644 --- a/libs/@local/graph/atlas/src/salt/semantic/artifact.rs +++ b/libs/@local/graph/atlas/src/salt/semantic/artifact.rs @@ -1,9 +1,9 @@ -//! The semantic graph's published form as one sparse matrix file and its mapped reader. +//! Semantic graph publication and mapped access to fixed attraction weights. //! -//! A [`SemanticGraph`] publishes as one [`crate::file::sprs`] file holding its -//! [`SemanticMatrix`](super::SemanticMatrix) verbatim. [`SemanticGraphArchive`] reopens the file -//! over a whole-file mapping and validates the graph invariants once, so training and release -//! evaluation read the same weights from the page cache without holding them on the heap. +//! A [`SemanticGraph`] with zero-based row pointers publishes as one [`crate::file::sprs`] file +//! preserving its [`SemanticMatrix`](super::SemanticMatrix) entries. [`SemanticGraphArchive`] +//! validates the graph invariants over the mapped matrix. Reopening the same artifact fixes the +//! semantic weights without copying the matrix regions to heap allocations. #![cfg_attr( not(test), expect( @@ -39,12 +39,14 @@ where /// Writes the graph as a sparse matrix file. /// - /// Returns the SHA-256 of the written bytes: the identity the repository records for the - /// published file. - /// /// # Errors /// /// Returns an error when the underlying writer fails. + /// + /// # Panics + /// + /// This panics when the matrix's first row pointer is nonzero. [`SemanticGraph::new`] accepts + /// such matrices, but the sparse-file writer requires zero-based pointers. fn write_into(&self, write: impl io::Write) -> io::Result { let mut writer = Writer { accumulator: Sha256::new(), @@ -53,8 +55,8 @@ where write_matrix(&self.matrix(), &mut writer).map_err(|error| match error { WriteSprsError::Io(error) => error, - // A validated graph is row-compressed, unsliced, and at - // least 2 x 2, so no non-IO write failure exists for it. + // validation establishes nonzero dimensions. Zero-based pointers are an additional + // writer requirement, established by SemanticGraph::build but not SemanticGraph::new. error @ (WriteSprsError::Sliced | WriteSprsError::ZeroDimension { .. }) => { unreachable!("a validated graph is writable: {error}") } @@ -66,7 +68,7 @@ where impl WriteAs for SemanticGraph where N: Id {} -/// An opened sparse matrix file does not hold a valid semantic graph. +/// Failure to interpret a sparse matrix file as a semantic graph. #[derive(Debug)] pub(crate) enum InvalidSemanticFile { /// The file does not hold the graph's matrix layout. @@ -106,10 +108,11 @@ impl Error for InvalidSemanticFile { /// A published semantic graph opened over its mapped file. /// -/// Construction checks the graph invariants once, so an open graph only serves valid views; the -/// matrix regions stay in the page cache under memory pressure and off the heap. Each -/// [`view`](Self::view) re-checks the compressed-row structure ([`SprsFile::matrix`]'s contract), -/// so stages call it once and hold the view. +/// Construction checks the graph invariants once. Each [`Self::view`] rechecks the sparse structure +/// through [`SprsFile::matrix`] and borrows the mapped entries. Retain a view across repeated reads +/// to avoid repeating that structural scan. +/// +/// Validated views require the backing file to remain immutable for the mapping's lifetime. #[derive(Debug)] pub(crate) struct SemanticGraphArchive { file: SprsFile, @@ -124,8 +127,8 @@ where /// /// # Errors /// - /// Returns an error when the file does not hold the graph's matrix layout or the matrix - /// violates a [`SemanticGraph`] invariant. + /// Returns [`InvalidSemanticFile`] when the file cannot provide a matrix satisfying the graph's + /// layout and invariants. pub(crate) fn new(file: SprsFile) -> Result { let matrix = file.matrix().map_err(InvalidSemanticFile::Matrix)?; validate(matrix)?; @@ -136,7 +139,11 @@ where }) } - /// Borrows the validated graph. + /// Borrows the validated graph after rechecking its sparse structure. + /// + /// # Complexity + /// + /// This takes O(n + m) work for `n` rows and `m` stored entries. #[must_use] pub(crate) fn view(&self) -> SemanticGraphView<'_, N> { let matrix = self diff --git a/libs/@local/graph/atlas/src/salt/semantic/bandwidth.rs b/libs/@local/graph/atlas/src/salt/semantic/bandwidth.rs index f549e739802..82998dfbf4c 100644 --- a/libs/@local/graph/atlas/src/salt/semantic/bandwidth.rs +++ b/libs/@local/graph/atlas/src/salt/semantic/bandwidth.rs @@ -1,28 +1,37 @@ //! Per-row smooth-kNN bandwidth calibration. //! -//! Each node row receives a local connectivity radius `ρ` and a bandwidth `σ` turning its neighbour -//! distances `d_j` into fuzzy memberships +//! Each node row receives a local connectivity radius `ρ` and a bandwidth `σ` to normalize its +//! distance scale into fuzzy memberships. For `k` neighbours with finite distances `d_j ≥ 0`, `ρ` +//! is the smallest positive distance, or zero when every distance is zero. With finite `σ > 0`, the +//! mathematical model is //! //! ```text -//! p_j = exp(-max(d_j - ρ, 0) / σ). +//! p_j = exp(-max(d_j - ρ, 0) / σ), +//! Σ_j p_j = target, target = log₂(k). //! ``` //! -//! `ρ` is the smallest positive distance in the row, so the nearest distinct neighbour always holds -//! full membership. `σ` solves +//! The nearest positive-distance neighbour and every zero-distance neighbour hold full membership. +//! Matching the same sum across rows makes dense and sparse regions comparable through their local +//! distance scales. //! -//! ```text -//! Σ_j p_j = target -//! ``` +//! Calibration starts at `σ = 1`, expanding the upper bound by doubling as needed and bisecting +//! once it has a bracket. The number of neighbours at `d_j ≤ ρ` bounds the achievable sum from +//! below. When that count exceeds `target`, the equation has no solution. A row of exact duplicates +//! is the extreme case: its sum is `k` for every positive `σ`. On such rows the search lowers `σ` +//! while the residual is at least the tolerance, subject to the iteration limit. For a finite +//! positive final bandwidth, the sum remains at least as large as the tie count. Other neighbours +//! can still contribute above it. +//! +//! The bandwidth floor is a multiple of the row's mean distance, using the corpus mean when every +//! distance is zero. Applying the floor after bisection can raise the sum above the target. A zero +//! scale supplies no positive floor. The default iteration limit keeps the trial bandwidth positive +//! even in that case. //! -//! by bisection, `target` being `log2(k)` for a `k`-neighbour table, which is what makes dense and -//! sparse regions comparable. Neighbours at `d_j ≤ ρ` hold full membership at every `σ`, so their -//! count bounds the achievable sum from below. A row where more than `target` distances tie at or -//! below `ρ` has no solution (a row of exact duplicates, whose sum is `k` for every `σ`, is the -//! extreme case). On such rows the bisection drives `σ` toward zero, the floor takes over, and the -//! sum settles at the tie count above the target. Membership sums accumulate in double precision, -//! so accumulation noise stays well below the bisection tolerance. The floor is proportional to the -//! row's mean distance (the corpus mean when every distance ties at zero) and keeps `σ` positive -//! everywhere. +//! Adjusted distances, bandwidths and exponential evaluations use `f32`. Membership sums accumulate +//! in `f64`, reducing summation error without removing the kernel's approximation error or +//! guaranteeing an arbitrary tolerance. Stored memberships additionally clamp to +//! [`f32::MIN_POSITIVE`] to retain every directed edge, whereas the bisection sum uses unclamped +//! kernel results. use core::simd::{f32x8, f64x8, num::SimdFloat as _}; use std::simd::Simd; @@ -30,20 +39,23 @@ use std::simd::Simd; use super::SmoothingOptions; use crate::math::{MatrixN, NonNegative, kernel::exp_f32x8}; +/// Distances per SIMD kernel evaluation. const LANES: usize = 8; -/// One row's calibrated radius and bandwidth. +/// One row's calibrated local distance scale. #[derive(Debug, Copy, Clone, PartialEq)] pub(super) struct Bandwidth { + /// Smallest positive neighbour distance, or zero if every distance is zero. pub rho: f32, + /// Trial bandwidth after applying the configured distance-based floor. pub sigma: f32, } -/// Reusable per-row state: adjusted distances, padded for the kernel. +/// Reusable calibration state for rows of one fixed neighbour count. /// -/// One solver serves many rows; construction sizes the scratch once and -/// [`calibrate`](Self::calibrate) refills it per row. Padding lanes hold positive infinity, which -/// the membership kernel maps to an exact zero, so partial trailing lanes need no masking. +/// Construction sizes the scratch once, and [`Self::calibrate`] refills it per row. Padding lanes +/// hold positive infinity. At finite positive bandwidths their kernel inputs are negative infinity, +/// which contributes exactly zero to the bisection sum. Output copies only the real lanes. pub(super) struct RowSolver { adjusted: MatrixN<8>, } @@ -58,8 +70,13 @@ impl RowSolver { /// Calibrates one row's bandwidth against its neighbour distances. /// - /// `target` is the membership sum to solve for and `fallback_scale` replaces the row's mean - /// distance in the `σ` floor when the row has no positive distance to measure a scale from. + /// `distances` must have the neighbour count supplied to [`Self::new`]. `target` is the finite + /// membership sum to approach, normally `log₂(k)`. `fallback_scale` must be finite and + /// nonnegative. It replaces the row's mean distance in the `σ` floor when every distance is + /// zero. [`SmoothingOptions`] describes the convergence limits. + /// + /// The returned bandwidth can leave a residual larger than the tolerance, including when the + /// tie count makes the target unattainable. pub(super) fn calibrate( &mut self, distances: &[NonNegative], @@ -70,7 +87,7 @@ impl RowSolver { const INF: Simd = Simd::splat(f32::INFINITY); const ZERO: Simd = Simd::splat(0.0); - // `NonNegative` is `repr(transparent)` over `f32` + // `NonNegative` has the same representation as `f32`. let distances: &[f32] = zerocopy::transmute_ref!(distances); let rho = distances @@ -81,7 +98,8 @@ impl RowSolver { let rho = if rho.is_finite() { rho } else { 0.0 }; let rho_x8 = Simd::splat(rho); - // Adjusted distances: max(d - ρ, 0) per lane, padding from the load's infinity fill. + // subtracting the local radius makes every neighbour at or below it a full member. + // infinity padding contributes zero to the sum at finite positive bandwidths. let rows = self.adjusted.lanes_mut(); for (row, distance) in rows.iter_mut().zip(distances.chunks(LANES)) { *row = (Simd::load_or(distance, INF) - rho_x8).simd_max(ZERO); @@ -120,8 +138,10 @@ impl RowSolver { /// Writes the row's memberships under `bandwidth` into `out`. /// - /// This clamps every membership to at least [`f32::MIN_POSITIVE`], so a stored edge never - /// carries an exact zero. + /// `bandwidth` must describe the last row passed to [`Self::calibrate`], and `out` must have + /// that row's length. Only `bandwidth.sigma` participates here: calibration already subtracted + /// the radius from the stored distances. Every output membership clamps to at least + /// [`f32::MIN_POSITIVE`], and a stored edge never carries an exact zero. pub(super) fn memberships(&self, bandwidth: Bandwidth, out: &mut [f32]) { let sigma = f32x8::splat(bandwidth.sigma); let floor = f32x8::splat(f32::MIN_POSITIVE); @@ -132,7 +152,9 @@ impl RowSolver { } } - /// Sums the row's memberships under a candidate `sigma`, accumulated in double precision. + /// Sums the last row's unclamped kernel memberships in double precision. + /// + /// `sigma` must be finite and positive for padding to contribute exactly zero. fn membership_sum(&self, sigma: f32) -> f64 { let sigma = f32x8::splat(sigma); let mut sum = f64x8::splat(0.0); @@ -145,7 +167,9 @@ impl RowSolver { } } -/// Returns the arithmetic mean of `values`. +/// Computes the arithmetic mean in single precision. +/// +/// `values` must be nonempty for a defined mean. Both the sum and the length conversion can round. #[expect( clippy::cast_precision_loss, reason = "neighbour counts stay far below exact f32 integer precision" diff --git a/libs/@local/graph/atlas/src/salt/semantic/error.rs b/libs/@local/graph/atlas/src/salt/semantic/error.rs index fc92a22e970..620df9d99c9 100644 --- a/libs/@local/graph/atlas/src/salt/semantic/error.rs +++ b/libs/@local/graph/atlas/src/salt/semantic/error.rs @@ -1,8 +1,6 @@ -//! Semantic-graph validation errors. - use core::{error::Error, fmt}; -/// A matrix violated a [`SemanticGraph`](super::SemanticGraph) invariant. +/// A violation of a [`SemanticGraph`](super::SemanticGraph) invariant. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) enum SemanticValidationError { /// The matrix uses column compression. diff --git a/libs/@local/graph/atlas/src/salt/semantic/mod.rs b/libs/@local/graph/atlas/src/salt/semantic/mod.rs index 66134d5e1fe..c112c97b140 100644 --- a/libs/@local/graph/atlas/src/salt/semantic/mod.rs +++ b/libs/@local/graph/atlas/src/salt/semantic/mod.rs @@ -1,24 +1,25 @@ //! The semantic graph of fuzzy edge weights over the k-NN table. //! -//! [`SemanticGraph`] is a symmetric sparse matrix over the node-row domain whose entry `(i, j)` -//! weights the semantic edge between rows `i` and `j` in `(0, 1]`. It is the weighted form of the -//! [`Knn`](super::knn::table::Knn) table: distances calibrate into directed fuzzy memberships per -//! row ([`bandwidth`]), and the directed memberships combine into one undirected weight by the -//! probabilistic union +//! [`SemanticGraph`] is a symmetric sparse matrix over the node-row domain. Each stored entry `(i, +//! j)` weights the semantic edge between distinct rows `i` and `j` in `(0, 1]`. +//! [`SemanticGraph::build`] calibrates the distances in a [`Knn`](super::knn::table::Knn) table +//! into directed fuzzy memberships per row ([`bandwidth`]). The directed memberships combine into +//! an undirected weight by the probabilistic union. In mathematical notation, //! //! ```text -//! w(i, j) = p(i → j) + p(j → i) - p(i → j) · p(j → i), +//! w(i, j) = p(i → j) + p(j → i) - p(i → j) · p(j → i). //! ``` //! -//! an absent direction contributing zero, so a one-sided edge keeps its directed membership. The -//! union's support is the union of the directed supports. A row carries its `k` outgoing edges plus -//! one edge for every other row that names it, so degrees start at `k` and only the row domain -//! bounds them. A hub named by many rows carries many edges. The graph stores every edge in both of -//! its rows with bit-equal weight. +//! Here `p(i → j)` is the calibrated membership of neighbour `j` in row `i`. An absent direction +//! contributes zero, and a one-sided edge keeps its directed membership. The union's support is the +//! union of the directed supports. A built row carries its `k` outgoing neighbours plus each +//! incoming neighbour absent from that outgoing set. Its degree lies between `k` and `n - 1` for +//! `n` node rows. A hub named by many rows can have a high degree. The graph stores every edge in +//! both of its rows with bit-equal weight. //! -//! The graph is the training-side attraction structure, and training and release evaluation both -//! read it from its published artifact, so backend variation in the k-NN build cannot confound -//! model comparisons ([`artifact::SemanticGraphArchive`] reopens the published file). +//! Publishing these attraction weights fixes the semantic input for model comparisons that reopen +//! the same artifact through [`artifact::SemanticGraphArchive`]. Rebuilding the k-NN table can +//! change that input. use core::marker::PhantomData; @@ -40,25 +41,28 @@ mod error; #[cfg(test)] mod tests; -/// The graph's matrix layout, shared with the k-NN table. +/// A sparse `f32` weight matrix with `u32` columns and `u64` row pointers. pub(crate) type SemanticMatrix = CsMatI; /// A borrowed [`SemanticMatrix`]. pub(crate) type SemanticMatrixView<'view> = CsMatViewI<'view, f32, u32, u64>; -/// Smooth-kNN calibration settings. +/// Smooth-kNN convergence limits and the distance-scaled bandwidth floor. /// /// The defaults are the established UMAP fuzzy-set kernel constants. #[derive(Debug, Copy, Clone, PartialEq)] pub(crate) struct SmoothingOptions { - /// Absolute tolerance on the membership-sum equation at which the bisection stops early. + /// Absolute membership-sum residual below which bisection stops early. + /// + /// The tolerance is `1.0e-5` by default. Reaching the iteration limit can leave a larger residual. pub tolerance: DPositive = d_positive!(1.0e-5), - /// Scale factor of the `σ` floor. + /// Scale factor of the distance-based `σ` floor. /// - /// `σ` never falls below this fraction of the row's mean distance (the corpus mean for rows - /// without a positive distance). + /// This is `1.0e-3` by default. For a finite nonnegative factor, `σ` never falls below its product with the row's mean distance (the corpus mean for rows without a positive distance). A zero or underflowed product supplies no positive floor. pub bandwidth_floor: f32 = 1.0e-3, - /// Bisection iterations per row when the tolerance is not met earlier. + /// Maximum bisection iterations per row. + /// + /// This is `64` by default. Zero iterations keep the initial trial bandwidth of `1.0` before applying the floor. Large limits can drive the trial bandwidth to zero or infinity in `f32`. pub bisection_iterations: usize = 64, } @@ -68,7 +72,11 @@ const impl Default for SmoothingOptions { } } -/// Checks every graph invariant over a borrowed matrix. +/// Checks the graph's shape, weight domain and symmetry. +/// +/// # Errors +/// +/// Returns [`SemanticValidationError`] when the matrix violates a graph invariant. #[expect( clippy::float_cmp, reason = "the union weight is computed from commutative operations, so the two directions of \ @@ -135,9 +143,11 @@ fn validate(matrix: SemanticMatrixView<'_>) -> Result<(), SemanticValidationErro /// /// Row `i` stores the weights of every semantic edge at node row `i`, keyed by the other endpoint /// in ascending row order. Weights are finite in `(0, 1]`, no row references itself, and the graph -/// stores every edge in both of its rows with bit-equal weight. A row's edges are the union of the -/// directed supports: at least the `k` outgoing edges of a `k`-neighbour table, plus one for every -/// other row naming it. +/// stores every edge in both of its rows with bit-equal weight. The square matrix spans at least +/// two rows. [`Self::new`] also accepts empty rows and graphs with no edges. [`Self::build`] +/// establishes the k-NN support relationship described by this module. +/// +/// `N` must represent every row in the matrix's domain for typed traversal. #[derive(Debug, Clone)] pub(crate) struct SemanticGraph(SemanticMatrix, PhantomData); @@ -149,9 +159,7 @@ where /// /// # Errors /// - /// Returns an error when the matrix is not row-compressed, not square over at least two rows, - /// self-referencing, stores a weight outside the finite `(0, 1]` range, or stores an edge whose - /// two directions are missing or unequal. + /// Returns [`SemanticValidationError`] when the matrix violates a graph invariant. pub(crate) fn new(matrix: SemanticMatrix) -> Result { validate(matrix.view())?; Ok(Self(matrix, PhantomData)) @@ -159,10 +167,21 @@ where /// Weighs a k-NN table into the symmetric semantic graph. /// - /// Each row's distances calibrate a [`bandwidth`] whose exponential memberships sum to - /// `log2(k)`. The directed memberships then combine by the probabilistic union. Rows calibrate - /// in parallel and deterministically: each row writes its result into its own slot regardless - /// of completion order. + /// Calibration uses each row's distances to approach a membership sum of `log₂(k)` through a + /// [`bandwidth`]. Tied distances can make that target unattainable, and iteration limits or the + /// floors can also leave a residual. The directed memberships then combine by the probabilistic + /// union. + /// + /// Row completion order preserves the association between each membership and its edge. With + /// the default settings, each row's weights have a fixed arithmetic order for a given numerical + /// environment. Cross-target bit equality is outside this contract. + /// + /// # Complexity + /// + /// For `n` rows, `k` neighbours per row and at most `b` calibration iterations, calibration + /// takes O(n · k · (b + 1)) work. The directed matrices and union use O(n · k) additional + /// space. Final validation checks each reverse edge by binary search, adding O(m · log n) work + /// for `m` union entries. #[expect( clippy::cast_precision_loss, clippy::cast_possible_truncation, @@ -175,6 +194,8 @@ where let (_, indices, distances) = knn.matrix().into_raw_storage(); let target = (neighbours as f64).log2(); + // the parallel sum can vary its rounding order. Only all-zero rows use this fallback, + // and their memberships are one for every finite positive bandwidth. let corpus_mean = (distances .par_iter() .map(|&distance| f64::from(distance)) @@ -201,10 +222,10 @@ where let transposed = directed.transpose_view().to_csr(); - // (a + b) - a · b: both operations are commutative, so the two directions of an edge - // compute bit-equal weights. The expression is ≤ 1 over memberships in [0, 1] - at the - // maximum the two roundings leave a half-ulp that ties-to-even returns to 1.0 - and the - // clamp is the unit fraction's ceiling. + // Swapping finite memberships preserves both their rounded sum and their exact product. + // The fused operation subtracts that product from the same rounded sum in either order. + // Therefore both directions compute bit-equal weights. The final clamp enforces the + // unit-fraction ceiling at 1.0. let union = csmat_binop(directed.view(), transposed.view(), |&lhs, &rhs| { lhs.mul_add(-rhs, lhs + rhs).min(1.0) }); @@ -244,10 +265,10 @@ impl<'view, N> SemanticGraphView<'view, N> where N: Id, { - /// Wraps a matrix whose invariants already hold. + /// Borrows a matrix satisfying the graph invariants. /// - /// The caller promises the matrix passed [`validate`]; the wrapper performs no checks of its - /// own. + /// `matrix` must satisfy [`validate`]. + // These invariants govern graph correctness, not memory safety. #[inline] #[must_use] pub(super) const fn new_unchecked(matrix: SemanticMatrixView<'view>) -> Self { @@ -272,7 +293,8 @@ where /// /// # Panics /// - /// This panics when `row` is outside the graph's row domain. + /// This panics when [`Id::as_usize`] maps `row` outside the matrix's row domain. Iteration + /// panics if `N` cannot represent a stored endpoint. pub(crate) fn row(&self, row: N) -> impl Iterator> + '_ { let (columns, weights) = self .0 @@ -292,7 +314,12 @@ where /// Sums the graph's positive edge weight in double precision. /// - /// Every stored entry contributes, so each undirected edge counts once per endpoint row. + /// Every stored entry contributes in row and endpoint order, and each undirected edge counts + /// once per endpoint row. + /// + /// # Panics + /// + /// This panics if `N` cannot represent a matrix row. #[must_use] pub(crate) fn total_weight(&self) -> DNonNegative { let mut total = DNonNegative::ZERO; diff --git a/libs/@local/graph/atlas/src/salt/semantic/tests.rs b/libs/@local/graph/atlas/src/salt/semantic/tests.rs index c597789ca04..38d2e511c3d 100644 --- a/libs/@local/graph/atlas/src/salt/semantic/tests.rs +++ b/libs/@local/graph/atlas/src/salt/semantic/tests.rs @@ -21,11 +21,17 @@ use crate::{ salt::knn::table::{Knn, KnnMatrix}, }; +/// Returns the default calibration settings for fixtures. fn options() -> SmoothingOptions { SmoothingOptions { .. } } -/// Builds a validated k-NN table from uniform per-row neighbour lists. +/// Builds a k-NN fixture from uniform, ascending-column neighbour lists. +/// +/// # Panics +/// +/// This panics if `rows` is empty, the lists violate the sparse structure or k-NN invariants, or +/// the matrix size exceeds its index encoding. fn knn_from_rows(rows: &[Vec<(u32, NonNegative)>]) -> Knn { let count = rows.len(); let neighbours = rows[0].len(); @@ -46,7 +52,15 @@ fn knn_from_rows(rows: &[Vec<(u32, NonNegative)>]) -> Knn { Knn::new(matrix).expect("the fixture satisfies the table invariants") } -/// Brute-force cosine k-NN over random points on the unit circle arc. +/// Builds brute-force cosine k-NN over seeded random points on a quarter-circle arc. +/// +/// Each row retains at most `neighbours` entries, sorted by column after nearest-distance +/// selection. +/// +/// # Panics +/// +/// This panics if `rows` is below two, `neighbours` is zero, or the row count exceeds the matrix's +/// `u32` dimension encoding. fn random_knn(rows: usize, neighbours: usize, seed: u64) -> Knn { let mut rng = Xoshiro256PlusPlus::seed_from_u64(seed); let points: Vec<[f32; 2]> = core::iter::repeat_with(|| { @@ -81,9 +95,14 @@ fn random_knn(rows: usize, neighbours: usize, seed: u64) -> Knn { knn_from_rows(&table) } -/// The scalar fuzzy-weight reference. +/// Computes a scalar smooth-kNN reference with libm exponentials. +/// +/// The result maps each direction of every union edge to its weight. The scalar reference evaluates +/// the union without fusion, independently of the SIMD path. /// -/// The smooth-kNN kernel with libm exponentials, keyed by directed edge. +/// # Panics +/// +/// This panics if `knn` has a nonzero row-pointer offset. #[expect( clippy::suboptimal_flops, clippy::cast_precision_loss, @@ -208,7 +227,7 @@ fn calibration_solves_the_membership_sum_equation() { let mut solver = RowSolver::new(distances.len()); let bandwidth = solver.calibrate(&distances, target, 1.0, &options()); - // The defining equation, recomputed through scalar libm. + // scalar libm supplies an independent evaluation of the membership equation. let sum: f64 = distances .iter() .map(|&distance| { @@ -242,7 +261,7 @@ fn duplicate_rows_saturate_at_full_membership() { let mut solver = RowSolver::new(distances.len()); let bandwidth = solver.calibrate(&distances, 2.0, 0.25, &options()); - // No positive distance: the floor scales from the fallback. + // without a positive distance, the floor scales from the fallback. assert_eq!(bandwidth.rho, 0.0); assert_eq!(bandwidth.sigma, options().bandwidth_floor * 0.25); @@ -251,17 +270,14 @@ fn duplicate_rows_saturate_at_full_membership() { assert_eq!(memberships, [1.0; 4]); } -/// The SIMD row pipeline matches the membership definition lane for lane. -/// -/// The scalar oracle routes through the same vendored kernel, so agreement is bit-exact: the pin -/// covers the padded tail lanes (partial, exact, and full-plus-partial chunks), the ρ subtraction -/// with its zero clamp, the membership floor, and same-length solver reuse. #[test] #[expect( clippy::cast_precision_loss, reason = "the fixture lengths are far below exact f64 integer precision" )] fn simd_rows_match_the_scalar_definition_lane_for_lane() { + // partial and complete SIMD chunks exercise padding and the copied output tail. + // the scalar oracle uses the same kernel to isolate row preparation from kernel accuracy. for neighbours in [1_usize, 3, 7, 8, 9, 15, 16, 17] { let distances: Vec = (0..neighbours) .map(|slot| { @@ -276,8 +292,7 @@ fn simd_rows_match_the_scalar_definition_lane_for_lane() { let target = (neighbours as f64).log2().max(1.0); let mut solver = RowSolver::new(neighbours); - // A first calibration over reversed distances, so the checked row - // reuses scratch another row has written. + // reversing the distances checks that same-length reuse replaces the prior row's scratch. let reversed: Vec = distances.iter().rev().copied().collect(); solver.calibrate(&reversed, target, 0.5, &options()); @@ -362,9 +377,8 @@ fn union_support_covers_every_directed_edge() { #[test] fn one_sided_edges_keep_their_directed_membership() { - // Row 2 is nobody's neighbour: both of its edges are one-sided, so - // their union weights equal its directed memberships, which the - // scalar reference computes independently. + // row 2 is absent from every neighbour list. Its one-sided edges retain their directed + // memberships, which the scalar reference computes independently. let knn = knn_from_rows(&[ vec![(1, non_negative!(0.1)), (3, non_negative!(0.2))], vec![(0, non_negative!(0.1)), (3, non_negative!(0.3))], @@ -431,6 +445,7 @@ fn published_graph_reopens_mapped() { assert_eq!(digest, hasher.finalize()); } +/// Builds a two-row matrix with one edge stored at the given directional weights. fn symmetric_pair(weight_forward: f32, weight_reverse: f32) -> SemanticMatrix { SemanticMatrix::try_new( (2, 2), diff --git a/libs/@local/hashql/core/src/id/slice.rs b/libs/@local/hashql/core/src/id/slice.rs index 1936fa52e6c..8d27a9aff1e 100644 --- a/libs/@local/hashql/core/src/id/slice.rs +++ b/libs/@local/hashql/core/src/id/slice.rs @@ -39,12 +39,11 @@ const fn greatest_chunk_start(length: usize, size: NonZero) -> usize { /// A slice that uses typed IDs for indexing instead of raw `usize` values. /// -/// `IdSlice` is a transparent wrapper around `[T]` that enforces type-safe indexing -/// using ID types that implement the [`Id`] trait. +/// `IdSlice` has the same layout as `[T]` and uses types implementing [`Id`] for indexing. +/// Raw conversions preserve the slice length, including lengths beyond `I`'s range. Methods +/// producing IDs require the converted indices to be representable by `I`. /// -/// The API is not complete by design, new methods will be added as needed. -/// -/// # Examples +/// # Example /// /// ``` /// # use hashql_core::id::{IdSlice, Id as _, newtype}; @@ -77,26 +76,32 @@ impl IdSlice where I: Id, { - /// Creates a reference to an empty `IdSlice`. + /// Creates a reference to an empty typed slice. #[inline] #[must_use] pub const fn empty<'this>() -> &'this Self { Self::from_raw(&[]) } - /// Creates an `IdSlice` from a raw slice reference. + /// Creates a typed slice from a raw slice reference. #[inline] #[expect(unsafe_code, reason = "repr(transparent)")] pub const fn from_raw(raw: &[T]) -> &Self { - // SAFETY: `IdSlice` is repr(transparent) and has the same layout as `[T]`. + // SAFETY: `repr(transparent)` gives `IdSlice` the layout of `raw: [T]`. Its + // `PhantomData` field has size 0, alignment 1 and no validity requirements. + // The cast preserves the address, element-count metadata and input borrow lifetime. + // Therefore the shared reference is valid for the returned borrow. unsafe { &*(ptr::from_ref::<[T]>(raw) as *const Self) } } - /// Creates a mutable `IdSlice` from a raw mutable slice reference. + /// Creates a mutable typed slice from a raw mutable slice reference. #[inline] #[expect(unsafe_code, reason = "repr(transparent)")] pub const fn from_raw_mut(raw: &mut [T]) -> &mut Self { - // SAFETY: `IdSlice` is repr(transparent) and has the same layout as `[T]`. + // SAFETY: `repr(transparent)` gives `IdSlice` the layout of `raw: [T]`. Its + // `PhantomData` field has size 0, alignment 1 and no validity requirements. + // The cast preserves the address and element-count metadata. The returned borrow + // retains the input borrow's lifetime and exclusivity, making the reference valid. unsafe { &mut *(ptr::from_mut(raw) as *mut Self) } } @@ -116,48 +121,58 @@ where unsafe { &mut *(ptr::from_mut(self) as *mut [T]) } } - /// Creates an `IdSlice` from a boxed slice. + /// Converts a boxed slice into a boxed typed slice. #[inline] #[expect(unsafe_code, reason = "repr(transparent)")] pub fn from_boxed_slice(slice: Box<[T], A>) -> Box { let (ptr, alloc) = Box::into_raw_with_allocator(slice); - // SAFETY: `IdSlice` is repr(transparent) and we simply cast the underlying pointer. + // SAFETY: `Box::from_raw_in` requires valid contents and unique ownership under the + // original allocator and layout. `IdSlice` is `repr(transparent)` over `[T]`, with + // an unconstrained size-0, alignment-1 `PhantomData` companion. The cast preserves + // the address and element-count metadata, including for empty and zero-sized slices. + // Reconstructing once with the returned allocator preserves ownership and layout. unsafe { Box::from_raw_in(ptr as *mut Self, alloc) } } - /// Converts a boxed `IdSlice` back into its raw boxed slice. + /// Converts a boxed typed slice back into its raw boxed slice. /// - /// The inverse of [`from_boxed_slice`](Self::from_boxed_slice), dropping only the typed index - /// domain. Intended for the boundaries where a typed collection leaves the domain-indexed - /// world, such as handing storage to an external format that speaks raw indices. + /// The inverse of [`from_boxed_slice`](Self::from_boxed_slice), removing only the typed index + /// domain while preserving the allocation and elements. #[inline] #[expect(unsafe_code, reason = "repr(transparent)")] pub fn into_boxed_raw(slice: Box) -> Box<[T], A> { let (ptr, alloc) = Box::into_raw_with_allocator(slice); - // SAFETY: `IdSlice` is repr(transparent) and we simply cast the underlying pointer. + // SAFETY: `Box::from_raw_in` requires valid contents and unique ownership under the + // original allocator and layout. `repr(transparent)` gives `IdSlice` and `[T]` the + // same layout. The cast preserves the address and element-count metadata, including + // for empty and zero-sized slices. Reconstructing once with the returned allocator + // preserves ownership and layout. unsafe { Box::from_raw_in(ptr as *mut [T], alloc) } } - /// Converts to `Box, A>`. + /// Converts a boxed slice of initialized slots into a boxed typed slice. /// /// See [`Box::assume_init`] for additional details. /// /// # Safety /// - /// As with [`MaybeUninit::assume_init`], it is up to the caller to guarantee that the values - /// really are in an initialized state. Calling this when the content is not yet fully - /// initialized causes immediate undefined behavior. + /// All elements must contain initialized values that satisfy `T`'s validity requirements. + /// See [`MaybeUninit::assume_init`]. #[expect(unsafe_code)] pub unsafe fn boxed_assume_init( slice: Box>, A>, ) -> Box { let (ptr, alloc) = Box::into_raw_with_allocator(slice); - // SAFETY: The caller guarantees all elements are initialized and valid `T`s. - // `MaybeUninit` is #[repr(transparent)] over `T`, and `IdSlice` is #[repr(transparent)] - // over `[T]`, so the pointer cast (and its slice metadata) is layout-correct. + // SAFETY: `MaybeUninit` guarantees `T`'s size and alignment, and `Box::from_raw_in` + // requires valid contents and unique ownership under the original allocator and layout. + // The caller guarantees all elements contain valid, initialized `T`s. Both `IdSlice` + // types are `repr(transparent)` over their slice tails, with size-0, alignment-1 + // `PhantomData` companions requiring no initialized bytes. The casts preserve the + // address and element-count metadata, including for empty and zero-sized slices. + // Reconstructing once with the returned allocator is therefore valid. unsafe { Box::from_raw_in(ptr as *mut [MaybeUninit] as *mut [T] as *mut Self, alloc) } } @@ -208,9 +223,9 @@ where /// Returns the prefix of the slice below `bound`, keeping the index domain. /// - /// Range indexing returns a raw slice because a general sub-slice re-bases its indices to - /// offsets; a prefix never re-bases (every element keeps its original ID), so the typed view - /// is the honest return. + /// Range indexing returns a raw slice with indices relative to the subslice start. + /// A prefix starts at zero. Its indices never change, and every element keeps its original ID. + /// Returning a typed prefix preserves those IDs. /// /// # Panics /// @@ -242,12 +257,13 @@ where /// Returns the exclusive upper bound ID for this slice. /// - /// This is equivalent to the ID that would be assigned to a new element if one were added. - /// Useful for bounds checking: `id < slice.bound()` tests if `id` is valid for this slice. + /// The ID's numeric value equals the slice length. The comparison `id < slice.bound()` + /// tests whether `id` indexes an element of the slice. /// /// # Panics /// - /// The slice occupies the complete ID domain and no successor ID exists. + /// Panics if the slice length is outside `I`'s range. This includes a slice occupying the + /// complete ID domain, whose exclusive upper bound has no representable ID. #[inline] pub fn bound(&self) -> I { I::from_usize(self.len()) @@ -261,9 +277,13 @@ where self.raw.is_empty() } - /// Returns an iterator over all valid IDs for this slice. + /// Returns an iterator over all element IDs for this slice. + /// + /// The iterator converts indices in `0..self.len()` into IDs. + /// + /// # Panics /// - /// The iterator yields IDs from 0 up to (but not including) `bound()`. + /// Panics if an element index, or zero for an empty slice, is outside `I`'s range. pub fn ids(&self) -> impl DoubleEndedIterator + ExactSizeIterator + Clone + 'static { let length = self.len(); @@ -273,9 +293,13 @@ where (0..length).map(I::from_usize) } - /// Returns a parallel iterator over all valid IDs for this slice. + /// Returns a parallel iterator over all element IDs for this slice. /// /// The parallel counterpart of [`Self::ids`]. + /// + /// # Panics + /// + /// Panics if an element index, or zero for an empty slice, is outside `I`'s range. #[cfg(feature = "rayon")] pub fn par_ids(&self) -> impl IndexedParallelIterator + 'static { let length = self.len(); @@ -308,7 +332,11 @@ where /// Returns an iterator over ID-element pairs. /// - /// Similar to [`Iterator::enumerate`] but yields typed IDs instead of `usize` indices. + /// Like [`Iterator::enumerate`], with each index converted to `I`. + /// + /// # Panics + /// + /// Panics if an element index, or zero for an empty slice, is outside `I`'s range. pub fn iter_enumerated( &self, ) -> impl DoubleEndedIterator + ExactSizeIterator + Clone { @@ -323,8 +351,11 @@ where /// Returns a parallel iterator over ID-element pairs. /// - /// The parallel counterpart of [`Self::iter_enumerated`]: yields typed IDs instead of `usize` - /// indices. + /// The parallel counterpart of [`Self::iter_enumerated`]. + /// + /// # Panics + /// + /// Panics if an element index, or zero for an empty slice, is outside `I`'s range. #[cfg(feature = "rayon")] pub fn par_iter_enumerated(&self) -> impl IndexedParallelIterator where @@ -361,7 +392,11 @@ where /// Returns a mutable iterator over ID-element pairs. /// - /// Similar to [`Iterator::enumerate`] but yields typed IDs instead of `usize` indices. + /// Like [`Iterator::enumerate`], with each index converted to `I`. + /// + /// # Panics + /// + /// Panics if an element index, or zero for an empty slice, is outside `I`'s range. pub fn iter_enumerated_mut( &mut self, ) -> impl DoubleEndedIterator + ExactSizeIterator { @@ -376,8 +411,11 @@ where /// Returns a parallel mutable iterator over ID-element pairs. /// - /// The parallel counterpart of [`Self::iter_enumerated_mut`]: yields typed IDs instead of - /// `usize` indices. + /// The parallel counterpart of [`Self::iter_enumerated_mut`]. + /// + /// # Panics + /// + /// Panics if an element index, or zero for an empty slice, is outside `I`'s range. #[cfg(feature = "rayon")] pub fn par_iter_enumerated_mut(&mut self) -> impl IndexedParallelIterator where @@ -392,7 +430,7 @@ where .map(|(index, value)| (I::from_usize(index), value)) } - /// Swaps two elements in the vector. + /// Swaps two elements in the slice. /// /// See [`slice::swap`] for details. /// @@ -409,6 +447,10 @@ where /// Returns an iterator over contiguous array windows of size `N`. /// /// See [`slice::array_windows`] for details. + /// + /// # Panics + /// + /// Panics if `N` is zero. #[inline] pub fn windows(&self) -> impl ExactSizeIterator { self.raw.array_windows() @@ -416,8 +458,14 @@ where /// Returns an iterator over ID-window pairs for windows of size `N`. /// - /// Each window carries the ID of its first element, so a window at `(id, [a, b])` spans - /// `id` and `id.plus(1)`. See [`slice::array_windows`] for the window semantics. + /// Each window pairs with the ID of its first element. A window at `(id, [a, b])` contains + /// the elements at offsets `id.as_usize()` and `id.as_usize() + 1` in the original slice. + /// See [`slice::array_windows`] for the window semantics. + /// + /// # Panics + /// + /// Panics if `N` is zero or a window-start index is outside `I`'s range. A slice shorter + /// than `N` still requires `I` to represent zero. #[inline] pub fn windows_enumerated( &self, @@ -464,8 +512,12 @@ where /// Returns an iterator over chunks of size `size`, each with the ID of its first element. /// /// Each chunk is a slice of `size` elements, except the last chunk may be smaller. A chunk - /// at `(id, chunk)` spans `id` through `id.plus(chunk.len() - 1)`, mirroring - /// [`Self::windows_enumerated`]. + /// at `(id, chunk)` contains the original slice's offsets in + /// `id.as_usize()..id.as_usize() + chunk.len()`, mirroring [`Self::windows_enumerated`]. + /// + /// # Panics + /// + /// Panics if a chunk-start index, or zero for an empty slice, is outside `I`'s range. #[inline] pub fn chunks_enumerated( &self, @@ -483,6 +535,10 @@ where /// Returns a parallel iterator over chunks, each with the ID of its first element. /// /// The parallel counterpart of [`Self::chunks_enumerated`]. + /// + /// # Panics + /// + /// Panics if a chunk-start index, or zero for an empty slice, is outside `I`'s range. #[cfg(feature = "rayon")] #[inline] pub fn par_chunks_enumerated( @@ -531,9 +587,12 @@ where /// Returns a mutable iterator over chunks, each with the ID of its first element. /// - /// Each chunk is a mutable slice of `size` elements, except the last chunk may be smaller. A - /// chunk at `(id, chunk)` spans `id` through `id.plus(chunk.len() - 1)`, mirroring - /// [`Self::chunks_enumerated`]. + /// Each chunk is a mutable slice of `size` elements, except the last chunk may be smaller. + /// Its ID identifies the first element, as in [`Self::chunks_enumerated`]. + /// + /// # Panics + /// + /// Panics if a chunk-start index, or zero for an empty slice, is outside `I`'s range. #[inline] pub fn chunks_enumerated_mut( &mut self, @@ -551,6 +610,10 @@ where /// Returns a parallel mutable iterator over chunks, each with the ID of its first element. /// /// The parallel counterpart of [`Self::chunks_enumerated_mut`]. + /// + /// # Panics + /// + /// Panics if a chunk-start index, or zero for an empty slice, is outside `I`'s range. #[cfg(feature = "rayon")] #[inline] pub fn par_chunks_enumerated_mut( @@ -570,7 +633,7 @@ where .map(move |(index, chunk)| (I::from_usize(index * size.get()), chunk)) } - /// Sorts the slice in place with the given comparator, in unstable order. + /// Sorts the slice in place with `compare`, in unstable order. /// /// See [`slice::sort_unstable_by`](prim@slice#method.sort_unstable_by) for details. #[inline] @@ -578,7 +641,7 @@ where self.raw.sort_unstable_by(compare); } - /// Sorts the slice in place with the given comparator, in parallel and unstable order. + /// Sorts the slice in place with `compare`, in parallel and unstable order. /// /// The parallel counterpart of [`slice::sort_unstable_by`](prim@slice#method.sort_unstable_by). #[cfg(feature = "rayon")] @@ -590,7 +653,7 @@ where self.raw.par_sort_unstable_by(compare); } - /// Sorts the slice in place using the given key function, in unstable order. + /// Sorts the slice in place by the key from `func`, in unstable order. /// /// See [`slice::sort_unstable_by_key`](prim@slice#method.sort_unstable_by_key) for details. #[inline] @@ -601,7 +664,7 @@ where self.raw.sort_unstable_by_key(func); } - /// Sorts the slice in place using the given key function, in parallel and unstable order. + /// Sorts the slice in place by the key from `func`, in parallel and unstable order. /// /// The parallel counterpart of [`Self::sort_unstable_by_key`]. #[cfg(feature = "rayon")] @@ -625,7 +688,7 @@ where self.raw.is_sorted() } - /// Returns `true` if the slice is sorted by the given comparator. + /// Returns `true` if the slice is sorted according to `compare`. /// /// See [`slice::is_sorted_by`](prim@slice#method.is_sorted_by) for details. #[inline] @@ -633,19 +696,18 @@ where self.raw.is_sorted_by(compare) } - /// Searches for an item in the slice using binary search, returning the index of the item if - /// found. + /// Finds an item's ID by binary search. /// /// See [`slice::binary_search`](prim@slice#method.binary_search) for details. /// /// # Errors /// - /// Returns the ID where a matching element could be inserted while maintaining sorted order. + /// When no element matches, returns the ID where inserting `item` preserves sorted order. /// /// # Panics /// - /// The insertion position lies beyond the ID domain. This occurs when the slice occupies the - /// complete domain and `item` belongs after every element. + /// Panics if the matching index or insertion position is outside `I`'s range. A slice + /// occupying the complete ID domain panics when `item` belongs after every element. #[inline] pub fn binary_search(&self, item: &T) -> Result where @@ -667,7 +729,7 @@ where IdVec::from_raw(self.raw.to_vec()) } - /// Clones the slice into a new [`IdVec`] using the given allocator. + /// Clones the slice into a new [`IdVec`] using `alloc`. /// /// See [`slice::to_vec_in`](prim@slice#method.to_vec_in) for details. pub fn to_vec_in(&self, alloc: A) -> IdVec @@ -680,11 +742,13 @@ where /// Returns the ID of the first element for which `predicate` is false. /// + /// Returns the exclusive upper bound if every element satisfies `predicate`. /// See [`slice::partition_point`](prim@slice#method.partition_point) for details. /// /// # Panics /// - /// Every element satisfies `predicate` and the exclusive end lies beyond the ID domain. + /// Panics if the partition position is outside `I`'s range, including the exclusive end + /// when every element satisfies `predicate`. #[inline] pub fn partition_point(&self, predicate: impl Fn(&T) -> bool) -> I { let index = self.raw.partition_point(predicate); @@ -724,27 +788,29 @@ impl IdSlice> where I: Id, { - /// Converts `&mut IdSlice>` to `&mut IdSlice`. + /// Borrows all initialized elements as a mutable typed slice. /// /// # Safety /// - /// As with [`MaybeUninit::assume_init`], it is up to the caller to guarantee that the values - /// really are in an initialized state. Calling this when the content is not yet fully - /// initialized causes immediate undefined behavior. + /// All elements must contain initialized values that satisfy `T`'s validity requirements. + /// See [`MaybeUninit::assume_init`]. pub const unsafe fn assume_init_mut(&mut self) -> &mut IdSlice { - // SAFETY: The caller must ensure that all elements are initialized. + // SAFETY: the slice operation requires initialized values satisfying `T`'s validity + // requirements in all slots. The caller must guarantee that condition. The mutable + // borrow preserves exclusivity and lifetime, making the conversion valid. IdSlice::from_raw_mut(unsafe { self.raw.assume_init_mut() }) } - /// Converts `&IdSlice>` to `&IdSlice`. + /// Borrows all initialized elements as a shared typed slice. /// /// # Safety /// - /// As with [`MaybeUninit::assume_init`], it is up to the caller to guarantee that the values - /// really are in an initialized state. Calling this when the content is not yet fully - /// initialized causes immediate undefined behavior. + /// All elements must contain initialized values that satisfy `T`'s validity requirements. + /// See [`MaybeUninit::assume_init`]. pub const unsafe fn assume_init_ref(&self) -> &IdSlice { - // SAFETY: The caller must ensure that all elements are initialized. + // SAFETY: the slice operation requires initialized values satisfying `T`'s validity + // requirements in all slots. The caller must guarantee that condition. The shared + // borrow preserves the input lifetime, making the conversion valid. IdSlice::from_raw(unsafe { self.raw.assume_init_ref() }) } } @@ -753,12 +819,12 @@ impl IdSlice> where I: Id, { - /// Removes and returns the value at the given ID index. + /// Removes and returns the value at `index`. /// /// Returns `None` if the index is out of bounds or if the value was already `None`. - /// The vector is not shrunk after removal. + /// The slice length is unchanged after removal. /// - /// # Examples + /// # Example /// /// ``` /// # use hashql_core::id::{IdVec, Id as _, newtype}; @@ -773,9 +839,9 @@ where self.get_mut(index)?.take() } - /// Returns `true` if the vector contains a value (not `None`) at the given ID index. + /// Returns `true` if the slice contains a value at `index`. /// - /// # Examples + /// # Example /// /// ``` /// # use hashql_core::id::{IdVec, Id as _, newtype}; @@ -793,7 +859,7 @@ where /// /// Returns [`None`] if the index is out of bounds or if the value at that index is [`None`]. /// - /// # Examples + /// # Example /// /// ``` /// # use hashql_core::id::{IdVec, Id as _, newtype}; @@ -813,7 +879,7 @@ where /// /// Returns [`None`] if the index is out of bounds or if the value at that index is [`None`]. /// - /// # Examples + /// # Example /// /// ``` /// # use hashql_core::id::{IdVec, Id as _, newtype}; @@ -914,15 +980,23 @@ where unsafe_code, reason = "repr(transparent): the clone writes through `[T]`'s own layout" )] -// SAFETY: `IdSlice` is `repr(transparent)` over `[T]`, so a clone of the inner slice written to -// `dest` is a valid clone of the whole value under the same layout. +// SAFETY: A normal return must leave `*dest` a valid `IdSlice` under `self`'s pointer +// metadata. `IdSlice` is `repr(transparent)`: `raw: [T]` is the transparent tail and +// `_marker: PhantomData` its size-0, alignment-1 companion, valid under every byte +// pattern. `raw` starts at offset 0 and shares `self`'s size, alignment and element-count +// metadata. The body forwards `dest` unchanged to `<[T]>::clone_to_uninit(&self.raw, dest)`, +// whose normal return leaves a valid `[T]` of `self.raw.len()` elements at `dest`. Beside the +// marker, that is a fully initialized `IdSlice`. unsafe impl CloneToUninit for IdSlice where T: Clone, { unsafe fn clone_to_uninit(&self, dest: *mut u8) { - // SAFETY: the caller passes `dest` valid for writes of `self`'s layout, which - // `repr(transparent)` makes exactly `self.raw`'s layout. + // SAFETY: `<[T]>::clone_to_uninit` requires `dest` valid for writes of + // `size_of_val(&self.raw)` bytes and aligned to `align_of_val(&self.raw)`. The caller + // guarantees both for `self`, and the `repr(transparent)` layout gives `self` and + // `self.raw` exactly the same size and alignment: zero bytes for an empty slice or a + // zero-sized `T`, and `T`'s own alignment for an over-aligned `T`. unsafe { <[T]>::clone_to_uninit(&self.raw, dest); } @@ -1168,7 +1242,7 @@ mod tests { } let id_slice = IdSlice::::from_boxed_slice(uninit); - // SAFETY: All elements were initialized in the loop above + // SAFETY: the loop above initializes all elements. let init = unsafe { IdSlice::boxed_assume_init(id_slice) }; assert_eq!(init.len(), 4);