diff --git a/.claude/board/LATEST_STATE.md b/.claude/board/LATEST_STATE.md index 8e036aec3..7f4e92863 100644 --- a/.claude/board/LATEST_STATE.md +++ b/.claude/board/LATEST_STATE.md @@ -1,3 +1,13 @@ +## 2026-10-04 — read-mode resolved once per population (branch `ccr-f6094d67-h6ulb3`, unmerged) + +### Current Contract Inventory — net delta (`nan_projection.rs`, `soa_graph.rs`) +- `project_energy_nonfinite_resolved(rows, ValueSchema)` and + `energy_all_finite_resolved(rows, ValueSchema)`: no per-row lookup. The + mixed-batch API resolves once per classid run. Entry `D-HPS-2`. +- `soa_graph` helpers take the domain's resolved `TailVariant`. +- `lance-graph-mask-risc`: `Pred::MatchFacet16Strided` (16-byte strided + ternary match, over `ndarray::simd::ternary_match_strided16_to_mask`). + ## 2026-10-04 — Register128 slab reading + bounded power sums (branch `ccr-1d39fce9-gdgy6k`, unmerged, D-LXC-29) ### Current Contract Inventory — net delta (`register128.rs`, `hotplug.rs`, `canonical_node.rs`) diff --git a/.claude/board/entries/2026-10-04-resolve-once-population-execution.md b/.claude/board/entries/2026-10-04-resolve-once-population-execution.md new file mode 100644 index 000000000..63ea71d2f --- /dev/null +++ b/.claude/board/entries/2026-10-04-resolve-once-population-execution.md @@ -0,0 +1,39 @@ +# D-HPS-2 — resolve once, then run the population (2026-10-04) + +**STATUS:** measured (branch `ccr-f6094d67-h6ulb3`, unmerged) + +Follows `D-HPS-1`. The metadata is resolved once per population, and the row +loops do no registry / ClassView / read-mode lookup. + +- `soa_graph`: `project_snapshot` and `nearest_anchor` resolve the domain's + `TailVariant` once (`domain_tail`) and pass it to `hhtl_path` / `family_of` / + `identity_of`. Before: one `classid_read_mode` per helper call per row. +- `nan_projection`: new `project_energy_nonfinite_resolved(rows, schema)` and + `energy_all_finite_resolved(rows, schema)` take the population's + `ValueSchema` (e.g. `ResolvedReading::read_mode.value_schema`) and do no + lookup. The mixed API resolves once per run of equal classid and delegates + to them; its per-row residue is one classid compare. The schema gate and the + exponent-mask test are unchanged. +- Falsifiers: a test-only counter in `classid_read_mode` pins lookups at 1 for + both 1 and 1000 rows (soa_graph, mixed wrapper) and at 0 for the resolved + path. Disable runs, all red: per-row lookup in `hhtl_path`; per-row lookup in + the resolved loop; schema gate dropped; mixed wrapper reduced to runs of 1. +- Measured (release, `target-cpu=native`, 100k homogeneous rows): + per-row lookup 23.7–24.6 ns/row, resolved 5.0 ns/row, mixed wrapper + 10.0–10.3 ns/row. soa_graph was not timed. +- The 128-bit matcher lives in ndarray (`ternary_match_strided16_to_mask`, + ndarray #340). 12 B vs 16 B at stride 512, 65,536 rows: 5.1–5.3 ns/row for + both; no measurable cost. + +- mask-risc: `Pred::MatchFacet16Strided {lane, pattern: [u8; 16], care: [u8; 16]}` + lowers to `ternary_match_strided16_to_mask` (ndarray #340, merged). The view + is validated 16 bytes wide; the oracle compares all 16 bytes. Disable runs, + all red: executor on the 12 B kernel; oracle on 12 bytes; width check at 12. + No new IR shape; Quack unchanged (no caller needs a spelling yet). + +**OPEN:** +- The bake paths were not changed: q2 `osint-bake/src/bin/fma.rs` does a + homogeneous per-node `classid_read_mode(CLASSID_FMA)` (another repo), and + deepnsm-v2 `promote.rs` `key_at` does one per call. +- symbiont `domino.rs` is not migrated: its boards are classid 0, so the mixed + wrapper already costs it one lookup. diff --git a/.claude/board/entries/README.md b/.claude/board/entries/README.md index 61f3ecaff..a37e51eb6 100644 --- a/.claude/board/entries/README.md +++ b/.claude/board/entries/README.md @@ -25,12 +25,13 @@ index row, (3) no duplicate entry id. Checks 1 and 2 are deliberately opposite directions; the stranding this convention prevents shows up in exactly one of them, never both. -212 entries, 2026-08-06 .. 2026-10-04. +213 entries, 2026-08-06 .. 2026-10-04. | date | entry id | finding | file | |---|---|---|---| | 2026-10-04 | `D-LXC-22` | | [2026-10-04-wordnet-clam-chaoda-scope-and-grammar-read-params.md](2026-10-04-wordnet-clam-chaoda-scope-and-grammar-read-params.md) | | 2026-10-04 | `D-HPS-1` | | [2026-10-04-spog-slab-hotplug-resolution.md](2026-10-04-spog-slab-hotplug-resolution.md) | +| 2026-10-04 | `D-HPS-2` | | [2026-10-04-resolve-once-population-execution.md](2026-10-04-resolve-once-population-execution.md) | | 2026-10-04 | `D-LXC-29-R` | | [2026-10-04-register128-bounded-power-sums.md](2026-10-04-register128-bounded-power-sums.md) | | 2026-10-04 | `D-LXC-25` | | [2026-10-04-deepnsm-v2-wechsel-lane-quorum.md](2026-10-04-deepnsm-v2-wechsel-lane-quorum.md) | | 2026-10-04 | `D-LXC-27` | | [2026-10-04-deepnsm-v2-unseen-noun-gender.md](2026-10-04-deepnsm-v2-unseen-noun-gender.md) | diff --git a/crates/lance-graph-contract/src/canonical_node.rs b/crates/lance-graph-contract/src/canonical_node.rs index de68f4250..f2489777e 100644 --- a/crates/lance-graph-contract/src/canonical_node.rs +++ b/crates/lance-graph-contract/src/canonical_node.rs @@ -1724,12 +1724,36 @@ static BUILTIN_READ_MODES: LazyLock> = LazyLock::new(|| { m }); +#[cfg(test)] +thread_local! { + /// Test-only count of [`classid_read_mode`] calls on this thread. Lets a + /// test assert that a population path resolves its reading a constant + /// number of times, not once per row. Thread-local, so parallel tests do + /// not see each other's lookups. + pub(crate) static READ_MODE_LOOKUPS: core::cell::Cell = + const { core::cell::Cell::new(0) }; +} + +/// Number of [`classid_read_mode`] calls on this thread since the last reset. +#[cfg(test)] +pub(crate) fn read_mode_lookups() -> usize { + READ_MODE_LOOKUPS.with(core::cell::Cell::get) +} + +/// Reset this thread's [`classid_read_mode`] call count. +#[cfg(test)] +pub(crate) fn reset_read_mode_lookups() { + READ_MODE_LOOKUPS.with(|c| c.set(0)); +} + /// Resolve a `classid` to its [`ReadMode`] — the single source both consumers /// and OGAR inherit. Reads the [`BUILTIN_READ_MODES`] registry, falling through /// to [`ReadMode::DEFAULT`] for any unconfigured classid (the key's own /// zero-fallback ladder). [`NodeGuid::read_mode`] is the carrier-method form. #[inline] pub fn classid_read_mode(classid: u32) -> ReadMode { + #[cfg(test)] + READ_MODE_LOOKUPS.with(|c| c.set(c.get() + 1)); BUILTIN_READ_MODES .get(&classid) .copied() diff --git a/crates/lance-graph-contract/src/nan_projection.rs b/crates/lance-graph-contract/src/nan_projection.rs index 76665eeb1..3db893ee6 100644 --- a/crates/lance-graph-contract/src/nan_projection.rs +++ b/crates/lance-graph-contract/src/nan_projection.rs @@ -24,17 +24,20 @@ //! happened to zero out. Each row is therefore gated on its OWN resolved //! `[ValueSchema::has]` before its `Energy` bytes are read at all. //! -//! **What "branchless" still means after the gate, precisely (codex review, -//! 2026-07-30).** The FINITENESS TEST — the exponent-mask compare on the -//! four already-loaded bytes — is unchanged: still zero branches on the -//! value. That is not the same claim as "the sweep costs what it did -//! before." [`row_has_energy`] calls [`NodeGuid::read_mode`], which resolves -//! through [`classid_read_mode`] — a `HashMap` lookup behind a `LazyLock`, -//! not a bitmask. That lookup is real per-row work, added on top of the old -//! four-byte load, and can plausibly dominate it for an in-cache homogeneous -//! batch. Not benchmarked; do not read "branchless" below as "free" — if -//! this projection lands on a genuinely hot path, that lookup is the first -//! place to look before assuming the schema gate is costless. +//! **Where the schema is resolved.** The finiteness test — the exponent-mask +//! compare on four loaded bytes — has no branch on the value. The schema +//! lookup ([`classid_read_mode`], a `HashMap` behind a `LazyLock`) is kept +//! out of the per-row loop: +//! +//! - [`project_energy_nonfinite_resolved`] / [`energy_all_finite_resolved`] +//! take the population's already-resolved [`ValueSchema`] (e.g. +//! `ResolvedReading::read_mode.value_schema`). One schema check per call; +//! no registry lookup at all. +//! - [`project_energy_nonfinite`] / [`energy_all_finite`] accept a mixed +//! batch. They resolve once per RUN of equal classid and hand each run to +//! the resolved path, so a homogeneous batch costs one lookup, and a batch +//! alternating classids costs one per change. The only per-row metadata +//! work left there is comparing the row's classid with the run's. //! //! [`NanReport::skipped`] makes the gate's effect observable rather than a //! silent no-op, per the workspace's can-it-fire testing rule. @@ -48,7 +51,7 @@ //! [`NodeGuid::read_mode`]: crate::canonical_node::NodeGuid::read_mode //! [`classid_read_mode`]: crate::canonical_node::classid_read_mode -use crate::canonical_node::{NodeRow, ValueTenant}; +use crate::canonical_node::{classid_read_mode, NodeRow, ValueSchema, ValueTenant}; /// `true` iff an `f32` bit pattern is non-finite (Inf or NaN): the exponent /// field is all-ones. No float materialised. @@ -92,8 +95,8 @@ impl NanReport { } /// Read one board's `Energy` tenant as a raw `f32` bit pattern (no float load). -/// Caller MUST have already confirmed the row's schema materialises `Energy` -/// ([`row_has_energy`]) — this function does not gate. +/// Caller MUST have already confirmed the population's schema materialises +/// `Energy` — this function does not gate. #[inline] fn energy_bits(row: &NodeRow) -> u32 { let off = ValueTenant::Energy.value_offset(); @@ -105,45 +108,93 @@ fn energy_bits(row: &NodeRow) -> u32 { ]) } -/// Does this row's OWN resolved schema materialise `Energy`? The one branch -/// this module adds — on schema presence, never on the float value. -#[inline] -fn row_has_energy(row: &NodeRow) -> bool { - row.key.read_mode().value_schema.has(ValueTenant::Energy) -} - -/// Project a batch of canonical boards onto the NaN-detection surface by reading -/// each one's `Energy` tenant — schema-gated per row (see module docs). Read-only; -/// returns the indices of non-finite boards among those actually inspected. -/// This is the demoted singleton BindSpace — a projection, never a carrier. -pub fn project_energy_nonfinite(rows: &[NodeRow]) -> NanReport { - let mut total = 0usize; - let mut skipped = 0usize; +/// Project a population whose reading is ALREADY RESOLVED onto the +/// NaN-detection surface. `schema` is the population's value schema (for a +/// [`crate::hotplug::ResolvedReading`], its `read_mode.value_schema`); every +/// row is read under it. No registry or read-mode lookup happens here. +/// +/// If `schema` does not materialise `Energy`, no `Energy` bytes are read: every +/// row is `skipped` and the report is clean. Otherwise each row's `Energy` is +/// tested with the integer exponent mask. +/// +/// The caller owns the homogeneity claim. For a batch that may mix classids, +/// use [`project_energy_nonfinite`]. +pub fn project_energy_nonfinite_resolved(rows: &[NodeRow], schema: ValueSchema) -> NanReport { + if !schema.has(ValueTenant::Energy) { + return NanReport { + total: 0, + nonfinite: Vec::new(), + skipped: rows.len(), + }; + } let mut nonfinite = Vec::new(); for (i, row) in rows.iter().enumerate() { - if !row_has_energy(row) { - skipped += 1; - continue; - } - total += 1; if f32_bits_nonfinite(energy_bits(row)) { nonfinite.push(i as u32); } } NanReport { - total, + total: rows.len(), nonfinite, - skipped, + skipped: 0, + } +} + +/// Clean/dirty answer for a population whose reading is already resolved — +/// the sibling of [`project_energy_nonfinite_resolved`]. Early-outs on the +/// first non-finite board; `true` without reading anything when `schema` has +/// no `Energy`. +pub fn energy_all_finite_resolved(rows: &[NodeRow], schema: ValueSchema) -> bool { + !schema.has(ValueTenant::Energy) || rows.iter().all(|row| !f32_bits_nonfinite(energy_bits(row))) +} + +/// Split `rows` into maximal runs of equal classid, resolving each run's +/// value schema once. Yields `(start index, run, schema)`. +fn schema_runs(rows: &[NodeRow]) -> impl Iterator { + let mut start = 0usize; + core::iter::from_fn(move || { + if start >= rows.len() { + return None; + } + let classid = rows[start].key.classid(); + let len = rows[start..] + .iter() + .take_while(|r| r.key.classid() == classid) + .count(); + let run = &rows[start..start + len]; + let at = start; + start += len; + Some((at, run, classid_read_mode(classid).value_schema)) + }) +} + +/// Project a batch of canonical boards onto the NaN-detection surface by reading +/// each one's `Energy` tenant — schema-gated per row (see module docs). Read-only; +/// returns the indices of non-finite boards among those actually inspected. +/// This is the demoted singleton BindSpace — a projection, never a carrier. +/// +/// Accepts a batch mixing classids. Each run of equal classid is resolved once +/// and handed to [`project_energy_nonfinite_resolved`]; a caller that already +/// knows its population's reading should call that directly. +pub fn project_energy_nonfinite(rows: &[NodeRow]) -> NanReport { + let mut report = NanReport::default(); + for (at, run, schema) in schema_runs(rows) { + let r = project_energy_nonfinite_resolved(run, schema); + report.total += r.total; + report.skipped += r.skipped; + report + .nonfinite + .extend(r.nonfinite.into_iter().map(|i| i + at as u32)); } + report } /// Fast clean/dirty answer without materialising the index list — the cheapest /// projection (early-outs on the first non-finite board). Rows whose schema -/// omits `Energy` are skipped, not treated as a violation. +/// omits `Energy` are skipped, not treated as a violation. Mixed batches are +/// resolved once per run of equal classid. pub fn energy_all_finite(rows: &[NodeRow]) -> bool { - rows.iter() - .filter(|row| row_has_energy(row)) - .all(|row| !f32_bits_nonfinite(energy_bits(row))) + schema_runs(rows).all(|(_, run, schema)| energy_all_finite_resolved(run, schema)) } #[cfg(test)] @@ -256,4 +307,73 @@ mod tests { "energy_all_finite must agree with project_energy_nonfinite" ); } + + // ── Resolved path: schema resolved once by the caller ───────────────────── + + #[test] + fn resolved_path_does_no_read_mode_lookup() { + use crate::canonical_node::{read_mode_lookups, reset_read_mode_lookups}; + let cognitive = classid_read_mode(NodeGuid::CLASSID_OSINT).value_schema; + assert!(cognitive.has(ValueTenant::Energy)); + for n in [1usize, 1000] { + let mut rows: Vec = (0..n).map(|i| board_with(i as f32)).collect(); + rows[n - 1] = board_with(f32::NAN); + reset_read_mode_lookups(); + let r = project_energy_nonfinite_resolved(&rows, cognitive); + let clean = energy_all_finite_resolved(&rows, cognitive); + assert_eq!(read_mode_lookups(), 0, "n = {n}"); + // anti-vacuity: the rows were actually read + assert_eq!(r.total, n); + assert_eq!(r.nonfinite, vec![(n - 1) as u32]); + assert!(!clean); + } + } + + #[test] + fn resolved_path_keeps_the_schema_gate() { + let compressed = classid_read_mode(NodeGuid::CLASSID_FMA).value_schema; + assert!(!compressed.has(ValueTenant::Energy)); + let rows = vec![ + board_with_classid(NodeGuid::CLASSID_FMA, f32::NAN), + board_with_classid(NodeGuid::CLASSID_FMA, f32::INFINITY), + ]; + // the bytes really are poisoned, so a missing gate would report them + assert!(f32_bits_nonfinite(energy_bits(&rows[0]))); + let r = project_energy_nonfinite_resolved(&rows, compressed); + assert_eq!((r.total, r.skipped), (0, 2)); + assert!(r.nonfinite.is_empty()); + assert!(energy_all_finite_resolved(&rows, compressed)); + } + + #[test] + fn mixed_wrapper_resolves_once_per_classid_run() { + use crate::canonical_node::{read_mode_lookups, reset_read_mode_lookups}; + for n in [1usize, 1000] { + let rows: Vec = (0..n).map(|i| board_with(i as f32)).collect(); + reset_read_mode_lookups(); + let r = project_energy_nonfinite(&rows); + assert_eq!(read_mode_lookups(), 1, "homogeneous batch, n = {n}"); + assert_eq!(r.total, n); + reset_read_mode_lookups(); + assert!(energy_all_finite(&rows)); + assert_eq!(read_mode_lookups(), 1, "homogeneous batch, n = {n}"); + } + } + + #[test] + fn mixed_wrapper_reports_indices_into_the_whole_batch() { + let rows = vec![ + board_with_classid(NodeGuid::CLASSID_FMA, f32::NAN), + board_with_classid(NodeGuid::CLASSID_OSINT, f32::NAN), + board_with_classid(NodeGuid::CLASSID_FMA, f32::NAN), + board_with_classid(NodeGuid::CLASSID_OSINT, 1.0), + board_with_classid(NodeGuid::CLASSID_OSINT, f32::INFINITY), + ]; + let r = project_energy_nonfinite(&rows); + assert_eq!(r.nonfinite, vec![1, 4]); + assert_eq!((r.total, r.skipped), (3, 2)); + assert!(!energy_all_finite(&rows)); + // the FMA rows alone are skipped, not read + assert!(energy_all_finite(&[rows[0], rows[2]])); + } } diff --git a/crates/lance-graph-contract/src/soa_graph.rs b/crates/lance-graph-contract/src/soa_graph.rs index e4035869d..ad70cda45 100644 --- a/crates/lance-graph-contract/src/soa_graph.rs +++ b/crates/lance-graph-contract/src/soa_graph.rs @@ -57,7 +57,7 @@ //! [`NodeGuid::CLASSID_FMA`]). New domains are just another `DomainSpec` — //! the projector is domain-agnostic. -use crate::canonical_node::{NodeGuid, NodeRow}; +use crate::canonical_node::{NodeGuid, NodeRow, TailVariant}; use crate::graph_render::{GraphSnapshot, RenderEdge, RenderNode}; use crate::hhtl::NiblePath; use std::collections::HashMap; @@ -183,18 +183,27 @@ fn family_node_id(family: u32) -> String { /// never collapses to [`NiblePath::EMPTY`]. Every other classid uses the canonical /// v1 lowering (`classid_lo·HEEL·HIP·TWIG`), which falls back to /// [`NiblePath::EMPTY`] only for the v1-fold case of a non-zero high `classid` u16. +/// +/// `tail` is the domain's `tail_variant`, resolved ONCE by the caller +/// ([`domain_tail`]) — never looked up per row. #[inline] -fn hhtl_path(guid: &NodeGuid) -> NiblePath { +fn hhtl_path(guid: &NodeGuid, tail: TailVariant) -> NiblePath { #[cfg(feature = "guid-v3-tail")] - { - use crate::canonical_node::{classid_read_mode, TailVariant}; - if classid_read_mode(guid.classid()).tail_variant == TailVariant::V3 { - return NiblePath::from_guid_prefix_v3(guid); - } + if tail == TailVariant::V3 { + return NiblePath::from_guid_prefix_v3(guid); } + let _ = tail; NiblePath::from_guid_prefix(guid).unwrap_or(NiblePath::EMPTY) } +/// The domain's `tail_variant`, resolved once per projection. Every row the +/// projectors read has already been filtered to `domain.classid`, so one +/// registry lookup serves the whole population. +#[inline] +fn domain_tail(domain: &DomainSpec) -> TailVariant { + crate::canonical_node::classid_read_mode(domain.classid).tail_variant +} + /// The node's basin-`family` id, decoded per its `tail_variant` — the /// family-grouping/anchoring counterpart of [`hhtl_path`]'s V3 routing branch. /// A V2/V3 tail stores `family` in bytes 12..14 ([`NodeGuid::family_v2`]); the V1 @@ -204,17 +213,12 @@ fn hhtl_path(guid: &NodeGuid) -> NiblePath { /// mapping exactly like the path is. Under no tail feature every classid is V1, so /// this is just `family()`. #[inline] -fn family_of(guid: &NodeGuid) -> u32 { +fn family_of(guid: &NodeGuid, tail: TailVariant) -> u32 { #[cfg(feature = "guid-v2-tail")] - { - use crate::canonical_node::{classid_read_mode, TailVariant}; - if matches!( - classid_read_mode(guid.classid()).tail_variant, - TailVariant::V2 | TailVariant::V3 - ) { - return guid.family_v2() as u32; - } + if matches!(tail, TailVariant::V2 | TailVariant::V3) { + return guid.family_v2() as u32; } + let _ = tail; guid.family() } @@ -222,17 +226,12 @@ fn family_of(guid: &NodeGuid) -> u32 { /// [`family_of`]): V2/V3 read [`NodeGuid::identity_v2`] (bytes 14..16), V1 reads /// [`NodeGuid::identity`] (bytes 13..16). #[inline] -fn identity_of(guid: &NodeGuid) -> u32 { +fn identity_of(guid: &NodeGuid, tail: TailVariant) -> u32 { #[cfg(feature = "guid-v2-tail")] - { - use crate::canonical_node::{classid_read_mode, TailVariant}; - if matches!( - classid_read_mode(guid.classid()).tail_variant, - TailVariant::V2 | TailVariant::V3 - ) { - return guid.identity_v2() as u32; - } + if matches!(tail, TailVariant::V2 | TailVariant::V3) { + return guid.identity_v2() as u32; } + let _ = tail; guid.identity() } @@ -247,6 +246,7 @@ pub fn project_snapshot(rows: &[NodeRow], domain: &DomainSpec) -> GraphSnapshot .iter() .filter(|r| r.key.classid() == domain.classid) .collect(); + let tail = domain_tail(domain); // family → member count, and a COLLISION-AWARE family-low-byte → family map. // codex P1: with >256 families two ids can share a low byte; a duplicate @@ -255,7 +255,7 @@ pub fn project_snapshot(rows: &[NodeRow], domain: &DomainSpec) -> GraphSnapshot let mut by_family: HashMap = HashMap::new(); let mut family_by_low: HashMap> = HashMap::new(); for row in &domain_rows { - let fam = family_of(&row.key); + let fam = family_of(&row.key, tail); *by_family.entry(fam).or_insert(0) += 1; family_by_low .entry((fam & 0xFF) as u8) @@ -300,16 +300,19 @@ pub fn project_snapshot(rows: &[NodeRow], domain: &DomainSpec) -> GraphSnapshot // Member nodes + their edges (all head-only, family-adapter resolution). for row in &domain_rows { let g = row.key; - let fam = family_of(&g); + let fam = family_of(&g, tail); nodes.push(RenderNode { id: g.to_string(), - label: format!("{:06x}", identity_of(&g)), + label: format!("{:06x}", identity_of(&g, tail)), kind: domain.name.to_string(), confidence: 1.0, props: vec![ ("classid".to_string(), format!("{:08x}", g.classid())), ("family".to_string(), format!("{fam:06x}")), - ("hhtl_depth".to_string(), hhtl_path(&g).depth().to_string()), + ( + "hhtl_depth".to_string(), + hhtl_path(&g, tail).depth().to_string(), + ), ], }); // member → own family containment @@ -389,19 +392,20 @@ pub fn nearest_anchor(rows: &[NodeRow], domain: &DomainSpec) -> Vec { .iter() .filter(|r| r.key.classid() == domain.classid) .collect(); + let tail = domain_tail(domain); // Representative HHTL path per anchor family (first member encountered). let mut anchor_paths: Vec<(u32, NiblePath)> = Vec::new(); for row in &domain_rows { - let fam = family_of(&row.key); + let fam = family_of(&row.key, tail); if domain.anchor_families.contains(&fam) && !anchor_paths.iter().any(|(f, _)| *f == fam) { - anchor_paths.push((fam, hhtl_path(&row.key))); + anchor_paths.push((fam, hhtl_path(&row.key, tail))); } } domain_rows .iter() .map(|row| { let g = row.key; - let p = hhtl_path(&g); + let p = hhtl_path(&g, tail); let mut anchor_family = u32::MAX; let mut hops = u8::MAX; for &(fam, ap) in &anchor_paths { @@ -464,6 +468,55 @@ mod tests { } } + /// The projectors resolve the domain's reading once per call, never once + /// per row. Two-sided on the population size: 1 row and 1000 rows must + /// cost the same number of registry lookups. Instrumented at the registry + /// itself (`classid_read_mode`), so a lookup hidden behind any helper is + /// counted. + #[test] + fn domain_projection_resolves_its_reading_once_not_per_row() { + use crate::canonical_node::{read_mode_lookups, reset_read_mode_lookups}; + let dom = DomainSpec { + classid: NodeGuid::CLASSID_FMA, + name: "fma", + anchor_families: &[0x0001, 0x0002], + in_family_edge: "adjacent-to", + out_family_edge: "part-of", + member_edge: "part-of", + }; + let mut counts = Vec::new(); + for n in [1u32, 1000] { + let rows: Vec = (0..n) + .map(|i| { + node( + &dom, + (1, (i % 7) as u16, (i % 3) as u16), + 1 + i % 4, + i + 1, + &[2], + &[1], + ) + }) + .collect(); + reset_read_mode_lookups(); + let snap = project_snapshot(&rows, &dom); + let after_snapshot = read_mode_lookups(); + let hops = nearest_anchor(&rows, &dom); + let after_anchor = read_mode_lookups(); + assert!( + snap.nodes.len() as u32 > n, + "anti-vacuity: rows were projected" + ); + assert_eq!(hops.len() as u32, n, "anti-vacuity: every row was ranked"); + counts.push((after_snapshot, after_anchor - after_snapshot)); + } + assert_eq!(counts[0], (1, 1), "one lookup per projection call"); + assert_eq!( + counts[0], counts[1], + "lookup count must not scale with rows" + ); + } + #[cfg(feature = "guid-v3-tail")] #[test] fn v3_rows_decode_family_and_identity_via_tail_variant() { @@ -485,9 +538,9 @@ mod tests { ); // The tail-aware helpers read the V3 basin (family_v2 / identity_v2)… - assert_eq!(family_of(&g), 0xBBBB, "V3 family = family_v2 (12..14)"); + assert_eq!(family_of(&g, tv), 0xBBBB, "V3 family = family_v2 (12..14)"); assert_eq!( - identity_of(&g), + identity_of(&g, tv), 0xCCCC, "V3 identity = identity_v2 (14..16)" ); diff --git a/crates/lance-graph-mask-risc/src/exec.rs b/crates/lance-graph-mask-risc/src/exec.rs index 60b783a79..27a67c540 100644 --- a/crates/lance-graph-mask-risc/src/exec.rs +++ b/crates/lance-graph-mask-risc/src/exec.rs @@ -38,9 +38,9 @@ use ndarray::simd::{ masked_group_sum_sym_i32_pair, masked_group_sum_sym_i32_via, masked_key_run_count_u32, masked_max_i32, masked_min_i32, masked_strided_group_sum, masked_sum_i32, ne_i32_to_mask, ne_i32_to_mask_under, ne_u32_to_mask, ne_u32_to_mask_under, popcount_batch_u64, - ternary_match_strided_to_mask, ternary_match_u32_to_mask, ternary_match_u32_to_mask_under, - ternary_match_u64_to_mask, ternary_match_u64_to_mask_under, CrossPowerSums, KeyRunCarry, - PowerSums, + ternary_match_strided16_to_mask, ternary_match_strided_to_mask, ternary_match_u32_to_mask, + ternary_match_u32_to_mask_under, ternary_match_u64_to_mask, ternary_match_u64_to_mask_under, + CrossPowerSums, KeyRunCarry, PowerSums, }; use crate::ir::{ @@ -725,6 +725,29 @@ fn run_pred<'a>( mask_and_assign(dst, read(planes, s, u, t)); } } + ( + Pred::MatchFacet16Strided { + lane, + pattern, + care, + }, + under, + ) => { + if let Some(sv) = lane_strided(planes, lane) { + ternary_match_strided16_to_mask( + sv.bytes, + strided_tile_offset(&sv, t), + sv.stride, + t.rows, + &pattern, + &care, + dst, + ); + } + if let Some(u) = under { + mask_and_assign(dst, read(planes, s, u, t)); + } + } } } diff --git a/crates/lance-graph-mask-risc/src/ir.rs b/crates/lance-graph-mask-risc/src/ir.rs index 0745cf2f4..48953973c 100644 --- a/crates/lance-graph-mask-risc/src/ir.rs +++ b/crates/lance-graph-mask-risc/src/ir.rs @@ -192,6 +192,16 @@ pub enum Pred { pattern: [u8; 12], care: [u8; 12], }, + /// `((field_i[k] ^ pattern[k]) & care[k]) == 0` for every `k < 16` over a + /// [`LaneRef::Strided`] view: the same ternary match as + /// [`Pred::MatchFacetStrided`], over all 16 bytes of the field + /// (`ndarray::simd::ternary_match_strided16_to_mask`). Bytes `12..16` + /// participate; a zero `care` byte is don't-care, as above. + MatchFacet16Strided { + lane: u16, + pattern: [u8; 16], + care: [u8; 16], + }, } /// One instruction. Destinations are always [`Operand::Scratch`]; input planes @@ -1149,7 +1159,8 @@ impl Program { pred: Pred::EqU32Strided { .. } | Pred::NeU32Strided { .. } - | Pred::MatchFacetStrided { .. }, + | Pred::MatchFacetStrided { .. } + | Pred::MatchFacet16Strided { .. }, under, .. } => { diff --git a/crates/lance-graph-mask-risc/src/reference.rs b/crates/lance-graph-mask-risc/src/reference.rs index 394b04d29..999a0f083 100644 --- a/crates/lance-graph-mask-risc/src/reference.rs +++ b/crates/lance-graph-mask-risc/src/reference.rs @@ -86,7 +86,8 @@ fn pred_lane_and_kind(pred: Pred) -> Option<(u16, LaneKind)> { Pred::EqU32Via { fk, .. } => (fk, LaneKind::U32), Pred::EqU32Strided { lane, .. } | Pred::NeU32Strided { lane, .. } - | Pred::MatchFacetStrided { lane, .. } => (lane, LaneKind::Strided), + | Pred::MatchFacetStrided { lane, .. } + | Pred::MatchFacet16Strided { lane, .. } => (lane, LaneKind::Strided), Pred::Range { .. } => return None, }) } @@ -392,6 +393,9 @@ pub(crate) fn validate( Pred::MatchFacetStrided { lane, .. } => { check_strided(planes, lane, 12)?; } + Pred::MatchFacet16Strided { lane, .. } => { + check_strided(planes, lane, 16)?; + } _ => {} } if let Pred::Range { lo, hi } = pred { @@ -826,6 +830,20 @@ fn eval_pred(planes: &Planes<'_>, foreign: &Foreign<'_>, pred: Pred, row: usize) let field = strided_facet_at(lane); (0..12).all(|k| (field[k] ^ pattern[k]) & care[k] == 0) } + Pred::MatchFacet16Strided { + lane, + pattern, + care, + } => { + let field: [u8; 16] = match planes.lanes[usize::from(lane)] { + LaneRef::Strided(v) => { + let off = v.first_offset + row * v.stride; + v.bytes[off..off + 16].try_into().unwrap() + } + _ => [0u8; 16], + }; + (0..16).all(|k| (field[k] ^ pattern[k]) & care[k] == 0) + } } } diff --git a/crates/lance-graph-mask-risc/tests/strided.rs b/crates/lance-graph-mask-risc/tests/strided.rs index 123a45d1e..c74226437 100644 --- a/crates/lance-graph-mask-risc/tests/strided.rs +++ b/crates/lance-graph-mask-risc/tests/strided.rs @@ -971,3 +971,201 @@ fn ternlog_combines_three_strided_views_into_one_count() { } } } + +// ───────────────────────────────────────────────────────────────────────── +// MatchFacet16Strided: the 16-byte ternary match over a strided view. +// ───────────────────────────────────────────────────────────────────────── + +fn field16_at(f: &Fixture, row: usize) -> [u8; 16] { + let base = row * RECORD_STRIDE + 4; + f.bytes[base..base + 16].try_into().unwrap() +} + +fn expected_facet16_count( + f: &Fixture, + pattern: [u8; 16], + care: [u8; 16], + selected: impl Fn(usize) -> bool, +) -> usize { + (0..f.n) + .filter(|&r| selected(r)) + .filter(|&r| { + let b = field16_at(f, r); + (0..16).all(|k| (b[k] ^ pattern[k]) & care[k] == 0) + }) + .count() +} + +fn facet16_count_program(pattern: [u8; 16], care: [u8; 16]) -> Program { + Program::new( + vec![MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern, + care, + }, + under: None, + dst: 0, + }], + Terminal::Count { + mask: Operand::Scratch(0), + }, + ) +} + +/// FAILS IF: the executor disagrees with the oracle or with an independent +/// byte count, or bytes 12..16 do not take part in the match. +#[test] +fn match_facet16_strided_matches_oracle_and_reads_bytes_past_twelve() { + for n in ROWS { + let f = Fixture::new(n, 211); + let lanes = [LaneRef::Strided(f.facet_view())]; + let planes = Planes { + n_rows: n, + masks: &[], + lanes: &lanes, + }; + let pattern = if n == 0 { [0u8; 16] } else { field16_at(&f, 0) }; + + // Partial care across both halves of the 16 bytes. + let mut care = [0u8; 16]; + care[0] = 0x0F; + care[5] = 0xF0; + care[13] = 0x0F; + let want = expected_facet16_count(&f, pattern, care, |_| true); + let v = run_none( + &f, + &planes, + &facet16_count_program(pattern, care), + "facet16 partial", + ); + assert_eq!(v, Value::Count(want), "n={n}: partial care"); + + // Only byte 15 is inspected. The 12-byte predicate cannot see it. + let mut care15 = [0u8; 16]; + care15[15] = 0xFF; + let want15 = expected_facet16_count(&f, pattern, care15, |_| true); + let v15 = run_none( + &f, + &planes, + &facet16_count_program(pattern, care15), + "facet16 byte15", + ); + assert_eq!(v15, Value::Count(want15), "n={n}: byte 15 only"); + if n >= 70 { + assert!(want15 >= 1, "row 0 matches its own byte 15"); + assert!(want15 < n, "n={n}: byte 15 must actually filter rows"); + } + + // Silence twin: care == 0 matches every row. + let v_all = run_none( + &f, + &planes, + &facet16_count_program(pattern, [0u8; 16]), + "facet16 all-wild", + ); + assert_eq!( + v_all, + Value::Count(n), + "n={n}: care==0 must match every row" + ); + } +} + +/// FAILS IF: a gated 16-byte match is not `match & gate`. +#[test] +fn match_facet16_strided_under_a_gate() { + for n in ROWS { + let f = Fixture::new(n, 307); + let lanes = [LaneRef::Strided(f.facet_view())]; + let planes = Planes { + n_rows: n, + masks: &[], + lanes: &lanes, + }; + let pattern = if n == 0 { [0u8; 16] } else { field16_at(&f, 0) }; + let mut care = [0u8; 16]; + care[14] = 0x03; + let hi = (n / 2) as u32; + let p = Program::new( + vec![ + MaskOp::Pred { + pred: Pred::Range { lo: 0, hi }, + under: None, + dst: 0, + }, + MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern, + care, + }, + under: Some(Operand::Scratch(0)), + dst: 1, + }, + ], + Terminal::Count { + mask: Operand::Scratch(1), + }, + ); + let want = expected_facet16_count(&f, pattern, care, |r| r < n / 2); + let ungated = expected_facet16_count(&f, pattern, care, |_| true); + if n >= 70 { + assert!(want < ungated, "n={n}: the gate must remove some matches"); + } + let v = run_none(&f, &planes, &p, "facet16 gated"); + assert_eq!(v, Value::Count(want), "n={n}: gated count"); + } +} + +/// FAILS IF: a view wide enough for 12 bytes but not 16 is accepted by the +/// 16-byte predicate (or refused by the 12-byte one). +#[test] +fn match_facet16_strided_rejects_a_view_four_bytes_short() { + let n = 70; + let f = Fixture::new(n, 401); + let len = (n - 1) * RECORD_STRIDE + 4 + 12; + let short = &f.bytes[..len]; + let view = StridedRef { + bytes: short, + first_offset: 4, + stride: RECORD_STRIDE, + records: n, + }; + let lanes = [LaneRef::Strided(view)]; + let planes = Planes { + n_rows: n, + masks: &[], + lanes: &lanes, + }; + let p16 = facet16_count_program([0u8; 16], [0xFF; 16]); + let want = Err(ExecError::StridedOutOfBounds { + lane: 0, + need: len + 4, + have: len, + }); + let mut s = scratch(&p16, &f); + let got = execute_into(&p16, &planes, &Foreign::NONE, &mut s, Out::None); + let oracle = reference_execute_into(&p16, &planes, &Foreign::NONE, Out::None); + assert_eq!(got, want, "executor: 16-byte read past the view"); + assert_eq!(oracle, want, "oracle: 16-byte read past the view"); + + // Twin: the same view is long enough for the 12-byte predicate. + let p12 = Program::new( + vec![MaskOp::Pred { + pred: Pred::MatchFacetStrided { + lane: 0, + pattern: [0u8; 12], + care: [0u8; 12], + }, + under: None, + dst: 0, + }], + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + let mut s = scratch(&p12, &f); + let got12 = execute_into(&p12, &planes, &Foreign::NONE, &mut s, Out::None); + assert_eq!(got12, Ok(Value::Count(n))); +}