diff --git a/libs/@local/graph/atlas/src/lib.rs b/libs/@local/graph/atlas/src/lib.rs index 8e5ea01d567..8fab98f0e2f 100644 --- a/libs/@local/graph/atlas/src/lib.rs +++ b/libs/@local/graph/atlas/src/lib.rs @@ -185,3 +185,8 @@ pub(crate) mod progress; pub(crate) mod random; pub(crate) mod runs; pub(crate) mod salt; +#[expect( + dead_code, + reason = "the read API that consumes the serving layer lands above this PR in the stack" +)] +pub(crate) mod serve; diff --git a/libs/@local/graph/atlas/src/serve/codec.rs b/libs/@local/graph/atlas/src/serve/codec.rs new file mode 100644 index 00000000000..ed4c8139988 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/codec.rs @@ -0,0 +1,519 @@ +//! Keyed row-id obfuscation with exact decoding. +//! +//! Internal row ids are dense and assignment-ordered. Sending them verbatim exposes gaps between +//! visible rows and a lower bound on the allocated row count. [`RowCodec`] replaces that +//! representation with a keyed bijection of the full `u32` range, independent of the accepted row +//! count. Any `u32` is a well-formed wire value, although only the image of the accepted domain +//! decodes. +//! +//! A uniformly random permutation maps any fixed set of distinct rows to a uniformly distributed +//! subset of the same size. Indistinguishability from that ideal at the observed query volume is +//! this construction's design target, not an established bound for its 32-bit domain and round +//! function. The codec guarantees invertibility and claims no cryptographic confidentiality or +//! authentication guarantee. Repeated ids remain linkable, returned counts remain visible, and the +//! finite domain permits guessing and enumeration. A decoded row still requires a visibility check. +//! +//! # Model +//! +//! An eight-round balanced Feistel network permutes the `u32` range. Each state consists of halves +//! L, R ∈ [0, 2¹⁶). Round i ∈ [0, 8) maps (L, R) to (R, L ⊕ Fᵢ(R)), where ⊕ is bitwise XOR and Fᵢ +//! is the low 16 bits of SipHash-2-4 under the round's key. The input to `SipHash` is R as a +//! four-byte little-endian `u32`, including its zero high bytes. +//! +//! XOR with the same value is its own inverse. A round's output (A, B) recovers its input as (B ⊕ +//! Fᵢ(A), A), and decoding applies these inverse rounds in reverse key order. Therefore the network +//! is a bijection for every choice of round keys, independently of any pseudorandomness assumption. +//! +//! The permutation does not depend on the accepted row count, and appending rows to a generation +//! leaves every existing wire id unchanged. Encoding accepts exactly the rows in [0, 2³²): +//! [`WIRE_ROW_BOUND`] is the exclusive bound. Decoding applies the inverse network, checks that the +//! result fits the ID type and bounds-checks it against the accepted [`RowDomain`]. For a +//! zero-based ID type and accepted row count N ≤ [`WIRE_ROW_BOUND`], exactly the N wire values in +//! the image of [0, N) decode, and every other value answers [`None`]. +//! +//! Taking the universe per call lets one codec, derived when the generation opens, serve an +//! accepted row set that delta slot allocation grows past the fitted rows. +//! +//! # Keys +//! +//! Round keys derive from `HKDF-SHA256` over the server secret, salted by the generation identity +//! and expanded under a per-domain label, when a generation opens for serving. Equal `(secret, +//! generation, label)` give equal mappings. This is intentional: encoded row ids remain stable +//! across restarts. Wire ids can coincide across two generations by chance. These coincidences are +//! acceptable because wire ids identify rows only within their generation. +//! +//! The label returned by [`EncodableId::label`] identifies the row type. Node rows, the only type +//! exposed to external callers, use `atlas.wire.node.v1`. + +use core::{fmt, hash::Hasher as _, marker::PhantomData}; + +use hashql_core::id::Id; +use hkdf::Hkdf; +use sha2::Sha256; +use siphasher::sip::SipHasher24; +use zeroize::Zeroizing; + +use crate::{file::generation::GenerationId, identity::NodeRowId}; + +/// The Feistel round count one codec applies. +const ROUNDS: usize = 8; + +/// The Feistel half width. +const HALF_BITS: u32 = 16; + +/// The low-half mask. +const HALF_MASK: u32 = 0xFFFF; + +/// The exclusive bound of the rows that have a wire id. +/// +/// The permutation is over `u32`, and a row is encodable exactly when it lies in +/// `[0, WIRE_ROW_BOUND)`. Row `u32::MAX` is encodable, and a [`RowDomain`] of exactly this size is +/// the widest a codec serves. +pub(crate) const WIRE_ROW_BOUND: u64 = 1 << u32::BITS; + +/// The exclusive bound on accepted row ids. +/// +/// The universe contains the rows representable by `N` below the bound. The bound belongs to +/// one snapshot of a generation: the fitted rows set the base bound, and delta slot allocation +/// widens it. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) struct RowDomain(N); + +impl RowDomain +where + N: Id, +{ + /// Bounds the universe at `rows`. + #[must_use] + #[cfg(test)] // Tests bound domains at rows `from_length` cannot name on every target. + pub(crate) const fn new(rows: N) -> Self { + Self(rows) + } + + /// Returns a universe with `length` as the exclusive row bound. + /// + /// # Panics + /// + /// Panics if `length` is outside `N`'s representable range. Use [`TryFrom`] to validate + /// an untrusted count first. + #[must_use] + pub(crate) const fn from_length(length: usize) -> Self + where + N: [const] Id, + { + Self(N::from_usize(length)) + } + + /// Returns the exclusive row bound converted to `usize`. + /// + /// # Warning + /// + /// [`Id::as_usize`] controls this conversion. For the generated integer-backed IDs, a bound + /// wider than `usize` truncates. [`Self::bound`] retains the bound without this conversion. + #[must_use] + pub(crate) const fn size(self) -> usize + where + N: [const] Id, + { + self.0.as_usize() + } + + /// Returns the exclusive row bound, the row [`grow`](Self::grow) allocates next. + #[must_use] + pub(crate) const fn bound(self) -> N { + self.0 + } + + /// Allocates the row at the bound, widening the universe past it. + /// + /// Returns the widened universe with the allocated row, or [`None`] once the id space runs + /// out. + pub(crate) const fn grow(self) -> Option<(Self, N)> + where + N: [const] Id, + { + let next = self.0.next()?; + Some((Self(next), self.0)) + } + + /// Returns whether `row` lies inside the universe. + pub(crate) const fn contains(self, row: N) -> bool + where + N: [const] PartialOrd, + { + row < self.0 + } +} + +/// A row id as it crosses the wire. +/// +/// The value relates to an internal row id only through the owning generation's [`RowCodec`]: +/// [`RowCodec::encode`] produces egress values, and deserialization admits client-echoed values +/// whose meaning only [`RowCodec::decode`] assigns. An arbitrary `u32` is a well-formed value that +/// decodes to [`None`] outside the encoded image. Ordering compares wire values rather than +/// internal row positions. The [codec's obfuscation limits](crate::serve::codec) apply when that +/// order determines a result or breaks a tie. +#[derive(Debug, PartialEq, Eq, PartialOrd, Ord, Hash, schemars::JsonSchema)] +#[repr(transparent)] +#[schemars(transparent)] +pub(crate) struct EncodedRowId(u32, #[schemars(skip)] PhantomData); + +impl EncodedRowId { + /// Admits a value already in wire form, without an encoding pass. + /// + /// Accepts every `u32` without establishing an association with a row or generation. + pub(crate) const fn new_unchecked(value: u32) -> Self { + Self(value, PhantomData) + } + + /// Returns the wire value. + #[inline] + #[must_use] + pub(crate) const fn get(self) -> u32 { + self.0 + } +} + +impl Copy for EncodedRowId {} + +impl Clone for EncodedRowId { + fn clone(&self) -> Self { + *self + } +} + +impl serde::Serialize for EncodedRowId { + fn serialize(&self, serializer: S) -> Result + where + S: serde::Serializer, + { + self.0.serialize(serializer) + } +} + +impl<'de, I> serde::Deserialize<'de> for EncodedRowId { + fn deserialize(deserializer: D) -> Result + where + D: serde::Deserializer<'de>, + { + u32::deserialize(deserializer).map(|value| Self(value, PhantomData)) + } +} + +/// A row domain that crosses the wire under its own keyed mapping. +/// +/// The label is the HKDF expansion `info` of the domain's codec. Equal labels under one secret +/// and generation derive equal codecs. +pub(crate) trait EncodableId: Id { + /// Returns the HKDF expansion label of this row domain. + /// + /// By default, the label uses [`core::any::type_name`], whose output may change with the + /// compiler or the type's path. Changing the label can change wire ids, and individual rows may + /// retain the same value. + /// + /// # Implementation Note + /// + /// A stable wire domain must override this method with fixed bytes. Domains requiring separate + /// mappings must use distinct labels. + fn label() -> &'static [u8] { + core::any::type_name::().as_bytes() + } +} + +impl EncodableId for NodeRowId { + fn label() -> &'static [u8] { + b"atlas.wire.node.v1" + } +} + +/// The keyed mapping between one dense row domain and its wire ids. +/// +/// One codec serves one row domain of one generation. The underlying permutation bijects the `u32` +/// range for every key. Encoding accepts rows below [`WIRE_ROW_BOUND`]. Decoding inverts the +/// permutation and returns [`None`] for rows outside the ID type or the accepted [`RowDomain`]. +/// Both are pure: the mapping never changes for a held codec, and only the accepted bound moves as +/// slots allocate. [`fmt::Debug`] reports the round count without exposing the keys. +/// +/// These guarantees require the ID's numeric conversions to preserve represented values: conversion +/// to `u64` must be lossless, and conversion to `u32` must be lossless for encodable rows. The +/// codec does not enforce this property of the ID type. +/// +/// # Properties +/// +/// For every encodable row `r` contained in `domain`, `decode(encode(r), domain)` returns `Some(r)` +/// under the same codec. For every successfully decoded wire value `w`, encoding that row returns +/// `w`. +pub(crate) struct RowCodec { + /// The per-round SipHash-2-4 keys. + keys: Zeroizing<[[u8; 16]; ROUNDS]>, + _marker: PhantomData, +} + +impl RowCodec +where + I: Id, +{ + /// Derives the codec of one row domain from the server secret. + /// + /// The generation identity salts the extraction and [`EncodableId::label`] separates row + /// domains under one generation. Equal arguments derive equal codecs. + pub(crate) fn derive(secret: &[u8], generation: GenerationId) -> Self + where + I: EncodableId, + { + let salt = generation.digest().to_bytes(); + let label = I::label(); + + let mut keys = Zeroizing::new([[0_u8; 16]; ROUNDS]); + Hkdf::::new(Some(&salt), secret) + .expand(label, (*keys).as_flattened_mut()) + .expect("128 octets stay within HKDF-SHA256's expansion bound"); + + Self { + keys, + _marker: PhantomData, + } + } + + /// Tests whether `row`'s `u64` value is below [`WIRE_ROW_BOUND`]. + #[must_use] + pub(crate) fn encodes(row: I) -> bool { + row.as_u64() < WIRE_ROW_BOUND + } + + /// Encodes an internal row id as its wire id. + /// + /// # Panics + /// + /// Panics if `row` has no wire id under [`encodes`](Self::encodes). + pub(crate) fn encode(&self, row: I) -> EncodedRowId { + assert!( + Self::encodes(row), + "an encoded row must lie in [0, 2^32): the open bounds the fitted rows and node \ + allocation refuses the bound" + ); + + EncodedRowId::new_unchecked(self.permute(row.as_u32())) + } + + /// Decodes a wire value to a row accepted by `domain`. + /// + /// Returns [`None`] if the unpermuted value is outside the ID type's range or `domain`. + pub(crate) fn decode(&self, wire: EncodedRowId, domain: RowDomain) -> Option { + let row = I::try_from(self.unpermute(wire.get())).ok()?; + domain.contains(row).then_some(row) + } + + /// Applies the Feistel network once over the `u32` range. + fn permute(&self, mut state: u32) -> u32 { + for key in &*self.keys { + let left = state >> HALF_BITS; + let right = state & HALF_MASK; + state = (right << HALF_BITS) | (left ^ (round(key, right) & HALF_MASK)); + } + + state + } + + /// Applies the inverse network once over the `u32` range. + fn unpermute(&self, mut state: u32) -> u32 { + for key in self.keys.iter().rev() { + let right = state >> HALF_BITS; + let left = (state & HALF_MASK) ^ (round(key, right) & HALF_MASK); + state = (left << HALF_BITS) | right; + } + + state + } +} + +impl fmt::Debug for RowCodec { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("RowCodec") + .field("rounds", &ROUNDS) + .finish_non_exhaustive() + } +} + +/// Evaluates one round function: the low 32 bits of the keyed SipHash-2-4 of `half`. +#[expect( + clippy::cast_possible_truncation, + reason = "the caller masks to the half width; the narrowing keeps the used bits" +)] +#[expect( + clippy::little_endian_bytes, + reason = "the round function hashes one pinned byte order to preserve the mapping across hosts" +)] +fn round(key: &[u8; 16], half: u32) -> u32 { + let mut hasher = SipHasher24::new_with_key(key); + hasher.write(&half.to_le_bytes()); + hasher.finish() as u32 +} + +#[cfg(test)] +mod tests { + use hashql_core::id::{Id as _, newtype}; + + use super::{EncodableId, EncodedRowId, RowCodec, RowDomain, WIRE_ROW_BOUND}; + use crate::{file::generation::GenerationId, identity::NodeRowId, integrity::Sha256Digest}; + + newtype! { + /// A four-row ID domain for testing range rejection during decoding. + struct SmallRowId(u32 is 0..=3) + } + + impl EncodableId for SmallRowId {} + + /// The secret every vector below derives from. + const SECRET: &[u8] = b"atlas-codec-vector-secret"; + + /// The generation seeds of the two pinned tables. + /// + /// A generation identity is the SHA-256 digest of its seed. + const GENERATION_SEEDS: [&[u8]; 2] = [ + b"atlas-codec-vector-generation", + b"atlas-codec-vector-generation-2", + ]; + + /// `(row, wire)` pairs of [`SECRET`] under the first seed and `b"atlas.wire.node.v1"`. + /// + /// The values match an independent implementation of the module's construction: + /// HKDF-SHA256 per RFC 5869 with the generation digest as salt, the secret as keying material + /// and the label as info, eight 16-byte SipHash-2-4 keys in expansion order, and the + /// eight-round Feistel network of the module doc. That implementation reproduces RFC 5869 A.1 + /// and the SipHash-2-4 reference vector. A pair pins the label in force, the salt and info + /// roles, the key order, the round function's byte order and the Feistel structure at once. + /// The rows are the smallest three, both sides of the half-width boundary and the last + /// encodable row. + const VECTORS: [(u32, u32); 6] = [ + (0x0000_0000, 0x8BD5_91BE), + (0x0000_0001, 0x62C0_23C6), + (0x0000_0002, 0x2FAD_EF18), + (0x0000_FFFF, 0x4B4D_9B04), + (0x0001_0000, 0x44A3_0B72), + (0xFFFF_FFFF, 0x9EEB_C0B7), + ]; + + /// `(row, wire)` pairs of [`SECRET`] under the second seed and `b"atlas.wire.node.v1"`. + const VECTORS_SECOND_GENERATION: [(u32, u32); 6] = [ + (0x0000_0000, 0xAC60_3915), + (0x0000_0001, 0x79C4_1919), + (0x0000_0002, 0xB40B_B5B5), + (0x0000_FFFF, 0x046D_66FF), + (0x0001_0000, 0xFDC2_03AA), + (0xFFFF_FFFF, 0x4450_FDCA), + ]; + + /// Rows on both sides of the half-width boundary and at the ends of the wire range. + const SAMPLE_ROWS: [u32; 7] = [0, 1, 2, 0xFFFF, 0x1_0000, 0xFFFF_FFFE, 0xFFFF_FFFF]; + + /// Builds the generation identity of `seed`: the SHA-256 digest of its bytes. + fn generation(seed: &[u8]) -> GenerationId { + GenerationId::from_digest(Sha256Digest::of(seed)) + } + + /// Derives the node codec of [`SECRET`] under the generation seeded by `seed`. + fn codec(seed: &[u8]) -> RowCodec { + RowCodec::derive(SECRET, generation(seed)) + } + + #[test] + fn vectors_first_generation() { + assert_eq!(::label(), b"atlas.wire.node.v1"); + let codec = codec(GENERATION_SEEDS[0]); + for (row, wire) in VECTORS { + assert_eq!( + codec.encode(NodeRowId::from_u32(row)).get(), + wire, + "row {row:#010X} should encode to the pinned wire value" + ); + } + } + + /// Checks the second pinned table against its generation and the first table. + /// + /// Mappings under different generations may agree at a row by chance. These tables differ at + /// every recorded row. + #[test] + fn vectors_second_generation() { + let codec = codec(GENERATION_SEEDS[1]); + for ((row, wire), (first_row, first_wire)) in + VECTORS_SECOND_GENERATION.into_iter().zip(VECTORS) + { + assert_eq!(row, first_row, "the tables should record the same rows"); + assert_eq!( + codec.encode(NodeRowId::from_u32(row)).get(), + wire, + "row {row:#010X} should encode to the pinned wire value" + ); + assert_ne!( + wire, first_wire, + "row {row:#010X} should encode differently under the two generations" + ); + } + } + + #[test] + fn derive_equal_inputs() { + let left = codec(GENERATION_SEEDS[0]); + let right = codec(GENERATION_SEEDS[0]); + for row in SAMPLE_ROWS.map(NodeRowId::from_u32) { + assert_eq!(left.encode(row), right.encode(row)); + } + } + + #[test] + fn decode_inverts_within_domain() { + let codec = codec(GENERATION_SEEDS[0]); + let bound = NodeRowId::new(0x1_0001); + let domain = RowDomain(bound); + for row in [0, 1, 2, 0xFFFF, 0x1_0000].map(NodeRowId::from_u32) { + assert_eq!(codec.decode(codec.encode(row), domain), Some(row)); + } + for row in [ + bound, + NodeRowId::new(0x1_0002), + NodeRowId::from_u32(u32::MAX), + ] { + assert_eq!( + codec.decode(codec.encode(row), domain), + None, + "row {row} should lie outside the domain" + ); + } + } + + #[test] + fn decode_narrow_id() { + let codec = RowCodec::::derive(SECRET, generation(GENERATION_SEEDS[0])); + let domain = RowDomain(SmallRowId::MAX); + for value in 0..3 { + let row = SmallRowId::new(value); + assert_eq!(codec.decode(codec.encode(row), domain), Some(row)); + } + for value in [3, 4, u32::MAX] { + let wire = EncodedRowId::new_unchecked(codec.permute(value)); + assert_eq!(codec.decode(wire, domain), None); + } + } + + #[test] + fn decode_full_wire_domain() { + let codec = codec(GENERATION_SEEDS[0]); + let domain = RowDomain(NodeRowId::new(WIRE_ROW_BOUND)); + let last = NodeRowId::from_u32(u32::MAX); + assert!(domain.contains(last)); + assert_eq!(codec.decode(codec.encode(last), domain), Some(last)); + assert_eq!( + codec.decode(codec.encode(NodeRowId::MIN), domain), + Some(NodeRowId::MIN) + ); + } + + #[test] + #[should_panic(expected = "an encoded row must lie in [0, 2^32)")] + fn encode_beyond_wire_bound() { + let codec = codec(GENERATION_SEEDS[0]); + let _wire = codec.encode(NodeRowId::new(WIRE_ROW_BOUND)); + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/epoch.rs b/libs/@local/graph/atlas/src/serve/delta/epoch.rs new file mode 100644 index 00000000000..8031b55f8cb --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/epoch.rs @@ -0,0 +1,215 @@ +//! Immutable request captures of one published delta revision. + +use alloc::sync::Arc; +use core::{ops::Deref, ptr}; + +use arc_swap::Guard; + +use super::{ + Delta, DeltaReference, DeltaRevision, + importance::DeltaImportanceProvider, + layout::LayoutDelta, + overlay::{DeltaIdentityProvider, IdentityProviderResidual, NaiveIdentityProvider}, + topology::TopologyDelta, +}; +use crate::{ + dataset::auxiliary::{OwnedIcon, OwnedLegend}, + file::generation::GenerationId, + identity::{EdgeRowId, NodeRowId, OntologyRowId}, + postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, + serve::world::{ + NodeIndex, Ontology, layout::Layout, node_importance::ImportanceProvider, + topology::Topology, + }, +}; + +/// The captured delta, held either by an owned [`Arc`] or an [`arc_swap::ArcSwap`] [`Guard`]. +enum InternalEpoch { + Full(Arc), + Shared(Guard>), +} + +impl InternalEpoch { + /// Returns an owned handle to the same publication. + /// + /// A shared source remains unchanged and keeps its guard. The returned handle holds an [`Arc`] + /// instead of acquiring another [`Guard`]. + fn load_full(&self) -> Self { + match self { + Self::Full(arc) => Self::Full(Arc::clone(arc)), + Self::Shared(guard) => Self::Full(Arc::clone(guard)), + } + } +} + +impl Deref for InternalEpoch { + type Target = Delta; + + fn deref(&self) -> &Self::Target { + match self { + Self::Full(arc) => arc, + Self::Shared(guard) => guard, + } + } +} + +/// A request handle retaining one immutable delta publication. +/// +/// An [`ArcSwap`](arc_swap::ArcSwap) load selects one published [`Delta`], and later swaps do not +/// alter it. All component queries through this handle use the same world, lifetime tag and +/// revision. The handle remains readable after feed completion, runtime retirement or registry +/// closure, but its existence does not imply that the feed was healthy or caught up when +/// publication occurred. +pub(crate) struct Epoch { + delta: InternalEpoch, +} + +impl Epoch { + /// Returns the captured world's generation identifier. + pub(crate) fn generation(&self) -> GenerationId { + self.delta.world.generation().id() + } + + /// Returns the captured delta's identity and revision. + pub(crate) fn reference(&self) -> DeltaReference { + DeltaReference { + id: self.delta.id, + revision: self.delta.revision, + } + } + + /// Returns the captured revision. + pub(crate) fn revision(&self) -> DeltaRevision { + self.delta.revision + } + + /// Returns an independently owned epoch over the same captured delta. + /// + /// The returned epoch owns an [`Arc`] instead of another [`Guard`]. The source epoch remains + /// valid with unchanged ownership. A shared source retains its guard, while an owned source + /// retains its [`Arc`]. Forking does not block a concurrent publication. + pub(crate) fn fork(&self) -> Self { + Self { + delta: self.delta.load_full(), + } + } + + /// Returns whether a node has a visible placement at the captured revision. + pub(crate) fn contains_node(&self, node: NodeRowId) -> bool { + self.delta.world.layout.position(self, node).is_some() + } + + /// Borrows the captured layout changes after checking their world association. + /// + /// # Panics + /// + /// Panics if `layout` does not belong to the epoch's world. + pub(crate) fn layout(&self, layout: &Layout) -> &LayoutDelta { + assert!( + ptr::eq(layout, ptr::from_ref(&self.delta.world.layout)), + "layout must belong to the epoch's world", + ); + + &self.delta.layout + } + + /// Borrows captured node identities after checking their world association. + /// + /// # Panics + /// + /// Panics if `index` does not belong to the epoch's world. + pub(crate) fn nodes( + &self, + index: &NodeIndex, + ) -> &IdentityProviderResidual { + assert!( + ptr::eq(index, ptr::from_ref(&self.delta.world.layout.index)), + "index must belong to the epoch's world", + ); + + &self.delta.node + } + + /// Borrows captured edge identities after checking their world association. + /// + /// # Panics + /// + /// Panics if `topology` does not belong to the epoch's world. + pub(crate) fn edges( + &self, + topology: &Topology, + ) -> &IdentityProviderResidual { + assert!( + ptr::eq(topology, ptr::from_ref(&self.delta.world.topology)), + "topology must belong to the epoch's world", + ); + + &self.delta.edge + } + + /// Borrows captured ontology identities after checking their world association. + /// + /// # Panics + /// + /// Panics if `ontology` does not belong to the epoch's world. + pub(crate) fn ontology( + &self, + ontology: &Ontology, + ) -> &IdentityProviderResidual { + assert!( + ptr::eq(ontology, ptr::from_ref(&self.delta.world.ontology)), + "ontology must belong to the epoch's world", + ); + + &self.delta.ontology + } + + /// Returns priorities over this publication's allocated nodes. + /// + /// # Panics + /// + /// Panics if `layout` does not belong to the epoch's world. + pub(crate) fn importance<'epoch>( + &'epoch self, + layout: &Layout, + ) -> impl ImportanceProvider + use<'epoch> { + self.layout(layout); + DeltaImportanceProvider::from_parts( + &self.delta.world.layout, + DeltaIdentityProvider::from_parts( + &self.delta.node, + NaiveIdentityProvider::from_ref(&self.delta.world.layout.index.identity), + ), + ) + } + + /// Borrows the captured topology changes after checking their world association. + /// + /// # Panics + /// + /// Panics if `topology` does not belong to the epoch's world. + pub(crate) fn topology(&self, topology: &Topology) -> &TopologyDelta { + assert!( + ptr::eq(topology, ptr::from_ref(&self.delta.world.topology)), + "topology must belong to the epoch's world", + ); + &self.delta.topology + } +} + +impl core::fmt::Debug for Epoch { + fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + fmt.debug_struct("Epoch") + .field("generation", &self.generation()) + .field("revision", &self.revision()) + .finish_non_exhaustive() + } +} + +impl From>> for Epoch { + fn from(delta: Guard>) -> Self { + Self { + delta: InternalEpoch::Shared(delta), + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/history/mod.rs b/libs/@local/graph/atlas/src/serve/delta/history/mod.rs new file mode 100644 index 00000000000..418b5113bd8 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/history/mod.rs @@ -0,0 +1,160 @@ +//! Bounded visibility decisions without historical payload copies. +//! +//! [`History`] returns no decision once retention can no longer answer a query. [`Versioned`] keeps +//! birth separately so eviction cannot make a value appear before its creation. + +use super::DeltaRevision; + +#[cfg(test)] +mod tests; + +// Identity, layout and topology eviction tests share this retention capacity. +#[cfg(test)] +pub(super) use self::tests::CAPACITY; + +/// A visibility decision at one revision. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(super) enum EntryKind { + Live, + Withdrawn, +} + +/// The bitset backing [`History`]'s alive/withdrawn flags, one bit per retained revision. +type HistoryBitset = u8; +/// The number of revisions [`History`] retains, one bit of [`HistoryBitset`] per revision. +const HISTORY_SIZE: usize = HistoryBitset::BITS as usize; + +/// A fixed-capacity history of the eight newest visibility transitions, oldest first. +/// +/// Once full, [`Self::push`] evicts the oldest retained transition. The history cannot answer a +/// query older than every retained revision. [`Self::at`] returns [`None`]. Callers then use the +/// added value's live default or the inherited base provider. +#[derive(Debug, Copy, Clone)] +pub(super) struct History { + revisions: [DeltaRevision; HISTORY_SIZE], + alive: HistoryBitset, +} + +impl History { + /// Starts a history whose only transition is `kind` at `revision`. + pub(super) const fn new(kind: EntryKind, revision: DeltaRevision) -> Self { + Self { + revisions: [revision; HISTORY_SIZE], + alive: match kind { + EntryKind::Live => HistoryBitset::MAX, + EntryKind::Withdrawn => HistoryBitset::MIN, + }, + } + } + + /// Records a visibility transition, replacing one at the same revision. + /// + /// Returns whether visibility changed. Repeated states preserve retained history. + /// + /// # Panics + /// + /// Panics if `revision` precedes the latest recorded transition. + pub(super) fn push(&mut self, kind: EntryKind, revision: DeltaRevision) -> bool { + let latest = self.revisions[HISTORY_SIZE - 1]; + assert!( + revision >= latest, + "history revisions must be nondecreasing" + ); + + let prev = self.alive & 1; + let next = match kind { + EntryKind::Live => 1, + EntryKind::Withdrawn => 0, + }; + + if prev == next { + return false; + } + + if revision > latest { + self.revisions.shift_left([revision]); + self.alive <<= 1; + } else { + self.alive &= !1; + } + + self.alive |= next; + + next != prev + } + + /// Returns the current decision, the most recently pushed transition. + pub(super) const fn now(&self) -> EntryKind { + if self.alive & 1 != 0 { + EntryKind::Live + } else { + EntryKind::Withdrawn + } + } + + /// Returns the newest retained decision at or before `revision`. + /// + /// `None` lets the caller fall back to the initial state or the base provider. + pub(super) fn at(&self, revision: DeltaRevision) -> Option { + self.revisions + .iter() + .rev() + .position(|&history| history <= revision) + .map(|index| { + if (self.alive >> index) & 1 != 0 { + EntryKind::Live + } else { + EntryKind::Withdrawn + } + }) + } +} + +/// A value with permanent birth tracking and bounded visibility decisions. +#[derive(Debug, Copy, Clone)] +pub(super) struct Versioned { + data: T, + birth: DeltaRevision, + history: History, +} + +impl Versioned { + /// Wraps `data`, born live at `birth`. + pub(super) const fn new(data: T, birth: DeltaRevision) -> Self { + Self { + data, + birth, + history: History::new(EntryKind::Live, birth), + } + } + + /// Borrows the wrapped value regardless of visibility. + pub(super) const fn data(&self) -> &T { + &self.data + } + + /// Records a visibility transition without changing the birth revision. + /// + /// Returns whether visibility changed. + /// + /// # Panics + /// + /// Panics if `revision` precedes birth or the latest recorded transition. + pub(super) fn push(&mut self, kind: EntryKind, revision: DeltaRevision) -> bool { + self.history.push(kind, revision) + } + + /// Returns whether the value is visible, at the current state or at `revision`. + /// + /// A revision before birth is never live. A revision retention can no longer answer falls back + /// to live, matching an unplaced addition's default visibility. + pub(super) fn is_live(&self, revision: Option) -> bool { + revision.map_or_else( + || self.history.now() == EntryKind::Live, + |revision| { + revision >= self.birth + && self.history.at(revision).unwrap_or(EntryKind::Live) == EntryKind::Live + }, + ) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/history/tests.rs b/libs/@local/graph/atlas/src/serve/delta/history/tests.rs new file mode 100644 index 00000000000..8b094174af3 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/history/tests.rs @@ -0,0 +1,171 @@ +use super::{DeltaRevision, EntryKind, HISTORY_SIZE, History, Versioned}; + +/// The number of revisions [`History`] retains before falling back to the initial state. +pub(in crate::serve::delta) const CAPACITY: usize = HISTORY_SIZE; + +/// Defines visibility answers around the retained history boundaries. +/// +/// A revision before the initial entry has no answer. The initial entry holds through the entry +/// pushed after it, and the most recent push becomes the current and future answer. +#[test] +fn history_boundaries() { + let mut history = History::new(EntryKind::Withdrawn, DeltaRevision::new(3)); + assert_eq!(history.at(DeltaRevision::new(2)), None); + assert_eq!( + history.at(DeltaRevision::new(3)), + Some(EntryKind::Withdrawn) + ); + history.push(EntryKind::Live, DeltaRevision::new(7)); + assert_eq!(history.at(DeltaRevision::new(2)), None); + assert_eq!( + history.at(DeltaRevision::new(3)), + Some(EntryKind::Withdrawn) + ); + assert_eq!( + history.at(DeltaRevision::new(6)), + Some(EntryKind::Withdrawn) + ); + assert_eq!(history.at(DeltaRevision::new(7)), Some(EntryKind::Live)); + assert_eq!(history.at(DeltaRevision::new(8)), Some(EntryKind::Live)); + assert_eq!(history.now(), EntryKind::Live); +} + +/// Evicts only the oldest entries after pushes exceed retained capacity. +/// +/// The first `HISTORY_SIZE` revisions past the drop point still resolve to their recorded state. +/// The immediately preceding revision has no answer, while a revision after the last push falls +/// back to the current state. +#[test] +fn history_rollover() { + let mut history = History::new(EntryKind::Live, DeltaRevision::new(0)); + let end = u64::try_from(HISTORY_SIZE + 3).expect("should fit the history size"); + for revision in 1..=end { + let kind = if revision.is_multiple_of(2) { + EntryKind::Live + } else { + EntryKind::Withdrawn + }; + history.push(kind, DeltaRevision::new(revision)); + } + let first = end - u64::try_from(HISTORY_SIZE).expect("should fit the history size") + 1; + assert_eq!(history.at(DeltaRevision::new(first - 1)), None); + for revision in first..=end { + let kind = if revision.is_multiple_of(2) { + EntryKind::Live + } else { + EntryKind::Withdrawn + }; + assert_eq!(history.at(DeltaRevision::new(revision)), Some(kind)); + } + assert_eq!(history.at(DeltaRevision::new(end + 1)), Some(history.now())); +} + +/// Preserves retention capacity when a pushed entry changes nothing. +/// +/// Every redundant push reports no change. The sole real transition remains resolvable at and +/// after its revision. +#[test] +fn history_unchanged_retention() { + for initial in [EntryKind::Live, EntryKind::Withdrawn] { + let next = match initial { + EntryKind::Live => EntryKind::Withdrawn, + EntryKind::Withdrawn => EntryKind::Live, + }; + let mut history = History::new(initial, DeltaRevision::new(3)); + assert!(history.push(next, DeltaRevision::new(5))); + let end = 6 + u64::try_from(HISTORY_SIZE).expect("should fit the history size"); + for revision in 6..=end { + assert!(!history.push(next, DeltaRevision::new(revision))); + } + assert_eq!(history.at(DeltaRevision::new(2)), None); + assert_eq!(history.at(DeltaRevision::new(3)), Some(initial)); + assert_eq!(history.at(DeltaRevision::new(4)), Some(initial)); + assert_eq!(history.at(DeltaRevision::new(5)), Some(next)); + assert_eq!(history.at(DeltaRevision::new(end)), Some(next)); + assert_eq!(history.now(), next); + } +} + +/// Keeps only the last transition pushed at one revision. +/// +/// Earlier revisions are never affected. +#[test] +fn history_same_revision() { + let mut history = History::new(EntryKind::Live, DeltaRevision::new(3)); + for _ in 0..=HISTORY_SIZE { + history.push(EntryKind::Live, DeltaRevision::new(5)); + history.push(EntryKind::Withdrawn, DeltaRevision::new(5)); + } + assert_eq!(history.at(DeltaRevision::new(3)), Some(EntryKind::Live)); + assert_eq!( + history.at(DeltaRevision::new(5)), + Some(EntryKind::Withdrawn) + ); +} + +/// Panics when a pushed revision precedes the last recorded revision. +#[test] +#[should_panic(expected = "history revisions must be nondecreasing")] +fn history_backwards_revision() { + let mut history = History::new(EntryKind::Live, DeltaRevision::new(5)); + history.push(EntryKind::Withdrawn, DeltaRevision::new(4)); +} + +/// Accepts the maximum revision for both pushing and querying. +#[test] +fn history_max_revision() { + let mut history = History::new(EntryKind::Live, DeltaRevision::new(u64::MAX - 1)); + history.push(EntryKind::Withdrawn, DeltaRevision::new(u64::MAX)); + assert_eq!( + history.at(DeltaRevision::new(u64::MAX - 1)), + Some(EntryKind::Live) + ); + assert_eq!( + history.at(DeltaRevision::new(u64::MAX)), + Some(EntryKind::Withdrawn) + ); +} + +/// Preserves a value's birth revision after bounded-history eviction. +/// +/// Before birth, liveness is false. At and after birth, liveness comes from the (now-evicted) +/// initial entry, and the retained data remains unaffected by history rollover. +#[test] +fn versioned_birth_rollover() { + let mut value = Versioned::new("value", DeltaRevision::new(3)); + assert!(value.push(EntryKind::Withdrawn, DeltaRevision::new(4))); + let capacity = u64::try_from(HISTORY_SIZE).expect("should fit the history size"); + for offset in 1..=capacity { + let revision = 4 + 2 * offset; + assert!(value.push(EntryKind::Live, DeltaRevision::new(revision - 1))); + assert!(value.push(EntryKind::Withdrawn, DeltaRevision::new(revision))); + } + assert_eq!(value.history.at(DeltaRevision::new(4)), None); + assert!(!value.is_live(Some(DeltaRevision::new(2)))); + assert!(value.is_live(Some(DeltaRevision::new(3)))); + assert!(value.is_live(Some(DeltaRevision::new(4)))); + assert!(!value.is_live(None)); + assert_eq!(value.data(), &"value"); +} + +/// Reports whether a push changes the current visibility. +/// +/// Pushing the value's current visibility again reports no change. Pushing the opposite one does, +/// and liveness reflects the latest pushed kind. +#[test] +fn versioned_change_flag() { + let mut value = Versioned::new((), DeltaRevision::new(3)); + assert!(!value.push(EntryKind::Live, DeltaRevision::new(4))); + assert!(value.push(EntryKind::Withdrawn, DeltaRevision::new(5))); + assert!(!value.push(EntryKind::Withdrawn, DeltaRevision::new(6))); + assert!(value.push(EntryKind::Live, DeltaRevision::new(6))); + assert!(value.is_live(None)); +} + +/// Panics when a visibility decision predates the value's birth. +#[test] +#[should_panic(expected = "history revisions must be nondecreasing")] +fn versioned_prebirth_decision() { + let mut value = Versioned::new((), DeltaRevision::new(3)); + value.push(EntryKind::Withdrawn, DeltaRevision::new(2)); +} diff --git a/libs/@local/graph/atlas/src/serve/delta/id.rs b/libs/@local/graph/atlas/src/serve/delta/id.rs new file mode 100644 index 00000000000..5dd0d5c9a49 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/id.rs @@ -0,0 +1,26 @@ +//! Delta-local row offsets beyond a snapshotted base domain. + +use hashql_core::id::Id; + +use crate::serve::codec::RowDomain; + +hashql_core::id::newtype! { + /// A delta-local offset for a row past `origin`, the base domain snapshotted at construction. + pub(crate) struct DeltaRowId(u64) +} + +impl DeltaRowId { + /// Returns `index`'s offset past the base domain, or `None` for a base row. + /// + /// [`RowDomain::size`] must preserve `origin`'s bound for this classification to hold. + /// Generated integer-backed IDs truncate a bound wider than `usize` through that conversion. + pub(crate) const fn derive(origin: RowDomain, index: I) -> Option + where + I: [const] Id, + { + let index = index.as_u64(); + let offset = origin.size(); + + index.checked_sub(offset as u64).map(Self::new) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/importance/mod.rs b/libs/@local/graph/atlas/src/serve/delta/importance/mod.rs new file mode 100644 index 00000000000..20884bfaccc --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/importance/mod.rs @@ -0,0 +1,41 @@ +//! Priority lookup through existing rank and identity providers. +//! +//! Identity ordering extends the base priorities without allocating another per-node index. + +use super::overlay::VersionedIdentityProvider; +use crate::{ + identity::NodeRowId, + postgres::id::ArchivedEntityId, + serve::world::node_importance::{ImportanceProvider, NodePriority}, +}; + +#[cfg(test)] +mod tests; + +/// Base priorities extended with identity order for nodes the base ranks lack. +pub(super) struct DeltaImportanceProvider { + base: B, + identities: I, +} + +impl DeltaImportanceProvider { + /// Composes `base`'s ranks with `identities`' fallback order. + pub(super) const fn from_parts(base: B, identities: I) -> Self { + Self { base, identities } + } +} + +impl ImportanceProvider for DeltaImportanceProvider +where + B: ImportanceProvider, + I: VersionedIdentityProvider, +{ + /// Returns `base`'s rank priority, or `identities`' allocated-key order as a fallback. + fn provide_priority(&self, node: NodeRowId) -> Option { + self.base.provide_priority(node).or_else(|| { + self.identities + .provide_allocated_key_of(node) + .map(NodePriority::Identity) + }) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/importance/tests.rs b/libs/@local/graph/atlas/src/serve/delta/importance/tests.rs new file mode 100644 index 00000000000..059ca3ca925 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/importance/tests.rs @@ -0,0 +1,282 @@ +use alloc::sync::Arc; +use core::assert_matches; + +use arc_swap::Guard; +use hashql_core::id::Id as _; +use rand::{SeedableRng as _, rngs::StdRng}; +use uuid::Uuid; + +use super::DeltaImportanceProvider; +use crate::{ + dataset::auxiliary::{Label, Legend, OwnedLegend}, + identity::{NodeRowId, OntologyRowId}, + math::Vec2, + postgres::id::ArchivedEntityId, + salt::fit::prepare::IdentityProvider, + serve::{ + delta::{ + Delta, DeltaRevision, + epoch::Epoch, + overlay::{DeltaIdentityProvider, IdentityProviderResidual, NaiveIdentityProvider}, + }, + tests::fixture::{TamperFixture, secret}, + world::{ + World, + node_importance::{ImportanceProvider, NodePriority}, + }, + }, +}; + +/// Opens a fresh synthetic world under `name` and allocates a delta over it. +/// +/// # Panics +/// +/// Panics on failure during generation publication or world opening. +fn fixture(name: &str) -> (TamperFixture, Delta) { + let fixture = TamperFixture::publish(name); + let world = World::open(fixture.generation().clone(), &secret()) + .expect("should open the synthetic world"); + let delta = Delta::new(Arc::new(world), StdRng::seed_from_u64(23)) + .expect("should allocate a delta identity"); + (fixture, delta) +} + +/// Builds an entity id from a web and an entity half, for compact literal test identities. +fn entity(web: u128, id: u128) -> ArchivedEntityId { + ArchivedEntityId { + web_id: Uuid::from_u128(web).into(), + entity_uuid: Uuid::from_u128(id).into(), + } +} + +/// Builds a legend carrying `label` under a fixed ontology row. +fn legend(label: &str) -> OwnedLegend { + OwnedLegend::new(OntologyRowId::MIN, Label::new(label)) +} + +/// Snapshots `delta` into an epoch usable for read-side priority and position queries. +fn epoch(delta: &Delta) -> Epoch { + Epoch::from(Guard::from_inner(Arc::new(delta.clone()))) +} + +/// Orders identities independently of allocation order across webs. +#[test] +fn identity_allocation_order() { + let (_fixture, mut delta) = fixture("importance-identity-orders-over-allocation"); + let first_web = entity(2, 1); + let second_web = entity(1, 1); + + assert_eq!( + delta.update_node(first_web, legend("first"), Vec2::ZERO), + Some(true) + ); + assert_eq!( + delta.update_node(second_web, legend("second"), Vec2::ZERO), + Some(true) + ); + let first_row = delta + .node_row(first_web) + .expect("should allocate the first arrival row"); + let second_row = delta + .node_row(second_web) + .expect("should allocate the second arrival row"); + assert!( + first_row < second_row, + "should allocate rows in insertion order" + ); + + let epoch = epoch(&delta); + let first_priority = delta.world.layout.priority(&epoch, first_row); + let second_priority = delta.world.layout.priority(&epoch, second_row); + assert_eq!(first_priority, Some(NodePriority::Identity(first_web))); + assert_eq!(second_priority, Some(NodePriority::Identity(second_web))); + assert!( + second_priority < first_priority, + "should order identity priority by identity bytes rather than allocation order" + ); +} + +/// Returns no priority for unknown rows. +#[test] +fn priority_unknown() { + let (_fixture, delta) = fixture("importance-unknown-row-no-priority"); + let epoch = epoch(&delta); + assert_eq!( + delta.world.layout.priority(&epoch, NodeRowId::MAX), + None, + "should carry no priority for a row outside every known domain" + ); +} + +/// Preserves a rank through withdrawal and revival across captured publications. +#[test] +fn rank_withdrawal() { + let (_fixture, mut delta) = fixture("importance-fitted-priority-survives-withdrawal"); + let row = NodeRowId::new(0); + let fitted = delta + .world + .layout + .index + .identity + .key_of(row) + .expect("should resolve the fitted node's identity"); + + let published = epoch(&delta); + let before = delta + .world + .layout + .priority(&published, row) + .expect("should resolve a fitted row's rank priority"); + assert_matches!( + before, + NodePriority::Rank(_), + "should give a fitted row a rank priority" + ); + + delta.revision.increment_by(1); + assert!(delta.withdraw(fitted), "should withdraw the fitted node"); + let hidden = epoch(&delta); + assert_eq!(delta.world.layout.priority(&hidden, row), Some(before)); + assert_eq!( + delta.world.layout.position(&hidden, row), + None, + "should hide the withdrawn node's position while its priority persists" + ); + + delta.revision.increment_by(1); + assert_eq!( + delta.update_node(fitted, legend("revived"), Vec2::ZERO), + Some(true) + ); + let revived = epoch(&delta); + assert_eq!(delta.world.layout.priority(&revived, row), Some(before)); + assert_eq!( + delta.world.layout.priority(&published, row), + Some(before), + "should keep an old publication's priority reading unaffected by later revisions" + ); +} + +/// Preserves identity priority through withdrawal and revival without exposing future allocations. +#[test] +fn identity_withdrawal() { + let (_fixture, mut delta) = fixture("importance-arrival-priority-survives-withdrawal"); + let before_allocation = epoch(&delta); + let added = entity(3, 1); + assert_eq!( + delta.update_node(added, legend("first"), Vec2::ZERO), + Some(true) + ); + let row = delta + .node_row(added) + .expect("should allocate the arrival row"); + assert_eq!( + delta.world.layout.priority(&before_allocation, row), + None, + "should exclude rows allocated after a publication" + ); + + let published = epoch(&delta); + assert_eq!( + delta.world.layout.priority(&published, row), + Some(NodePriority::Identity(added)) + ); + + delta.revision.increment_by(1); + assert!(delta.withdraw(added), "should withdraw the arrival node"); + let hidden = epoch(&delta); + assert_eq!( + delta.world.layout.priority(&hidden, row), + Some(NodePriority::Identity(added)) + ); + assert_eq!(delta.world.layout.position(&hidden, row), None); + + delta.revision.increment_by(1); + assert_eq!( + delta.update_node(added, legend("revived"), Vec2::ZERO), + Some(true) + ); + let revived = epoch(&delta); + assert_eq!( + delta.world.layout.priority(&revived, row), + Some(NodePriority::Identity(added)) + ); + assert_eq!( + delta.world.layout.priority(&published, row), + Some(NodePriority::Identity(added)), + "should keep an old publication's priority reading unaffected by later revisions" + ); +} + +/// An empty priority source. +struct NoRank; + +impl ImportanceProvider for NoRank { + fn provide_priority(&self, _node: NodeRowId) -> Option { + None + } +} + +/// An empty identity source. +struct NoIdentities; + +impl IdentityProvider for NoIdentities { + fn count(&self) -> usize { + 0 + } + + fn key_of(&self, _row: NodeRowId) -> Option { + None + } + + fn row_of(&self, _key: ArchivedEntityId) -> Option { + None + } + + fn payload_of_key(&self, _key: ArchivedEntityId) -> Option<&Legend> { + None + } +} + +/// Retains a hidden lower row's identity across nested priority and identity providers. +#[test] +fn identity_nested() { + let base = NaiveIdentityProvider::from_ref(&NoIdentities); + let mut lower_data = IdentityProviderResidual::new(&base); + let key = entity(4, 1); + let (row, _) = lower_data + .insert(&base, DeltaRevision::new(1), key, legend("lower")) + .expect("should allocate the lower arrival row"); + assert!( + lower_data.withdraw(&base, DeltaRevision::new(2), key), + "should withdraw the lower identity" + ); + let lower = DeltaIdentityProvider::from_parts(&lower_data, &base); + let upper_data = IdentityProviderResidual::new(&lower); + let upper = DeltaIdentityProvider::from_parts(&upper_data, &lower); + assert_eq!(upper.key_of(row), None, "should hide the lower identity"); + + let from_identities = DeltaImportanceProvider::from_parts(NoRank, &upper); + assert_eq!( + from_identities.provide_priority(row), + Some(NodePriority::Identity(key)), + "should resolve the allocated key recursively without visibility filtering" + ); + + let lower_importance = DeltaImportanceProvider::from_parts(NoRank, &lower); + let upper_importance = DeltaImportanceProvider::from_parts(&lower_importance, &upper); + assert_eq!( + upper_importance.provide_priority(row), + Some(NodePriority::Identity(key)), + "should preserve the lower layer's identity priority" + ); +} + +/// Rejects an epoch from another world during priority lookup. +#[test] +#[should_panic(expected = "layout must belong to the epoch's world")] +fn priority_foreign_world() { + let (_fixture, delta) = fixture("importance-epoch-world"); + let (_other_fixture, other) = fixture("importance-foreign-world"); + other.world.layout.priority(&epoch(&delta), NodeRowId::MIN); +} diff --git a/libs/@local/graph/atlas/src/serve/delta/layout/mod.rs b/libs/@local/graph/atlas/src/serve/delta/layout/mod.rs new file mode 100644 index 00000000000..acfd47d5497 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/layout/mod.rs @@ -0,0 +1,155 @@ +//! Fixed node placements with revision-dependent visibility. +//! +//! Retaining the first placement keeps a node's coordinates stable until refitting. [`LayoutDelta`] +//! records visibility separately, allowing withdrawal and revival without reprojection. + +use hashql_core::{collections::FastHashMap, id::IdVec}; + +use super::{ + DeltaRevision, + history::{EntryKind, History, Versioned}, + id::DeltaRowId, +}; +use crate::{identity::NodeRowId, math::Vec2}; + +#[cfg(test)] +mod tests; + +pub(crate) mod provider; + +use self::provider::{DeltaLayoutProvider, VersionedLayoutProvider}; + +/// Added placements and inherited-node visibility decisions. +/// +/// Added nodes remain absent before their first placement's birth revision. Evicted decisions fall +/// back to the base provider for inherited nodes and to live for placed additions. +#[derive(Debug, Default)] +pub(crate) struct LayoutDelta { + positions: IdVec, Option>>, + history: FastHashMap, +} + +impl LayoutDelta { + /// Returns an added node's retained position or the base provider's position. + pub(super) fn recorded_position( + &self, + base: &(impl VersionedLayoutProvider + ?Sized), + node: NodeRowId, + ) -> Option { + if let Some(delta) = DeltaRowId::derive(base.provide_node_domain(), node) { + return self + .positions + .get(delta)? + .as_ref() + .map(|entry| *entry.data()); + } + base.provide_position(node) + } + + /// Returns a node's position at `revision`, or at the current state for `None`. + fn get( + &self, + base: &(impl VersionedLayoutProvider + ?Sized), + node: NodeRowId, + revision: Option, + ) -> Option { + if let Some(delta) = DeltaRowId::derive(base.provide_node_domain(), node) { + let entry = self.positions.get(delta)?.as_ref()?; + return entry.is_live(revision).then(|| *entry.data()); + } + + let decision = self.history.get(&node).and_then(|history| { + revision.map_or_else(|| Some(history.now()), |revision| history.at(revision)) + }); + + match decision { + Some(EntryKind::Withdrawn) => None, + Some(EntryKind::Live) | None => revision.map_or_else( + || base.provide_position(node), + |revision| base.provide_position_at(node, revision), + ), + } + } + + /// Activates a node, retaining its first successful placement. + /// + /// Inherited nodes keep the base provider's position. Returns whether a local visibility + /// decision changes or a placement is first recorded. + /// + /// # Panics + /// + /// Panics if `revision` precedes the node's latest recorded decision. + pub(crate) fn insert( + &mut self, + base: &(impl VersionedLayoutProvider + ?Sized), + node: NodeRowId, + position: Vec2, + revision: DeltaRevision, + ) -> bool { + let Some(delta) = DeltaRowId::derive(base.provide_node_domain(), node) else { + return self + .history + .get_mut(&node) + .is_some_and(|history| history.push(EntryKind::Live, revision)); + }; + + if let Some(entry) = self.positions.lookup_mut(delta) { + return entry.push(EntryKind::Live, revision); + } + + self.positions + .insert(delta, Versioned::new(position, revision)); + true + } + + /// Hides a placed node without discarding its coordinates. + /// + /// An unplaced addition remains unchanged. Returns whether the local visibility decision + /// changes. + /// + /// # Panics + /// + /// Panics if `revision` precedes the node's latest recorded decision. + pub(crate) fn withdraw( + &mut self, + base: &(impl VersionedLayoutProvider + ?Sized), + node: NodeRowId, + revision: DeltaRevision, + ) -> bool { + if let Some(delta) = DeltaRowId::derive(base.provide_node_domain(), node) { + let Some(entry) = self.positions.lookup_mut(delta) else { + return false; + }; + + return entry.push(EntryKind::Withdrawn, revision); + } + + let mut changed = false; + let history = self.history.entry(node).or_insert_with(|| { + changed = true; + History::new(EntryKind::Withdrawn, revision) + }); + + history.push(EntryKind::Withdrawn, revision) | changed + } + + /// Composes this delta with `base` into one layout provider. + pub(crate) const fn bind(&self, base: B) -> DeltaLayoutProvider<'_, B> { + DeltaLayoutProvider::from_parts(self, base) + } +} + +impl Clone for LayoutDelta { + fn clone(&self) -> Self { + Self { + positions: self.positions.clone(), + history: self.history.clone(), + } + } + + fn clone_from(&mut self, source: &Self) { + let Self { positions, history } = self; + positions.clone_from(&source.positions); + history.clone_from(&source.history); + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/layout/provider.rs b/libs/@local/graph/atlas/src/serve/delta/layout/provider.rs new file mode 100644 index 00000000000..f1bcfac15a8 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/layout/provider.rs @@ -0,0 +1,98 @@ +//! Layout queries over an ordinary or revisioned base. +//! +//! [`VersionedLayoutProvider`] preserves the base's historical visibility when deltas compose. + +use super::LayoutDelta; +use crate::{ + identity::NodeRowId, + math::Vec2, + serve::{codec::RowDomain, delta::DeltaRevision, world::layout::LayoutProvider}, +}; + +/// Position lookup at a retained revision, over the allocated row domain. +/// +/// The domain includes withdrawn and unplaced rows. Historical lookups apply the provider's +/// retention policy, including its fallback after eviction. +pub(crate) trait VersionedLayoutProvider: LayoutProvider { + /// Returns the snapshotted node row domain, including withdrawn and unplaced rows. + fn provide_node_domain(&self) -> RowDomain; + + /// Returns `node`'s position at `revision`, applying the retention policy on eviction. + fn provide_position_at(&self, node: NodeRowId, revision: DeltaRevision) -> Option; +} + +impl VersionedLayoutProvider for &T { + fn provide_node_domain(&self) -> RowDomain { + T::provide_node_domain(self) + } + + fn provide_position_at(&self, node: NodeRowId, revision: DeltaRevision) -> Option { + T::provide_position_at(self, node, revision) + } +} + +/// An ordinary layout whose coordinates are the same at every revision. +pub(crate) struct NaiveLayoutProvider(T); + +impl NaiveLayoutProvider { + /// Wraps `value`, an ordinary [`LayoutProvider`]. + pub(crate) const fn new(value: T) -> Self { + Self(value) + } +} + +impl LayoutProvider for NaiveLayoutProvider { + fn provide_node_count(&self) -> usize { + self.0.provide_node_count() + } + + fn provide_position(&self, node: NodeRowId) -> Option { + self.0.provide_position(node) + } +} + +impl VersionedLayoutProvider for NaiveLayoutProvider { + fn provide_node_domain(&self) -> RowDomain { + RowDomain::from_length(self.provide_node_count()) + } + + fn provide_position_at(&self, node: NodeRowId, _: DeltaRevision) -> Option { + self.provide_position(node) + } +} + +/// A layout with an additional layer of placements and visibility decisions. +/// +/// Inherited nodes retain the base provider's coordinates and revision-dependent visibility. +/// A local live decision removes a local withdrawal without overriding a withdrawal in the base. +pub(crate) struct DeltaLayoutProvider<'delta, B> { + data: &'delta LayoutDelta, + base: B, +} + +impl<'delta, B> DeltaLayoutProvider<'delta, B> { + /// Composes `data`'s local placements over `base`. + pub(crate) const fn from_parts(data: &'delta LayoutDelta, base: B) -> Self { + Self { data, base } + } +} + +impl LayoutProvider for DeltaLayoutProvider<'_, B> { + fn provide_node_count(&self) -> usize { + self.base.provide_node_count() + self.data.positions.len() + } + + fn provide_position(&self, node: NodeRowId) -> Option { + self.data.get(&self.base, node, None) + } +} + +impl VersionedLayoutProvider for DeltaLayoutProvider<'_, B> { + fn provide_node_domain(&self) -> RowDomain { + RowDomain::from_length(self.provide_node_count()) + } + + fn provide_position_at(&self, node: NodeRowId, revision: DeltaRevision) -> Option { + self.data.get(&self.base, node, Some(revision)) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/layout/tests.rs b/libs/@local/graph/atlas/src/serve/delta/layout/tests.rs new file mode 100644 index 00000000000..330ba6b77b0 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/layout/tests.rs @@ -0,0 +1,363 @@ +use hashql_core::id::IdVec; + +use super::{ + DeltaRowId, LayoutDelta, + provider::{NaiveLayoutProvider, VersionedLayoutProvider as _}, +}; +use crate::{ + identity::NodeRowId, + math::Vec2, + serve::{ + delta::{DeltaRevision, history::CAPACITY}, + world::layout::LayoutProvider, + }, +}; + +/// A fixed base layout with two fitted node positions. +struct Origin { + positions: IdVec, +} + +impl Origin { + /// Builds a fitted layout with two node positions at rows 0 and 1. + fn new() -> Self { + Self { + positions: [Vec2::new(2.0, 3.0), Vec2::new(-1.0, 4.0)] + .into_iter() + .collect(), + } + } +} + +impl LayoutProvider for Origin { + fn provide_node_count(&self) -> usize { + self.positions.len() + } + + fn provide_position(&self, node: NodeRowId) -> Option { + self.positions.get(node).copied() + } +} + +/// Delegates unchanged layout reads to the base provider. +/// +/// This holds at the current and base revisions, including for a row outside the base's domain. +#[test] +fn position_origin() { + let origin = Origin::new(); + let base = NaiveLayoutProvider::new(&origin); + let data = LayoutDelta::default(); + let provider = data.bind(&base); + assert_eq!(provider.provide_node_domain().size(), 2); + for node in [NodeRowId::new(0), NodeRowId::new(1), NodeRowId::new(99)] { + assert_eq!( + provider.provide_position(node), + origin.provide_position(node) + ); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(0)), + origin.provide_position(node) + ); + } +} + +/// Tracks withdrawal and revival of an inherited node by revision. +/// +/// Withdrawing an inherited node hides its position from the withdrawal revision onward while an +/// earlier revision still resolves through the base. A later re-insertion restores the base's +/// position from its own revision onward. +#[test] +fn inherited_withdrawal_revival() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut data = LayoutDelta::default(); + let node = NodeRowId::new(0); + assert!(data.withdraw(&base, node, DeltaRevision::new(2))); + { + let provider = data.bind(&base); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(1)), + base.provide_position(node) + ); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(2)), + None + ); + assert_eq!(provider.provide_position(node), None); + } + assert!(data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(4))); + let provider = data.bind(&base); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(3)), + None + ); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(4)), + base.provide_position(node) + ); + assert_eq!(provider.provide_position(node), base.provide_position(node)); +} + +/// Tracks an added node across birth, withdrawal and revival. +/// +/// An added node is absent before its birth revision and resolves to its placed position from +/// birth onward. It disappears again after a later withdrawal, and a re-insertion restores the +/// original position from the new revision onward. +#[test] +fn addition_birth_withdrawal_revival() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut data = LayoutDelta::default(); + let node = NodeRowId::new(2); + let position = Vec2::new(1.0, -2.0); + assert!(data.insert(&base, node, position, DeltaRevision::new(2))); + assert_eq!( + data.bind(&base) + .provide_position_at(node, DeltaRevision::new(1)), + None + ); + assert_eq!( + data.bind(&base) + .provide_position_at(node, DeltaRevision::new(2)), + Some(position) + ); + assert!(data.withdraw(&base, node, DeltaRevision::new(3))); + assert_eq!(data.bind(&base).provide_position(node), None); + assert!(data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(4))); + let provider = data.bind(&base); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(3)), + None + ); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(4)), + Some(position) + ); + assert_eq!(provider.provide_position(node), Some(position)); +} + +/// Keeps out-of-order row reservations independent. +/// +/// Reserving a higher row before a lower one extends the domain immediately and leaves the +/// reserved row unplaced. It does not disturb the visibility or later placement of the lower row. +#[test] +fn placement_out_of_order() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut data = LayoutDelta::default(); + let earlier = NodeRowId::new(2); + let later = NodeRowId::new(3); + let delta = DeltaRowId::derive(base.provide_node_domain(), later) + .expect("later should exceed the base's node domain"); + data.positions.fill_until(delta, || None); + assert_eq!(data.bind(&base).provide_node_count(), 4); + assert_eq!(data.bind(&base).provide_position(later), None); + assert!(!data.withdraw(&base, earlier, DeltaRevision::new(1))); + assert!(data.insert(&base, later, Vec2::splat(3.0), DeltaRevision::new(2))); + assert_eq!(data.bind(&base).provide_position(earlier), None); + assert!(data.insert(&base, earlier, Vec2::splat(2.0), DeltaRevision::new(3))); + let provider = data.bind(&base); + assert_eq!(provider.provide_position(earlier), Some(Vec2::splat(2.0))); + assert_eq!(provider.provide_position(later), Some(Vec2::splat(3.0))); + assert_eq!( + provider.provide_position_at(earlier, DeltaRevision::new(2)), + None + ); +} + +/// Leaves state unchanged when replay repeats a recorded decision. +/// +/// Both insertion and withdrawal replays report no change. +#[test] +fn decisions_replay_same_revision() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut data = LayoutDelta::default(); + for node in [NodeRowId::new(0), NodeRowId::new(2)] { + data.insert(&base, node, Vec2::splat(2.0), DeltaRevision::new(1)); + let position = data.bind(&base).provide_position(node); + assert!(!data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(1))); + assert!(data.withdraw(&base, node, DeltaRevision::new(2))); + assert!(!data.withdraw(&base, node, DeltaRevision::new(2))); + assert!(data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(2))); + assert_eq!( + data.bind(&base) + .provide_position_at(node, DeltaRevision::new(2)), + position + ); + assert!(!data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(3))); + assert_eq!(data.bind(&base).provide_position(node), position); + } +} + +/// Preserves the origin decision after retained history rolls over. +/// +/// A node's earliest visibility decision survives past the retained revision history's capacity, +/// for both an inherited and an added node, while a query before the added node's birth still +/// correctly resolves to absent. +#[test] +fn decisions_eviction() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut data = LayoutDelta::default(); + let inherited = NodeRowId::new(0); + let added = NodeRowId::new(2); + data.insert(&base, added, Vec2::ZERO, DeltaRevision::new(2)); + for node in [inherited, added] { + assert!(data.withdraw(&base, node, DeltaRevision::new(3))); + } + let capacity = u64::try_from(CAPACITY).expect("should fit the retention capacity"); + for offset in 1..=capacity { + let revision = 3 + 2 * offset; + for node in [inherited, added] { + assert!(data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(revision - 1))); + assert!(data.withdraw(&base, node, DeltaRevision::new(revision))); + } + } + let provider = data.bind(&base); + assert_eq!( + provider.provide_position_at(inherited, DeltaRevision::new(3)), + base.provide_position(inherited) + ); + assert_eq!( + provider.provide_position_at(added, DeltaRevision::new(3)), + Some(Vec2::ZERO) + ); + assert_eq!( + provider.provide_position_at(added, DeltaRevision::new(1)), + None + ); + assert_eq!(provider.provide_position(inherited), None); + assert_eq!(provider.provide_position(added), None); +} + +/// Resolves visibility through both layers of a composed delta. +/// +/// A lower-layer withdrawal hides an inherited node at the current revision but not before it, an +/// upper-layer re-insertion after a lower withdrawal still resolves absent, and an upper-only +/// addition is visible independent of the lower layer. +#[test] +fn nested_visibility() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut lower = LayoutDelta::default(); + let inherited = NodeRowId::new(0); + let lower_node = NodeRowId::new(2); + lower.insert(&base, lower_node, Vec2::splat(2.0), DeltaRevision::new(2)); + lower.withdraw(&base, inherited, DeltaRevision::new(3)); + lower.withdraw(&base, lower_node, DeltaRevision::new(4)); + let lower = lower.bind(&base); + let mut upper = LayoutDelta::default(); + upper.withdraw(&lower, lower_node, DeltaRevision::new(3)); + upper.insert(&lower, lower_node, Vec2::ZERO, DeltaRevision::new(5)); + upper.insert( + &lower, + NodeRowId::new(3), + Vec2::splat(3.0), + DeltaRevision::new(6), + ); + let provider = upper.bind(&lower); + assert_eq!(provider.provide_node_domain().size(), 4); + assert_eq!( + provider.provide_position_at(inherited, DeltaRevision::new(2)), + base.provide_position(inherited) + ); + assert_eq!(provider.provide_position(inherited), None); + assert_eq!( + provider.provide_position_at(lower_node, DeltaRevision::new(1)), + None + ); + assert_eq!( + provider.provide_position_at(lower_node, DeltaRevision::new(2)), + Some(Vec2::splat(2.0)) + ); + assert_eq!( + provider.provide_position_at(lower_node, DeltaRevision::new(3)), + None + ); + assert_eq!( + provider.provide_position_at(lower_node, DeltaRevision::new(5)), + None + ); + assert_eq!(provider.provide_position(lower_node), None); + assert_eq!( + provider.provide_position(NodeRowId::new(3)), + Some(Vec2::splat(3.0)) + ); +} + +/// Composes rolled-over visibility with a lower delta's history. +/// +/// An upper delta's decisions for a lower-layer node survive past the retention capacity, keeping +/// the base position resolvable at the lower layer's own revisions while the upper layer's current +/// state remains withdrawn. +#[test] +fn nested_eviction() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut lower = LayoutDelta::default(); + let node = NodeRowId::new(0); + lower.withdraw(&base, node, DeltaRevision::new(2)); + lower.insert(&base, node, Vec2::ZERO, DeltaRevision::new(4)); + let lower = lower.bind(&base); + let mut upper = LayoutDelta::default(); + assert!(upper.withdraw(&lower, node, DeltaRevision::new(5))); + let capacity = u64::try_from(CAPACITY).expect("should fit the retention capacity"); + for offset in 1..=capacity { + let revision = 5 + 2 * offset; + assert!(upper.insert(&lower, node, Vec2::ZERO, DeltaRevision::new(revision - 1))); + assert!(upper.withdraw(&lower, node, DeltaRevision::new(revision))); + } + let provider = upper.bind(&lower); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(3)), + None + ); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(4)), + base.provide_position(node) + ); + assert_eq!( + provider.provide_position_at(node, DeltaRevision::new(5)), + base.provide_position(node) + ); + assert_eq!(provider.provide_position(node), None); +} + +/// Replaces all target state while keeping the resulting clone independent. +/// +/// `clone_from` discards the target's history. Every resulting copy ignores later source changes, +/// including the one produced by `clone`. +#[test] +fn clone_replacement() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut source = LayoutDelta::default(); + let added = NodeRowId::new(2); + source.insert(&base, added, Vec2::splat(2.0), DeltaRevision::new(1)); + source.withdraw(&base, NodeRowId::new(0), DeltaRevision::new(2)); + let mut target = LayoutDelta::default(); + target.insert(&base, NodeRowId::new(5), Vec2::ZERO, DeltaRevision::new(3)); + target.withdraw(&base, NodeRowId::new(1), DeltaRevision::new(3)); + target.clone_from(&source); + source.withdraw(&base, added, DeltaRevision::new(4)); + let cloned = target.clone(); + for data in [&target, &cloned] { + let provider = data.bind(&base); + assert_eq!(provider.provide_node_count(), 3); + assert_eq!(provider.provide_position(added), Some(Vec2::splat(2.0))); + assert_eq!(provider.provide_position(NodeRowId::new(0)), None); + assert_eq!( + provider.provide_position(NodeRowId::new(1)), + base.provide_position(NodeRowId::new(1)) + ); + assert_eq!(provider.provide_position(NodeRowId::new(5)), None); + assert_eq!( + provider.provide_position_at(added, DeltaRevision::new(0)), + None + ); + } +} + +/// Panics when a withdrawal revision precedes the node's insertion. +#[test] +#[should_panic(expected = "history revisions must be nondecreasing")] +fn decisions_decreasing_revision() { + let base = NaiveLayoutProvider::new(Origin::new()); + let mut data = LayoutDelta::default(); + let node = NodeRowId::new(2); + data.insert(&base, node, Vec2::ZERO, DeltaRevision::new(2)); + data.withdraw(&base, node, DeltaRevision::new(1)); +} diff --git a/libs/@local/graph/atlas/src/serve/delta/mod.rs b/libs/@local/graph/atlas/src/serve/delta/mod.rs new file mode 100644 index 00000000000..341fb6c0b17 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/mod.rs @@ -0,0 +1,370 @@ +//! Mutable changes and immutable publications over one fitted generation. +//! +//! A delta lifetime starts at [`Delta::new`]. Clones, working revisions and published revisions +//! preserve its [`DeltaId`]. Opening another delta over the same world starts a distinct logical +//! lifetime with a newly sampled tag. + +#![expect( + clippy::empty_enums, + reason = "zerocopy derives generate uninhabited field-marker types" +)] + +pub(crate) mod epoch; +mod history; +mod id; +mod importance; +pub(crate) mod layout; +pub(crate) mod overlay; +pub(crate) mod topology; + +use alloc::sync::Arc; + +use hashql_core::id::Id as _; +use rand::TryCryptoRng; +use zerocopy::{NativeEndian, U64}; + +use self::{ + layout::{LayoutDelta, provider::NaiveLayoutProvider}, + overlay::{ + DeltaIdentityProvider, IdentityProviderResidual, NaiveIdentityProvider, + VersionedIdentityProvider as _, + }, + topology::{TopologyDelta, provider::NaiveTopologyProvider}, +}; +use super::{codec::RowCodec, world::World}; +use crate::{ + dataset::auxiliary::{OwnedIcon, OwnedLegend}, + identity::{EdgeRowId, NodeRowId, OntologyRowId}, + math::Vec2, + postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, +}; + +hashql_core::id::newtype! { + /// A sequence value for one delta's published states. + /// + /// Incrementing by one uses `usize` arithmetic. On a 32-bit target, the conversion discards the upper 32 bits before addition. + /// + /// # Panics + /// + /// With overflow checking, a unit increment panics when the converted value equals `usize::MAX`. + /// + /// # Warning + /// + /// Without overflow checking, a unit increment at `usize::MAX` wraps to zero. Reused revisions no longer distinguish publications. Recording a visibility decision before its history's latest transition panics. + #[id(unaligned)] + pub(crate) struct DeltaRevision(u64) +} + +/// A random tag used to distinguish one delta lifetime. +/// +/// A lifetime begins with [`Delta::new`] and includes its clones and published revisions. Cloning +/// and revision changes preserve the tag. Each initialization samples its own 64-bit value, +/// including when reusing the same [`World`]. Collisions are possible and are not detected. +/// Equality is a probabilistic lifetime identity rather than a global uniqueness guarantee. +#[derive( + Debug, + Copy, + Clone, + PartialEq, + Eq, + Hash, + zerocopy::IntoBytes, + zerocopy::Immutable, + zerocopy::Unaligned, + zerocopy::KnownLayout, + zerocopy::FromBytes, +)] +#[repr(transparent)] +pub(crate) struct DeltaId(U64); + +impl DeltaId { + /// Samples a lifetime tag from `rng`. + /// + /// # Errors + /// + /// Returns `rng`'s error if it fails to produce randomness. + fn new(mut rng: R) -> Result + where + R: TryCryptoRng, + { + let bytes = rng.try_next_u64()?; + Ok(Self(U64::new(bytes))) + } +} + +#[derive( + Debug, + Copy, + Clone, + PartialEq, + Eq, + zerocopy::IntoBytes, + zerocopy::Immutable, + zerocopy::Unaligned, + zerocopy::KnownLayout, + zerocopy::FromBytes, +)] +/// A delta lifetime tag and revision naming one published state. +/// +/// Revisions distinguish publications within a lifetime only before [`DeltaRevision`]'s narrowing +/// or wrap reuses a value. The lifetime component retains [`DeltaId`]'s probabilistic collision +/// semantics. +#[repr(C)] +pub(crate) struct DeltaReference { + /// The sampled tag shared by the lifetime's clones and revisions. + pub id: DeltaId, + /// The publication's revision within that lifetime. + pub revision: DeltaRevision, +} + +/// Mutable additions, withdrawals and relabellings past one generation's fitted base. +/// +/// The feed accumulates changes in a working value. Publication moves or clones that value behind +/// a [`DeltaReader`], after which retained epochs keep the published copy immutable. Cloning a +/// [`Delta`] preserves both its lifetime tag and current revision. Only [`Delta::new`] begins +/// another logical lifetime. +pub(crate) struct Delta { + world: Arc, + + id: DeltaId, + revision: DeltaRevision, + + ontology: IdentityProviderResidual, + + node: IdentityProviderResidual, + edge: IdentityProviderResidual, + + topology: TopologyDelta, + layout: LayoutDelta, +} + +impl Delta { + /// Starts an empty delta over `world` at [`DeltaRevision::MIN`]. + /// + /// The new logical lifetime samples a [`DeltaId`] from `rng` without checking whether the + /// value differs from an earlier lifetime. + /// + /// # Errors + /// + /// Returns `rng`'s error if it fails to produce randomness for the delta's identity. + pub(crate) fn new(world: Arc, rng: R) -> Result + where + R: TryCryptoRng, + { + let id = DeltaId::new(rng)?; + let revision = DeltaRevision::MIN; + + Ok(Self { + id, + revision, + ontology: IdentityProviderResidual::new(NaiveIdentityProvider::from_ref( + &world.ontology.identity, + )), + node: IdentityProviderResidual::new(NaiveIdentityProvider::from_ref( + &world.layout.index.identity, + )), + edge: IdentityProviderResidual::new(NaiveIdentityProvider::from_ref( + &world.topology.identity, + )), + topology: TopologyDelta::default(), + layout: LayoutDelta::default(), + world, + }) + } + + /// Resolves an allocated node row, including a withdrawn node. + fn node_row(&self, entity: ArchivedEntityId) -> Option { + DeltaIdentityProvider::from_parts( + &self.node, + NaiveIdentityProvider::from_ref(&self.world.layout.index.identity), + ) + .provide_allocated_row_of(entity) + } + + /// Returns a node's retained wire coordinates, including after withdrawal. + fn node_position(&self, entity: ArchivedEntityId) -> Option { + self.layout.recorded_position( + &NaiveLayoutProvider::new(&self.world.layout), + self.node_row(entity)?, + ) + } + + /// Hides an entity's identity and geometry without releasing its row. + /// + /// # Panics + /// + /// Panics if recording a withdrawal would precede the affected identity, placement or edge + /// history's latest transition. + fn withdraw(&mut self, entity: ArchivedEntityId) -> bool { + let node = self.node_row(entity); + let edge = DeltaIdentityProvider::from_parts( + &self.edge, + NaiveIdentityProvider::from_ref(&self.world.topology.identity), + ) + .provide_allocated_row_of(entity); + + let mut changed = self.node.withdraw( + NaiveIdentityProvider::from_ref(&self.world.layout.index.identity), + self.revision, + entity, + ); + changed |= self.edge.withdraw( + NaiveIdentityProvider::from_ref(&self.world.topology.identity), + self.revision, + entity, + ); + + if let Some(node) = node { + changed |= self.layout.withdraw( + &NaiveLayoutProvider::new(&self.world.layout), + node, + self.revision, + ); + } + + if let Some(edge) = edge { + changed |= self.topology.withdraw( + NaiveTopologyProvider::from_ref(&self.world.topology), + edge, + self.revision, + ); + } + changed + } + + /// Activates a node and replaces its legend, retaining its first placement. + /// + /// `position` uses the [wire frame](crate::salt::lod::stage::WIRE_FRAME). Returns whether state + /// changed, or `None` when no node row remains available: the id space has no row left, or the + /// next row lies at [`WIRE_ROW_BOUND`](super::codec::WIRE_ROW_BOUND) and has no wire id. An + /// entity that already holds a row, live or withdrawn, updates and revives on that row at every + /// capacity. + /// + /// # Panics + /// + /// Panics if the current revision precedes the node's latest recorded identity or placement + /// visibility transition. + fn update_node( + &mut self, + entity: ArchivedEntityId, + legend: OwnedLegend, + position: Vec2, + ) -> Option { + let base = NaiveIdentityProvider::from_ref(&self.world.layout.index.identity); + + // Allocation places a new row at the domain's bound, which must have a wire id. An + // allocated row already has one, and its entity passes whatever the bound. + let identities = DeltaIdentityProvider::from_parts(&self.node, base); + if identities.provide_allocated_row_of(entity).is_none() + && !RowCodec::::encodes(identities.provide_domain().bound()) + { + return None; + } + + let (node, mut changed) = self.node.insert(base, self.revision, entity, legend)?; + self.topology + .reserve_node(NaiveTopologyProvider::from_ref(&self.world.topology), node); + + changed |= self.layout.insert( + &NaiveLayoutProvider::new(&self.world.layout), + node, + position, + self.revision, + ); + Some(changed) + } + + /// Records an edge legend and activates a resolved endpoint pair. + /// + /// An edge retains its first bound pair. + /// + /// An unresolved pair reserves the edge row without binding it. Returns whether state changed, + /// or `None` when no edge row remains available. + /// + /// # Panics + /// + /// Panics if the current revision precedes the edge's latest recorded identity or topology + /// visibility transition. + fn update_edge( + &mut self, + entity: ArchivedEntityId, + legend: OwnedLegend, + endpoints: Option<[NodeRowId; 2]>, + ) -> Option { + let (edge, mut changed) = self.edge.insert( + NaiveIdentityProvider::from_ref(&self.world.topology.identity), + self.revision, + entity, + legend, + )?; + + let base = NaiveTopologyProvider::from_ref(&self.world.topology); + + self.topology.reserve_edge(base, edge); + if let Some(endpoints) = endpoints { + changed |= self.topology.insert(base, edge, endpoints, self.revision); + } + + Some(changed) + } + + /// Resolves an ontology row and replaces its icon. + /// + /// Returns the row and whether state changed, or `None` when no ontology row remains available. + /// + /// # Panics + /// + /// Panics if the current revision precedes the ontology row's latest recorded visibility + /// transition. + fn register_ontology( + &mut self, + ontology: ArchivedOntologyTypeUuid, + icon: OwnedIcon, + ) -> Option<(OntologyRowId, bool)> { + self.ontology.insert( + NaiveIdentityProvider::from_ref(&self.world.ontology.identity), + self.revision, + ontology, + icon, + ) + } +} + +impl Clone for Delta { + #[inline] + fn clone(&self) -> Self { + Self { + world: Arc::clone(&self.world), + id: self.id, + revision: self.revision, + ontology: self.ontology.clone(), + node: self.node.clone(), + edge: self.edge.clone(), + topology: self.topology.clone(), + layout: self.layout.clone(), + } + } + + #[inline] + fn clone_from(&mut self, source: &Self) { + let Self { + world, + id, + revision, + ontology, + node, + edge, + topology, + layout, + } = self; + + world.clone_from(&source.world); + id.clone_from(&source.id); + revision.clone_from(&source.revision); + ontology.clone_from(&source.ontology); + node.clone_from(&source.node); + edge.clone_from(&source.edge); + topology.clone_from(&source.topology); + layout.clone_from(&source.layout); + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/overlay/mod.rs b/libs/@local/graph/atlas/src/serve/delta/overlay/mod.rs new file mode 100644 index 00000000000..d33e420ec87 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/overlay/mod.rs @@ -0,0 +1,497 @@ +//! Stable row allocation with revision-dependent identity visibility. + +use core::{borrow::Borrow, hash::Hash}; + +use hashql_core::{collections::FastHashMap, id::IdVec}; + +use super::{ + DeltaRevision, + history::{EntryKind, History, Versioned}, + id::DeltaRowId, +}; +use crate::{ + file::identity::{Key, Row}, + salt::fit::prepare::IdentityProvider, + serve::codec::RowDomain, +}; + +#[cfg(test)] +mod tests; + +mod provider; + +pub(crate) use self::provider::VersionedIdentityProvider; + +/// An ordinary identity provider whose keys and payloads are the same at every revision. +#[repr(transparent)] +pub(crate) struct NaiveIdentityProvider(T); + +impl NaiveIdentityProvider { + /// Reinterprets a borrowed `T` as an ordinary identity provider, without copying it. + #[inline] + pub(crate) const fn from_ref(value: &T) -> &Self { + let ptr = &raw const *value; + // SAFETY: `Self` is transparent over `T` and adds no validity requirements. The cast + // preserves pointer metadata and the shared borrow's lifetime. + unsafe { &*(ptr as *const Self) } + } +} + +impl + ?Sized> IdentityProvider for NaiveIdentityProvider +where + R: Row, + K: Key, +{ + #[inline] + fn count(&self) -> usize { + self.0.count() + } + + #[inline] + fn key_of(&self, row: R) -> Option { + self.0.key_of(row) + } + + #[inline] + fn row_of(&self, key: K) -> Option { + self.0.row_of(key) + } + + #[inline] + fn payload_of_key(&self, key: K) -> Option<&::Payload> { + self.0.payload_of_key(key) + } + + #[inline] + fn payload_of_row(&self, row: R) -> Option<&K::Payload> { + self.0.payload_of_row(row) + } +} + +impl + ?Sized> VersionedIdentityProvider + for NaiveIdentityProvider +where + R: Row, + K: Key, +{ + #[inline] + fn provide_domain(&self) -> RowDomain { + RowDomain::from_length(self.count()) + } + + #[inline] + fn provide_allocated_row_of(&self, key: K) -> Option { + self.row_of(key) + } + + #[inline] + fn provide_allocated_key_of(&self, row: R) -> Option { + self.key_of(row) + } + + #[inline] + fn provide_key_of_at(&self, row: R, _: DeltaRevision) -> Option { + self.key_of(row) + } + + #[inline] + fn provide_row_of_at(&self, key: K, _: DeltaRevision) -> Option { + self.row_of(key) + } + + #[inline] + fn provide_payload_of_key_at(&self, key: K, _: DeltaRevision) -> Option<&K::Payload> { + self.payload_of_key(key) + } + + #[inline] + fn provide_payload_of_row_at(&self, row: R, _: DeltaRevision) -> Option<&K::Payload> { + self.payload_of_row(row) + } +} + +/// Identity rows added, withdrawn and relabelled over an immutable base provider. +/// +/// The residual snapshots the base's [`RowDomain`] at construction and allocates added rows past +/// it, in order. An added row's [`DeltaRowId`] is its offset from that snapshot, and every method +/// taking a `base` requires one reporting the snapshotted domain: the archive-backed +/// [`NaiveIdentityProvider`] over a read-only table reports a constant domain, and a provider +/// whose domain grows after the snapshot violates the requirement. +#[derive(Debug)] +pub(crate) struct IdentityProviderResidual { + domain: RowDomain, + + forward: FastHashMap, + inverse: IdVec, Versioned>, + payload: FastHashMap, + + history: FastHashMap, +} + +impl IdentityProviderResidual { + /// Starts an empty residual, snapshotting `base`'s domain. + #[inline] + pub(crate) fn new(base: &(impl VersionedIdentityProvider + ?Sized)) -> Self + where + K: Key>, + R: Row, + { + Self { + domain: base.provide_domain(), + + forward: FastHashMap::default(), + inverse: IdVec::default(), + payload: FastHashMap::default(), + history: FastHashMap::default(), + } + } + + /// Composes this residual's additions with `base` into one identity provider. + #[inline] + pub(crate) const fn bind(&self, base: B) -> DeltaIdentityProvider<'_, B, K, R, P> { + DeltaIdentityProvider::from_parts(self, base) + } + + /// Records local activation and a current payload without changing an allocated row. + /// + /// Returns the row and whether visibility or payload changed. `None` leaves the residual + /// unchanged when no row remains available. + /// + /// # Panics + /// + /// Panics if `revision` precedes the key's latest recorded visibility transition, or if `base` + /// reports a domain past an added row of this residual. + pub(crate) fn insert( + &mut self, + base: &impl VersionedIdentityProvider, + revision: DeltaRevision, + key: K, + payload: P, + ) -> Option<(R, bool)> + where + K: Key + Hash + Eq, + R: Row, + P: Borrow, + { + let (row, mut changed) = if let Some(&row) = self.forward.get(&key) { + let delta = DeltaRowId::derive(base.provide_domain(), row) + .expect("an added identity row must follow the fitted rows"); + (row, self.inverse[delta].push(EntryKind::Live, revision)) + } else if let Some(row) = base.provide_allocated_row_of(key) { + let changed = self + .history + .get_mut(&row) + .is_some_and(|history| history.push(EntryKind::Live, revision)); + (row, changed) + } else { + let (universe, row) = self.domain.grow()?; + self.domain = universe; + self.forward.insert(key, row); + self.inverse.push(Versioned::new(key, revision)); + (row, true) + }; + + let previous = self + .payload + .get(&key) + .map(Borrow::borrow) + .or_else(|| base.payload_of_key(key)); + if previous != Some(payload.borrow()) { + self.payload.insert(key, payload); + changed = true; + } + + Some((row, changed)) + } + + /// Hides a key while preserving its row and payload for revival. + /// + /// Returns whether visibility changed. + /// + /// # Panics + /// + /// Panics if `revision` precedes the key's latest recorded visibility transition, or if `base` + /// reports a domain past an added row of this residual. + pub(crate) fn withdraw( + &mut self, + base: &impl VersionedIdentityProvider, + revision: DeltaRevision, + key: K, + ) -> bool + where + K: Key + Hash + Eq, + R: Row, + { + if let Some(&row) = self.forward.get(&key) { + let delta = DeltaRowId::derive(base.provide_domain(), row) + .expect("an added identity row must follow the fitted rows"); + + self.inverse[delta].push(EntryKind::Withdrawn, revision) + } else if let Some(row) = base.provide_allocated_row_of(key) { + let mut has_changed = false; + + let entry = self.history.entry(row).or_insert_with(|| { + has_changed = true; + History::new(EntryKind::Withdrawn, revision) + }); + + has_changed | entry.push(EntryKind::Withdrawn, revision) + } else { + false + } + } +} + +impl Clone for IdentityProviderResidual { + #[inline] + fn clone(&self) -> Self { + Self { + domain: self.domain.clone(), + forward: self.forward.clone(), + inverse: self.inverse.clone(), + payload: self.payload.clone(), + history: self.history.clone(), + } + } + + #[inline] + fn clone_from(&mut self, source: &Self) { + let Self { + domain, + forward, + inverse, + payload, + history, + } = self; + + domain.clone_from(&source.domain); + forward.clone_from(&source.forward); + inverse.clone_from(&source.inverse); + payload.clone_from(&source.payload); + history.clone_from(&source.history); + } +} + +/// Identity queries composing a residual's additions with a base provider. +pub(crate) struct DeltaIdentityProvider<'ctx, B, K, R, P> { + data: &'ctx IdentityProviderResidual, + base: B, +} + +impl<'ctx, B, K, R, P> DeltaIdentityProvider<'ctx, B, K, R, P> { + /// Composes `data`'s additions over `base`. + pub(crate) const fn from_parts(data: &'ctx IdentityProviderResidual, base: B) -> Self { + Self { data, base } + } +} + +impl DeltaIdentityProvider<'_, B, K, R, ::Owned> +where + R: Row, + K: Key + Hash + Eq, + B: VersionedIdentityProvider, +{ + /// Returns whether `row` is visible, at the current state or at `revision`. + pub(crate) fn permits_row(&self, row: R, revision: Option) -> bool { + DeltaRowId::derive(self.base.provide_domain(), row).map_or_else( + || { + self.data + .history + .get(&row) + .and_then(|history| { + revision + .map_or_else(|| Some(history.now()), |revision| history.at(revision)) + }) + .is_none_or(|kind| kind == EntryKind::Live) + }, + |delta| { + self.data + .inverse + .get(delta) + .is_some_and(|entry| entry.is_live(revision)) + }, + ) + } + + /// Resolves `row`'s visible key at `revision`. + /// + /// `revision = None` selects the current state. The lookup returns `None` when the row is not + /// visible at the selected state. + fn lookup_key(&self, row: R, revision: Option) -> Option { + if !self.permits_row(row, revision) { + return None; + } + + DeltaRowId::derive(self.base.provide_domain(), row).map_or_else( + || { + revision.map_or_else( + || self.base.key_of(row), + |revision| self.base.provide_key_of_at(row, revision), + ) + }, + |delta| self.data.inverse.get(delta).map(|entry| *entry.data()), + ) + } + + /// Resolves the row visible for `key` at `revision`. + /// + /// `revision = None` selects the current state. The lookup returns `None` when the resolved row + /// is not visible at the selected state. + fn lookup_row(&self, key: K, revision: Option) -> Option { + let row = self.data.forward.get(&key).copied().or_else(|| { + revision.map_or_else( + || self.base.row_of(key), + |revision| self.base.provide_row_of_at(key, revision), + ) + })?; + + self.permits_row(row, revision).then_some(row) + } + + /// Resolves `key`'s current payload when visible at `revision`. + /// + /// `revision = None` selects the current state. A residual payload takes precedence over + /// `base`. The lookup returns `None` when the key's row is not visible at the selected state. + fn lookup_payload_of_key( + &self, + key: K, + revision: Option, + ) -> Option<&K::Payload> { + self.lookup_row(key, revision)?; + self.data.payload.get(&key).map(Borrow::borrow).or_else(|| { + revision.map_or_else( + || self.base.payload_of_key(key), + |revision| self.base.provide_payload_of_key_at(key, revision), + ) + }) + } + + /// Resolves `row`'s current payload when visible at `revision`. + /// + /// `revision = None` selects the current state. Re-deriving the row's key through a borrowed + /// copy of this provider preserves the payload's borrow. + fn lookup_payload_of_row( + &self, + row: R, + revision: Option, + ) -> Option<&K::Payload> { + DeltaIdentityProvider::from_parts(self.data, &self.base) + .into_lookup_payload_of_row(row, revision) + } +} + +impl<'payload, K, R, B> + DeltaIdentityProvider<'payload, &'payload B, K, R, ::Owned> +where + R: Row, + K: Key + Hash + Eq, + B: VersionedIdentityProvider + ?Sized, +{ + /// Borrows `row`'s current payload when it is visible at `revision`. + /// + /// The returned borrow outlives `self`. `revision = None` selects the current state. The row's + /// key follows [`DeltaIdentityProvider::lookup_key`]'s visibility rules. + fn into_lookup_payload_of_row( + self, + row: R, + revision: Option, + ) -> Option<&'payload K::Payload> { + let key = self.lookup_key(row, revision)?; + self.data.payload.get(&key).map(Borrow::borrow).or_else(|| { + revision.map_or_else( + || self.base.payload_of_row(row), + |revision| self.base.provide_payload_of_row_at(row, revision), + ) + }) + } + + /// Borrows `row`'s current payload beyond this provider when visible at `revision`. + pub(crate) fn into_payload_of_row_at( + self, + row: R, + revision: DeltaRevision, + ) -> Option<&'payload K::Payload> { + self.into_lookup_payload_of_row(row, Some(revision)) + } +} + +impl IdentityProvider + for DeltaIdentityProvider<'_, B, K, R, ::Owned> +where + R: Row, + K: Key + Hash + Eq, + B: VersionedIdentityProvider, +{ + /// Returns the allocated row-domain's size, including rows the residual has withdrawn. + #[inline] + fn count(&self) -> usize { + self.data.domain.size() + } + + /// Resolves `row`'s key at the current state, hiding a withdrawn row. + #[inline] + fn key_of(&self, row: R) -> Option { + self.lookup_key(row, None) + } + + /// Resolves `key`'s row at the current state, hiding a withdrawn key. + #[inline] + fn row_of(&self, key: K) -> Option { + self.lookup_row(key, None) + } + + /// Resolves `key`'s current payload, preferring the residual before `base`. + #[inline] + fn payload_of_key(&self, key: K) -> Option<&K::Payload> { + self.lookup_payload_of_key(key, None) + } + + /// Resolves `row`'s current payload, preferring the residual before `base`. + #[inline] + fn payload_of_row(&self, row: R) -> Option<&K::Payload> { + self.lookup_payload_of_row(row, None) + } +} + +impl VersionedIdentityProvider + for DeltaIdentityProvider<'_, B, K, R, ::Owned> +where + R: Row, + K: Key + Hash + Eq, + B: VersionedIdentityProvider, +{ + fn provide_domain(&self) -> RowDomain { + self.data.domain + } + + fn provide_allocated_row_of(&self, key: K) -> Option { + self.data + .forward + .get(&key) + .copied() + .or_else(|| self.base.provide_allocated_row_of(key)) + } + + fn provide_allocated_key_of(&self, row: R) -> Option { + DeltaRowId::derive(self.base.provide_domain(), row).map_or_else( + || self.base.provide_allocated_key_of(row), + |delta| self.data.inverse.get(delta).map(|entry| *entry.data()), + ) + } + + fn provide_key_of_at(&self, row: R, revision: DeltaRevision) -> Option { + self.lookup_key(row, Some(revision)) + } + + fn provide_row_of_at(&self, key: K, revision: DeltaRevision) -> Option { + self.lookup_row(key, Some(revision)) + } + + fn provide_payload_of_key_at(&self, key: K, revision: DeltaRevision) -> Option<&K::Payload> { + self.lookup_payload_of_key(key, Some(revision)) + } + + fn provide_payload_of_row_at(&self, row: R, revision: DeltaRevision) -> Option<&K::Payload> { + self.lookup_payload_of_row(row, Some(revision)) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/overlay/provider.rs b/libs/@local/graph/atlas/src/serve/delta/overlay/provider.rs new file mode 100644 index 00000000000..ddb20050a8b --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/overlay/provider.rs @@ -0,0 +1,76 @@ +//! Identity-provider contracts for current and retained revisions. + +use crate::{ + file::identity::{Key, Row}, + salt::fit::prepare::IdentityProvider, + serve::{codec::RowDomain, delta::DeltaRevision}, +}; + +/// Identity lookups with revision-dependent visibility and current payload values. +/// +/// Decisions apply inclusively at their revision. Added rows are absent before birth. Without a +/// retained decision, fitted rows use the base provider and added rows are live. +/// +/// # Warning +/// +/// Evicting decisions can change answers to older revision queries. +pub(crate) trait VersionedIdentityProvider: IdentityProvider +where + R: Row, + K: Key, +{ + /// Returns the snapshotted row domain, including withdrawn and unbound rows. + fn provide_domain(&self) -> RowDomain; + + /// Resolves an allocated row regardless of visibility. + fn provide_allocated_row_of(&self, key: K) -> Option; + + /// Resolves an allocated key regardless of visibility. + fn provide_allocated_key_of(&self, row: R) -> Option; + + /// Resolves `row`'s key at `revision`, applying the retention policy on eviction. + fn provide_key_of_at(&self, row: R, revision: DeltaRevision) -> Option; + + /// Resolves `key`'s row at `revision`, applying the retention policy on eviction. + fn provide_row_of_at(&self, key: K, revision: DeltaRevision) -> Option; + + /// Resolves `key`'s current payload when visible at `revision`, applying the eviction fallback. + fn provide_payload_of_key_at(&self, key: K, revision: DeltaRevision) -> Option<&K::Payload>; + + /// Resolves `row`'s current payload when visible at `revision`, applying the eviction fallback. + fn provide_payload_of_row_at(&self, row: R, revision: DeltaRevision) -> Option<&K::Payload>; +} + +impl + ?Sized> VersionedIdentityProvider for &T +where + R: Row, + K: Key, +{ + fn provide_domain(&self) -> RowDomain { + T::provide_domain(self) + } + + fn provide_allocated_row_of(&self, key: K) -> Option { + T::provide_allocated_row_of(self, key) + } + + fn provide_allocated_key_of(&self, row: R) -> Option { + T::provide_allocated_key_of(self, row) + } + + fn provide_key_of_at(&self, row: R, revision: DeltaRevision) -> Option { + T::provide_key_of_at(self, row, revision) + } + + fn provide_row_of_at(&self, key: K, revision: DeltaRevision) -> Option { + T::provide_row_of_at(self, key, revision) + } + + fn provide_payload_of_key_at(&self, key: K, revision: DeltaRevision) -> Option<&K::Payload> { + T::provide_payload_of_key_at(self, key, revision) + } + + fn provide_payload_of_row_at(&self, row: R, revision: DeltaRevision) -> Option<&K::Payload> { + T::provide_payload_of_row_at(self, row, revision) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/overlay/tests.rs b/libs/@local/graph/atlas/src/serve/delta/overlay/tests.rs new file mode 100644 index 00000000000..ee6932c2940 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/overlay/tests.rs @@ -0,0 +1,772 @@ +use core::cell::Cell; + +use hashql_core::id::Id as _; +use uuid::Uuid; + +use super::{ + DeltaIdentityProvider, DeltaRevision, DeltaRowId, EntryKind, History, IdentityProviderResidual, + NaiveIdentityProvider, Versioned, VersionedIdentityProvider, +}; +use crate::{ + dataset::auxiliary::{Icon, OwnedIcon}, + identity::OntologyRowId, + postgres::id::ArchivedOntologyTypeUuid, + salt::fit::prepare::IdentityProvider, + serve::{codec::RowDomain, delta::history::CAPACITY as HISTORY_SIZE}, +}; + +/// A single-row ontology identity provider that counts its payload reads. +struct Base { + key: ArchivedOntologyTypeUuid, + key_reads: Cell, + row_reads: Cell, +} + +impl Base { + /// Builds a single-row provider whose row 0 resolves to a fixed key with icon `"base"`. + const fn new() -> Self { + Self { + key: ArchivedOntologyTypeUuid::from(Uuid::from_u128(1)), + key_reads: Cell::new(0), + row_reads: Cell::new(0), + } + } +} + +impl IdentityProvider for Base { + fn count(&self) -> usize { + 1 + } + + fn key_of(&self, row: OntologyRowId) -> Option { + (row == OntologyRowId::new(0)).then_some(self.key) + } + + fn row_of(&self, key: ArchivedOntologyTypeUuid) -> Option { + (key == self.key).then_some(OntologyRowId::new(0)) + } + + /// Resolves the fixed key's payload, counting the read in `key_reads`. + fn payload_of_key(&self, key: ArchivedOntologyTypeUuid) -> Option<&Icon> { + self.key_reads.set(self.key_reads.get() + 1); + (key == self.key).then_some(Icon::new("base")) + } + + /// Resolves row 0's payload via [`Self::key_of`], counting the read in `row_reads`. + fn payload_of_row(&self, row: OntologyRowId) -> Option<&Icon> { + self.row_reads.set(self.row_reads.get() + 1); + self.key_of(row).map(|_| Icon::new("base")) + } +} + +/// An [`IdentityProviderResidual`] specialized to ontology rows and icons, for these tests. +type IconResidual = IdentityProviderResidual; + +/// Records an added key at the supplied `birth` revision. +/// +/// This bypasses [`IdentityProviderResidual::insert`]. +/// +/// # Panics +/// +/// Panics if the residual row domain has no free row. +fn add_arrival( + data: &mut IconResidual, + birth: DeltaRevision, +) -> (ArchivedOntologyTypeUuid, OntologyRowId) { + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let (universe, row) = data.domain.grow().expect("should have a free row"); + data.domain = universe; + data.forward.insert(key, row); + data.inverse.push(Versioned::new(key, birth)); + data.payload.insert(key, OwnedIcon::from("arrival")); + (key, row) +} + +/// Checks that `key` and `row` resolve bidirectionally to `payload` at `revision`. +/// +/// Both key- and row-based payload paths must agree. `None` requires every path to report absence. +/// +/// # Panics +/// +/// Panics if any lookup differs from `payload` or breaks the key-row correspondence. +#[track_caller] +fn assert_at( + provider: &(impl VersionedIdentityProvider + ?Sized), + key: ArchivedOntologyTypeUuid, + row: OntologyRowId, + revision: DeltaRevision, + payload: Option<&str>, +) { + assert_eq!( + provider.provide_key_of_at(row, revision), + payload.map(|_| key) + ); + assert_eq!( + provider.provide_row_of_at(key, revision), + payload.map(|_| row) + ); + assert_eq!( + provider + .provide_payload_of_key_at(key, revision) + .map(Icon::as_ref), + payload + ); + assert_eq!( + provider + .provide_payload_of_row_at(row, revision) + .map(Icon::as_ref), + payload + ); +} + +/// Checks that `key` and `row` currently resolve bidirectionally to `payload`. +/// +/// Both key- and row-based payload paths must agree. `None` requires every path to report absence. +/// +/// # Panics +/// +/// Panics if any lookup differs from `payload` or breaks the key-row correspondence. +#[track_caller] +fn assert_current( + provider: &(impl IdentityProvider + ?Sized), + key: ArchivedOntologyTypeUuid, + row: OntologyRowId, + payload: Option<&str>, +) { + assert_eq!(provider.key_of(row), payload.map(|_| key)); + assert_eq!(provider.row_of(key), payload.map(|_| row)); + assert_eq!(provider.payload_of_key(key).map(Icon::as_ref), payload); + assert_eq!(provider.payload_of_row(row).map(Icon::as_ref), payload); +} + +/// Exposes allocated keys and rejects unknown rows through both provider forms. +#[test] +fn allocated_base() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + assert_eq!( + VersionedIdentityProvider::provide_allocated_key_of(&origin, OntologyRowId::MIN), + Some(base.key), + "should forward the allocated key through a provider reference" + ); + assert_eq!( + origin.provide_allocated_key_of(OntologyRowId::MAX), + None, + "should reject an unallocated row" + ); +} + +/// Hides base and delta rows on withdrawal without erasing their allocated keys. +#[test] +fn allocated_hidden() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let (row, _) = data + .insert( + origin, + DeltaRevision::new(1), + key, + OwnedIcon::from("arrival"), + ) + .expect("should allocate a delta row"); + assert!( + data.withdraw(origin, DeltaRevision::new(2), base.key), + "should withdraw the base row" + ); + assert!( + data.withdraw(origin, DeltaRevision::new(2), key), + "should withdraw the delta row" + ); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + for (row, key) in [(OntologyRowId::MIN, base.key), (row, key)] { + assert_eq!(provider.key_of(row), None, "should hide the withdrawn row"); + assert_eq!( + provider.provide_allocated_key_of(row), + Some(key), + "should retain the withdrawn row's allocated key" + ); + } + assert_eq!(provider.provide_allocated_key_of(OntologyRowId::MAX), None); +} + +/// Records only a payload update for a fitted row, without allocating a delta row. +/// +/// A later insertion replaces the payload while leaving the residual domain unchanged. +#[test] +fn insert_fitted() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + let row = OntologyRowId::new(0); + assert_eq!( + data.insert( + origin, + DeltaRevision::new(1), + base.key, + OwnedIcon::from("base") + ), + Some((row, false)) + ); + assert!(data.payload.is_empty()); + assert_eq!( + data.insert( + origin, + DeltaRevision::new(2), + base.key, + OwnedIcon::from("updated") + ), + Some((row, true)) + ); + assert_eq!(data.domain.size(), 1); + assert!(data.inverse.is_empty()); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + assert_current(&provider, base.key, row, Some("updated")); +} + +/// Allocates one stable row for a newly inserted key. +/// +/// Repeating the same payload at a later revision reports no change. The row is absent before its +/// birth revision. +#[test] +fn insert_arrival() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(&base); + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let row = OntologyRowId::new(1); + assert_eq!( + data.insert( + &base, + DeltaRevision::new(4), + key, + OwnedIcon::from("arrival") + ), + Some((row, true)) + ); + assert_eq!( + data.insert( + &base, + DeltaRevision::new(5), + key, + OwnedIcon::from("arrival") + ), + Some((row, false)) + ); + assert_eq!(data.domain.size(), 2); + assert_eq!(data.inverse.len(), 1); + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_current(&provider, key, row, Some("arrival")); + assert_at(&provider, key, row, DeltaRevision::new(3), None); + assert_at(&provider, key, row, DeltaRevision::new(4), Some("arrival")); +} + +/// Reuses an added row across withdrawal and revival. +/// +/// An added key can be withdrawn and later revived on the same row with a new payload. The row +/// resolves through each visibility and payload transition at its matching revision. +#[test] +fn insert_revival() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(&base); + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let (row, _) = data + .insert( + &base, + DeltaRevision::new(4), + key, + OwnedIcon::from("arrival"), + ) + .expect("should allocate an arrival row"); + assert!(data.withdraw(&base, DeltaRevision::new(7), key)); + assert!(!data.withdraw(&base, DeltaRevision::new(8), key)); + assert_eq!( + data.insert(&base, DeltaRevision::new(9), key, OwnedIcon::from("latest")), + Some((row, true)) + ); + assert_eq!(data.domain.size(), 2); + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_current(&provider, key, row, Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(3), None); + assert_at(&provider, key, row, DeltaRevision::new(4), Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(7), None); + assert_at(&provider, key, row, DeltaRevision::new(9), Some("latest")); +} + +/// Reuses a fitted row across withdrawal and reinsertion. +/// +/// Withdrawing a fitted row hides it. A repeated withdrawal reports no change, and a later +/// re-insertion with a new payload revives the same row from its revival revision onward. +#[test] +fn withdraw_fitted() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + let row = OntologyRowId::new(0); + assert!(data.withdraw(origin, DeltaRevision::new(2), base.key)); + assert!(!data.withdraw(origin, DeltaRevision::new(3), base.key)); + assert_eq!( + data.insert( + origin, + DeltaRevision::new(4), + base.key, + OwnedIcon::from("base") + ), + Some((row, true)) + ); + assert_eq!(data.domain.size(), 1); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + assert_at(&provider, base.key, row, DeltaRevision::new(2), None); + assert_at( + &provider, + base.key, + row, + DeltaRevision::new(4), + Some("base"), + ); +} + +/// Leaves all residual state unchanged when row allocation has no capacity. +/// +/// The residual can still insert a fitted key already inside the domain. +#[test] +fn insert_exhausted() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + data.domain = RowDomain::new(OntologyRowId::MAX); + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + assert_eq!( + data.insert( + origin, + DeltaRevision::new(1), + key, + OwnedIcon::from("arrival") + ), + None + ); + assert_eq!(data.domain, RowDomain::new(OntologyRowId::MAX)); + assert!(data.forward.is_empty()); + assert!(data.inverse.is_empty()); + assert!(data.payload.is_empty()); + assert_eq!( + data.insert( + origin, + DeltaRevision::new(2), + base.key, + OwnedIcon::from("updated") + ), + Some((OntologyRowId::new(0), true)) + ); +} + +/// Retains one residual row through an added key's insertion, withdrawal and revival. +/// +/// The base remains constant, and each step writes to that row's delta index. +#[test] +fn withdraw_added_constant_base() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(&base); + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let (row, _) = data + .insert( + &base, + DeltaRevision::new(1), + key, + OwnedIcon::from("arrival"), + ) + .expect("should allocate an arrival row"); + assert_eq!(row, OntologyRowId::new(1)); + assert_eq!(data.forward.get(&key), Some(&row)); + assert!( + data.inverse[DeltaRowId::new(0)].is_live(None), + "should record the arrival live" + ); + + assert!( + data.withdraw(&base, DeltaRevision::new(2), key), + "should withdraw the added key" + ); + assert!( + !data.inverse[DeltaRowId::new(0)].is_live(None), + "should record the withdrawal on the added row" + ); + assert!( + !data.withdraw(&base, DeltaRevision::new(3), key), + "should leave a repeated withdrawal unchanged" + ); + + assert_eq!( + data.insert( + &base, + DeltaRevision::new(4), + key, + OwnedIcon::from("arrival") + ), + Some((row, true)), + "should revive the added key on its row" + ); + assert!(data.inverse[DeltaRowId::new(0)].is_live(None)); + assert_eq!(data.domain.size(), 2, "should allocate no second row"); + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_at(&provider, key, row, DeltaRevision::new(2), None); + assert_at(&provider, key, row, DeltaRevision::new(4), Some("arrival")); +} + +/// Panics on withdrawal after the base domain grows past an added residual row. +/// +/// The upper residual snapshots the lower's domain. After the lower allocates a row, the upper +/// allocates its own row at the stale bound. The lower now reports a domain past that row. +#[test] +#[should_panic(expected = "an added identity row must follow the fitted rows")] +fn withdraw_added_grown_base() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut lower_data = IdentityProviderResidual::new(&base); + let lower = DeltaIdentityProvider::from_parts(&lower_data, &base); + let mut upper_data = IdentityProviderResidual::new(&lower); + assert_eq!(upper_data.domain.size(), 1); + + let _arrival = add_arrival(&mut lower_data, DeltaRevision::new(1)); + let lower = DeltaIdentityProvider::from_parts(&lower_data, &base); + assert_eq!(lower.provide_domain().size(), 2); + + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(3)); + let (row, _) = upper_data + .insert(&lower, DeltaRevision::new(2), key, OwnedIcon::from("upper")) + .expect("should allocate at the stale bound"); + assert_eq!( + row, + OntologyRowId::new(1), + "should allocate at the snapshotted bound" + ); + + let _changed = upper_data.withdraw(&lower, DeltaRevision::new(3), key); +} + +/// Revives a base-withdrawn key in an upper delta without reallocating it. +/// +/// The row resolves through the upper payload at its birth revision while remaining absent from +/// the base layer's current state. +#[test] +fn insert_hidden_origin() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut lower_data = IdentityProviderResidual::new(&base); + let key = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let (row, _) = lower_data + .insert( + &base, + DeltaRevision::new(4), + key, + OwnedIcon::from("arrival"), + ) + .expect("should allocate the lower row"); + assert!(lower_data.withdraw(&base, DeltaRevision::new(7), key)); + let lower = DeltaIdentityProvider::from_parts(&lower_data, &base); + let mut upper_data = IdentityProviderResidual::new(&lower); + assert_eq!( + upper_data.insert( + &lower, + DeltaRevision::new(8), + key, + OwnedIcon::from("updated") + ), + Some((row, true)) + ); + assert_eq!(upper_data.domain.size(), 2); + assert!(upper_data.forward.is_empty()); + assert!(upper_data.inverse.is_empty()); + let upper = DeltaIdentityProvider::from_parts(&upper_data, &lower); + assert_eq!(upper.provide_allocated_row_of(key), Some(row)); + assert_current(&upper, key, row, None); + assert_at(&upper, key, row, DeltaRevision::new(4), Some("updated")); +} + +/// Delegates unmodified fitted-row payload reads to the base. +/// +/// The lookups count as base reads. An unallocated row resolves to no payload. +#[test] +fn payload_base() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let data = IdentityProviderResidual::new(origin); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + + assert_eq!( + provider.payload_of_key(base.key).map(Icon::as_ref), + Some("base") + ); + assert_eq!( + provider + .payload_of_row(OntologyRowId::new(0)) + .map(Icon::as_ref), + Some("base") + ); + assert_eq!(base.key_reads.get(), 1); + assert_eq!(base.row_reads.get(), 1); + assert_eq!(provider.payload_of_row(OntologyRowId::new(1)), None); +} + +/// Reads residual payload overrides without consulting the base. +/// +/// This covers fitted-key overrides and newly inserted keys. An unallocated row remains absent. +#[test] +fn payload_replacements() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + let arrival = ArchivedOntologyTypeUuid::from(Uuid::from_u128(2)); + let (domain, row) = data.domain.grow().expect("should have a free row"); + data.domain = domain; + data.inverse + .push(Versioned::new(arrival, DeltaRevision::new(4))); + data.forward.insert(arrival, row); + data.payload.insert(base.key, OwnedIcon::from("updated")); + data.payload.insert(arrival, OwnedIcon::from("arrival")); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + + assert_eq!(provider.count(), 2); + for (row, key, text) in [ + (OntologyRowId::new(0), base.key, "updated"), + (row, arrival, "arrival"), + ] { + assert_eq!(provider.key_of(row), Some(key)); + assert_eq!(provider.row_of(key), Some(row)); + assert_eq!(provider.payload_of_key(key).map(Icon::as_ref), Some(text)); + assert_eq!(provider.payload_of_row(row).map(Icon::as_ref), Some(text)); + } + assert_eq!(base.key_reads.get(), 0); + assert_eq!(base.row_reads.get(), 0); + assert_eq!(provider.key_of(OntologyRowId::new(2)), None); +} + +/// Tracks an added row's current payload through its visibility history. +/// +/// Birth initially exposes the first payload. After replacement during withdrawal, every live +/// revision returns the replacement, and revival makes the row currently visible again. +#[test] +fn arrival_lifetime() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(&base); + let (key, row) = add_arrival(&mut data, DeltaRevision::new(4)); + { + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_current(&provider, key, row, Some("arrival")); + assert_at(&provider, key, row, DeltaRevision::new(3), None); + assert_at(&provider, key, row, DeltaRevision::new(4), Some("arrival")); + } + + data.inverse[DeltaRowId::new(0)].push(EntryKind::Withdrawn, DeltaRevision::new(7)); + data.payload.insert(key, OwnedIcon::from("latest")); + { + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_eq!(provider.count(), 2); + assert_current(&provider, key, row, None); + assert_at(&provider, key, row, DeltaRevision::new(3), None); + assert_at(&provider, key, row, DeltaRevision::new(4), Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(6), Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(7), None); + } + + data.inverse[DeltaRowId::new(0)].push(EntryKind::Live, DeltaRevision::new(9)); + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_current(&provider, key, row, Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(3), None); + assert_at(&provider, key, row, DeltaRevision::new(4), Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(8), None); + assert_at(&provider, key, row, DeltaRevision::new(9), Some("latest")); + let unknown = ArchivedOntologyTypeUuid::from(Uuid::from_u128(99)); + assert_at( + &provider, + unknown, + OntologyRowId::new(2), + DeltaRevision::new(9), + None, + ); +} + +/// Tracks a fitted row's visibility independently of its payload. +/// +/// Direct history overrides hide the row while withdrawn. Every live revision exposes the current +/// payload, including those preceding its replacement. +#[test] +fn base_lifetime() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + let row = OntologyRowId::new(0); + data.history.insert( + row, + History::new(EntryKind::Withdrawn, DeltaRevision::new(5)), + ); + data.payload.insert(base.key, OwnedIcon::from("latest")); + { + let provider = DeltaIdentityProvider::from_parts(&data, origin); + assert_current(&provider, base.key, row, None); + assert_at( + &provider, + base.key, + row, + DeltaRevision::new(0), + Some("latest"), + ); + assert_at( + &provider, + base.key, + row, + DeltaRevision::new(4), + Some("latest"), + ); + assert_at(&provider, base.key, row, DeltaRevision::new(5), None); + } + data.history + .get_mut(&row) + .expect("should retain the row history") + .push(EntryKind::Live, DeltaRevision::new(9)); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + assert_current(&provider, base.key, row, Some("latest")); + assert_at(&provider, base.key, row, DeltaRevision::new(8), None); + assert_at( + &provider, + base.key, + row, + DeltaRevision::new(9), + Some("latest"), + ); +} + +/// Preserves a fitted row's origin after visibility history rolls over. +/// +/// Once retention evicts the initial withdrawal, its revision falls back to the visible base row +/// and the residual's current payload. The latest retained withdrawal still controls the current +/// state. +#[test] +fn base_origin_rollover() { + let base = Base::new(); + let origin = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(origin); + let row = OntologyRowId::new(0); + let mut history = History::new(EntryKind::Withdrawn, DeltaRevision::new(2)); + let capacity = u64::try_from(HISTORY_SIZE).expect("should fit the history size"); + let end = 2 + 2 * capacity; + for offset in 1..=capacity { + let revision = 2 + 2 * offset; + assert!(history.push(EntryKind::Live, DeltaRevision::new(revision - 1))); + assert!(history.push(EntryKind::Withdrawn, DeltaRevision::new(revision))); + } + assert_eq!(history.at(DeltaRevision::new(2)), None); + data.history.insert(row, history); + data.payload.insert(base.key, OwnedIcon::from("latest")); + let provider = DeltaIdentityProvider::from_parts(&data, origin); + assert_current(&provider, base.key, row, None); + assert_at( + &provider, + base.key, + row, + DeltaRevision::new(2), + Some("latest"), + ); + assert_at(&provider, base.key, row, DeltaRevision::new(end), None); +} + +/// Preserves an added row's birth after visibility history rolls over. +/// +/// Birth remains separate after retention evicts old visibility decisions. The initial +/// withdrawal's revision falls back to birth-based liveness, while the latest retained withdrawal +/// controls current state. +#[test] +fn arrival_origin_rollover() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut data = IdentityProviderResidual::new(&base); + let (key, row) = add_arrival(&mut data, DeltaRevision::new(1)); + let entry = &mut data.inverse[DeltaRowId::new(0)]; + assert!(entry.push(EntryKind::Withdrawn, DeltaRevision::new(2))); + let capacity = u64::try_from(HISTORY_SIZE).expect("should fit the history size"); + let end = 2 + 2 * capacity; + for offset in 1..=capacity { + let revision = 2 + 2 * offset; + assert!(entry.push(EntryKind::Live, DeltaRevision::new(revision - 1))); + assert!(entry.push(EntryKind::Withdrawn, DeltaRevision::new(revision))); + } + data.payload.insert(key, OwnedIcon::from("latest")); + let provider = DeltaIdentityProvider::from_parts(&data, &base); + assert_current(&provider, key, row, None); + assert_at(&provider, key, row, DeltaRevision::new(0), None); + assert_at(&provider, key, row, DeltaRevision::new(1), Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(2), Some("latest")); + assert_at(&provider, key, row, DeltaRevision::new(end), None); +} + +/// Prefers an upper delta's visibility at overlapping revisions. +/// +/// An upper layer's own visibility history for a lower-layer added row takes precedence over the +/// lower layer's history at revisions where both apply, while earlier revisions still resolve +/// through the lower layer alone. +#[test] +fn nested_origin_revision() { + let base = Base::new(); + let base = NaiveIdentityProvider::from_ref(&base); + let mut lower_data = IdentityProviderResidual::new(&base); + let (key, row) = add_arrival(&mut lower_data, DeltaRevision::new(4)); + lower_data.inverse[DeltaRowId::new(0)].push(EntryKind::Withdrawn, DeltaRevision::new(7)); + lower_data + .payload + .insert(key, OwnedIcon::from("lower latest")); + let lower = DeltaIdentityProvider::from_parts(&lower_data, &base); + let mut upper_data = IdentityProviderResidual::new(&lower); + { + let upper = DeltaIdentityProvider::from_parts(&upper_data, &lower); + assert_eq!(upper.provide_domain().size(), 2); + assert_current(&upper, key, row, None); + assert_at(&upper, key, row, DeltaRevision::new(3), None); + assert_at( + &upper, + key, + row, + DeltaRevision::new(4), + Some("lower latest"), + ); + assert_at(&upper, key, row, DeltaRevision::new(7), None); + } + upper_data.history.insert( + row, + History::new(EntryKind::Withdrawn, DeltaRevision::new(5)), + ); + upper_data + .payload + .insert(key, OwnedIcon::from("upper latest")); + let upper = DeltaIdentityProvider::from_parts(&upper_data, &lower); + assert_at( + &upper, + key, + row, + DeltaRevision::new(4), + Some("upper latest"), + ); + assert_at(&upper, key, row, DeltaRevision::new(5), None); +} + +/// Preserves [`NaiveIdentityProvider::from_ref`] behavior through borrowed unsized erasure. +/// +/// This holds at the domain's minimum and maximum revisions. +#[test] +fn naive_borrowed_unsized() { + let base = Base::new(); + let row = OntologyRowId::new(0); + let erased: &dyn IdentityProvider = &base; + let provider = NaiveIdentityProvider::from_ref(erased); + assert_at(provider, base.key, row, DeltaRevision::new(0), Some("base")); + assert_at( + provider, + base.key, + row, + DeltaRevision::new(u64::MAX), + Some("base"), + ); +} diff --git a/libs/@local/graph/atlas/src/serve/delta/topology/mod.rs b/libs/@local/graph/atlas/src/serve/delta/topology/mod.rs new file mode 100644 index 00000000000..1f14cafb7cf --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/topology/mod.rs @@ -0,0 +1,283 @@ +//! Edge topology with revision-dependent visibility and fixed endpoint pairs. +//! +//! Retaining an edge's first endpoint pair keeps its adjacency stable once bound. [`TopologyDelta`] +//! records visibility separately, allowing withdrawal and revival without losing the pair. + +use hashql_core::{collections::FastHashMap, id::IdVec}; + +use super::{ + DeltaRevision, + history::{EntryKind, History, Versioned}, + id::DeltaRowId, +}; +use crate::{ + identity::{EdgeRowId, NodeRowId}, + serve::codec::RowDomain, +}; + +#[cfg(test)] +mod tests; + +pub(crate) mod provider; + +use self::provider::{DeltaTopologyProvider, VersionedTopologyProvider}; + +/// Sorted, unique added-edge rows for each incident node. +#[derive(Debug, Default)] +struct AdjacencyRows { + extension: IdVec, Vec>, + patches: FastHashMap>, +} + +impl AdjacencyRows { + /// Records `edge` as incident to `node`, keeping each node's edge list sorted and unique. + fn insert(&mut self, origin: RowDomain, node: NodeRowId, edge: EdgeRowId) { + let edges = if let Some(delta) = DeltaRowId::derive(origin, node) { + self.extension.fill_until(delta, Vec::new) + } else { + self.patches.entry(node).or_default() + }; + + if let Err(index) = edges.binary_search(&edge) { + edges.insert(index, edge); + } + } + + /// Returns `node`'s recorded incident edges, in sorted order. + fn get( + &self, + origin: RowDomain, + node: NodeRowId, + ) -> impl Iterator { + let edges = DeltaRowId::derive(origin, node).map_or_else( + || self.patches.get(&node), + |delta| self.extension.get(delta), + ); + + edges.into_flat_iter().copied() + } +} + +impl Clone for AdjacencyRows { + fn clone(&self) -> Self { + Self { + extension: self.extension.clone(), + patches: self.patches.clone(), + } + } + + fn clone_from(&mut self, source: &Self) { + let Self { extension, patches } = self; + extension.clone_from(&source.extension); + patches.clone_from(&source.patches); + } +} + +/// Added-edge incidence for a topology's nodes, indexed by direction. +#[derive(Debug, Default)] +pub(crate) struct AdjacencyDelta { + incoming: AdjacencyRows, + outgoing: AdjacencyRows, +} + +impl Clone for AdjacencyDelta { + #[inline] + fn clone(&self) -> Self { + Self { + incoming: self.incoming.clone(), + outgoing: self.outgoing.clone(), + } + } + + #[inline] + fn clone_from(&mut self, source: &Self) { + let Self { incoming, outgoing } = self; + incoming.clone_from(&source.incoming); + outgoing.clone_from(&source.outgoing); + } +} + +/// Endpoint pairs and visibility decisions for a topology's edges. +#[derive(Debug, Default)] +pub(crate) struct EndpointDelta { + endpoints: IdVec, Option>>, + history: FastHashMap, +} + +impl EndpointDelta { + /// Returns `edge`'s endpoint pair at `revision`, or at the current state for `None`. + fn get( + &self, + base: &(impl VersionedTopologyProvider + ?Sized), + edge: EdgeRowId, + revision: Option, + ) -> Option<[NodeRowId; 2]> { + let origin = base.provide_edge_domain(); + if let Some(delta) = DeltaRowId::derive(origin, edge) { + let entry = self.endpoints.get(delta)?.as_ref()?; + return entry.is_live(revision).then(|| *entry.data()); + } + + let decision = self.history.get(&edge).and_then(|history| { + revision.map_or_else(|| Some(history.now()), |revision| history.at(revision)) + }); + + match decision { + Some(EntryKind::Withdrawn) => None, + Some(EntryKind::Live) | None => revision.map_or_else( + || base.provide_endpoints(edge), + |revision| base.provide_endpoints_at(edge, revision), + ), + } + } +} + +impl Clone for EndpointDelta { + #[inline] + fn clone(&self) -> Self { + Self { + endpoints: self.endpoints.clone(), + history: self.history.clone(), + } + } + + #[inline] + fn clone_from(&mut self, source: &Self) { + let Self { endpoints, history } = self; + endpoints.clone_from(&source.endpoints); + history.clone_from(&source.history); + } +} + +/// Revision-dependent edge visibility with fixed endpoint pairs. +/// +/// Added edges retain their first endpoint pair and remain absent before that pair's birth +/// revision. Evicted visibility decisions fall back to the origin for inherited edges and to live +/// for additions. +#[derive(Debug, Default)] +pub(crate) struct TopologyDelta { + adjacency: AdjacencyDelta, + endpoint: EndpointDelta, +} + +impl TopologyDelta { + /// Reserves an allocated node row, including nodes without incident edges. + pub(crate) fn reserve_node(&mut self, base: &impl VersionedTopologyProvider, node: NodeRowId) { + if let Some(delta) = DeltaRowId::derive(base.provide_node_domain(), node) { + self.adjacency + .incoming + .extension + .fill_until(delta, Vec::new); + + self.adjacency + .outgoing + .extension + .fill_until(delta, Vec::new); + } + } + + /// Reserves an allocated edge row before its endpoint pair is available. + pub(crate) fn reserve_edge(&mut self, base: &impl VersionedTopologyProvider, edge: EdgeRowId) { + if let Some(delta) = DeltaRowId::derive(base.provide_edge_domain(), edge) { + self.endpoint.endpoints.fill_until(delta, || None); + } + } + + /// Activates an edge, retaining its first endpoint pair. + /// + /// Returns whether the local visibility decision changes or an endpoint pair is first bound. An + /// inherited edge keeps its base provider's endpoints. + /// + /// # Panics + /// + /// Panics if `revision` precedes the edge's latest recorded decision. + pub(crate) fn insert( + &mut self, + base: &impl VersionedTopologyProvider, + edge: EdgeRowId, + endpoints: [NodeRowId; 2], + revision: DeltaRevision, + ) -> bool { + let origin = base.provide_edge_domain(); + let Some(delta) = DeltaRowId::derive(origin, edge) else { + return self + .endpoint + .history + .get_mut(&edge) + .is_some_and(|history| history.push(EntryKind::Live, revision)); + }; + + if let Some(entry) = self.endpoint.endpoints.lookup_mut(delta) { + return entry.push(EntryKind::Live, revision); + } + + self.endpoint + .endpoints + .insert(delta, Versioned::new(endpoints, revision)); + + let [source, target] = endpoints; + let nodes = base.provide_node_domain(); + self.adjacency.outgoing.insert(nodes, source, edge); + self.adjacency.incoming.insert(nodes, target, edge); + + true + } + + /// Withdraws a bound edge without discarding its endpoint pair. + /// + /// # Panics + /// + /// Panics if `revision` precedes the edge's latest recorded decision. + pub(crate) fn withdraw( + &mut self, + base: &impl VersionedTopologyProvider, + edge: EdgeRowId, + revision: DeltaRevision, + ) -> bool { + let origin = base.provide_edge_domain(); + if let Some(delta) = DeltaRowId::derive(origin, edge) { + let Some(Some(entry)) = self.endpoint.endpoints.get_mut(delta) else { + return false; + }; + + return entry.push(EntryKind::Withdrawn, revision); + } + + let mut changed = false; + let history = self.endpoint.history.entry(edge).or_insert_with(|| { + changed = true; + History::new(EntryKind::Withdrawn, revision) + }); + + history.push(EntryKind::Withdrawn, revision) | changed + } + + /// Composes this delta with `base` into one topology provider. + pub(crate) const fn bind<'delta, B: ?Sized>( + &'delta self, + base: &'delta B, + ) -> DeltaTopologyProvider<'delta, B> { + DeltaTopologyProvider::from_parts(self, base) + } +} + +impl Clone for TopologyDelta { + #[inline] + fn clone(&self) -> Self { + Self { + adjacency: self.adjacency.clone(), + endpoint: self.endpoint.clone(), + } + } + + #[inline] + fn clone_from(&mut self, source: &Self) { + let Self { + adjacency, + endpoint, + } = self; + + adjacency.clone_from(&source.adjacency); + endpoint.clone_from(&source.endpoint); + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/topology/provider.rs b/libs/@local/graph/atlas/src/serve/delta/topology/provider.rs new file mode 100644 index 00000000000..b41dc240631 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/topology/provider.rs @@ -0,0 +1,294 @@ +//! Topology queries over an ordinary or revisioned base. +//! +//! [`VersionedTopologyProvider`] preserves the base's historical visibility when deltas compose. + +use super::TopologyDelta; +use crate::{ + identity::{EdgeRowId, NodeRowId}, + serve::{codec::RowDomain, delta::DeltaRevision, world::topology::TopologyProvider}, +}; + +/// Endpoint and adjacency lookup at a retained revision. +/// +/// The row domains include withdrawn and unbound rows. Historical lookups apply the provider's +/// retention policy, including its fallback after eviction. +pub(crate) trait VersionedTopologyProvider: TopologyProvider { + /// Returns the snapshotted node row domain, including withdrawn and unbound rows. + fn provide_node_domain(&self) -> RowDomain; + + /// Returns the snapshotted edge row domain, including withdrawn and unbound rows. + fn provide_edge_domain(&self) -> RowDomain; + + /// Returns `edge`'s endpoint pair at `revision`, applying the retention policy on eviction. + fn provide_endpoints_at( + &self, + edge: EdgeRowId, + revision: DeltaRevision, + ) -> Option<[NodeRowId; 2]>; + + /// Returns `node`'s incoming edges at `revision`, applying the retention policy on eviction. + fn provide_incoming_at( + &self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator; + + /// Returns `node`'s outgoing edges at `revision`, applying the retention policy on eviction. + fn provide_outgoing_at( + &self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator; +} + +impl VersionedTopologyProvider for &T { + fn provide_node_domain(&self) -> RowDomain { + T::provide_node_domain(self) + } + + fn provide_edge_domain(&self) -> RowDomain { + T::provide_edge_domain(self) + } + + fn provide_endpoints_at( + &self, + edge: EdgeRowId, + revision: DeltaRevision, + ) -> Option<[NodeRowId; 2]> { + T::provide_endpoints_at(self, edge, revision) + } + + fn provide_incoming_at( + &self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator { + T::provide_incoming_at(self, node, revision) + } + + fn provide_outgoing_at( + &self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator { + T::provide_outgoing_at(self, node, revision) + } +} + +/// An ordinary topology with revision-independent queries. +#[repr(transparent)] +pub(crate) struct NaiveTopologyProvider(T); + +impl NaiveTopologyProvider { + /// Reinterprets a borrowed `T` as an ordinary topology provider, without copying it. + pub(crate) const fn from_ref(value: &T) -> &Self { + let ptr = &raw const *value; + // SAFETY: `Self` is transparent over `T` and adds no validity requirements. The cast + // preserves pointer metadata and the shared borrow's lifetime. + unsafe { &*(ptr as *const Self) } + } +} + +impl TopologyProvider for NaiveTopologyProvider { + fn provide_node_count(&self) -> usize { + self.0.provide_node_count() + } + + fn provide_edge_count(&self) -> usize { + self.0.provide_edge_count() + } + + fn provide_endpoints(&self, edge: EdgeRowId) -> Option<[NodeRowId; 2]> { + self.0.provide_endpoints(edge) + } + + fn provide_incoming(&self, node: NodeRowId) -> impl Iterator { + self.0.provide_incoming(node) + } + + fn provide_outgoing(&self, node: NodeRowId) -> impl Iterator { + self.0.provide_outgoing(node) + } +} + +impl VersionedTopologyProvider for NaiveTopologyProvider { + fn provide_node_domain(&self) -> RowDomain { + RowDomain::from_length(self.provide_node_count()) + } + + fn provide_edge_domain(&self) -> RowDomain { + RowDomain::from_length(self.provide_edge_count()) + } + + fn provide_endpoints_at(&self, edge: EdgeRowId, _: DeltaRevision) -> Option<[NodeRowId; 2]> { + self.provide_endpoints(edge) + } + + fn provide_incoming_at( + &self, + node: NodeRowId, + _: DeltaRevision, + ) -> impl Iterator { + self.provide_incoming(node) + } + + fn provide_outgoing_at( + &self, + node: NodeRowId, + _: DeltaRevision, + ) -> impl Iterator { + self.provide_outgoing(node) + } +} + +/// A topology with additional endpoint bindings and visibility decisions. +/// +/// Inherited edges retain the base provider's endpoints and revision-dependent visibility. A local +/// live decision removes a local withdrawal without overriding a withdrawal in the base. +pub(crate) struct DeltaTopologyProvider<'delta, B: ?Sized> { + data: &'delta TopologyDelta, + base: &'delta B, +} + +impl<'delta, B: ?Sized> DeltaTopologyProvider<'delta, B> { + /// Composes `data`'s local bindings over `base`. + pub(crate) const fn from_parts(data: &'delta TopologyDelta, base: &'delta B) -> Self { + Self { data, base } + } +} + +impl<'delta, B: VersionedTopologyProvider + ?Sized> DeltaTopologyProvider<'delta, B> { + /// Returns `node`'s incoming edges at `revision`, or at the current state for `None`. + fn lookup_incoming( + &self, + node: NodeRowId, + revision: Option, + ) -> impl Iterator + use<'delta, B> { + let &Self { data, base } = self; + + let current = revision.is_none().then(|| base.provide_incoming(node)); + let historical = revision + .into_iter() + .flat_map(move |revision| base.provide_incoming_at(node, revision)); + + current + .into_iter() + .flatten() + .chain(historical) + .chain( + data.adjacency + .incoming + .get(base.provide_node_domain(), node), + ) + .filter(move |&edge| data.endpoint.get(&base, edge, revision).is_some()) + } + + /// Returns `node`'s outgoing edges at `revision`, or at the current state for `None`. + fn lookup_outgoing( + &self, + node: NodeRowId, + revision: Option, + ) -> impl Iterator + use<'delta, B> { + let &Self { data, base } = self; + + let current = revision.is_none().then(|| base.provide_outgoing(node)); + let historical = revision + .into_iter() + .flat_map(move |revision| base.provide_outgoing_at(node, revision)); + + current + .into_iter() + .flatten() + .chain(historical) + .chain( + data.adjacency + .outgoing + .get(base.provide_node_domain(), node), + ) + .filter(move |&edge| data.endpoint.get(&base, edge, revision).is_some()) + } + + /// Returns incoming edges with a borrow of the stored data rather than this provider. + pub(crate) fn into_incoming_at( + self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator + use<'delta, B> { + self.lookup_incoming(node, Some(revision)) + } + + /// Returns outgoing edges with a borrow of the stored data rather than this provider. + pub(crate) fn into_outgoing_at( + self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator + use<'delta, B> { + self.lookup_outgoing(node, Some(revision)) + } +} + +impl TopologyProvider for DeltaTopologyProvider<'_, B> { + fn provide_node_count(&self) -> usize { + // An edge can reference the highest allocated node in only one direction. + self.base.provide_node_count() + + self + .data + .adjacency + .incoming + .extension + .len() + .max(self.data.adjacency.outgoing.extension.len()) + } + + fn provide_edge_count(&self) -> usize { + self.base.provide_edge_count() + self.data.endpoint.endpoints.len() + } + + fn provide_endpoints(&self, edge: EdgeRowId) -> Option<[NodeRowId; 2]> { + self.data.endpoint.get(self.base, edge, None) + } + + fn provide_incoming(&self, node: NodeRowId) -> impl Iterator { + self.lookup_incoming(node, None) + } + + fn provide_outgoing(&self, node: NodeRowId) -> impl Iterator { + self.lookup_outgoing(node, None) + } +} + +impl VersionedTopologyProvider + for DeltaTopologyProvider<'_, B> +{ + fn provide_node_domain(&self) -> RowDomain { + RowDomain::from_length(self.provide_node_count()) + } + + fn provide_edge_domain(&self) -> RowDomain { + RowDomain::from_length(self.provide_edge_count()) + } + + fn provide_endpoints_at( + &self, + edge: EdgeRowId, + revision: DeltaRevision, + ) -> Option<[NodeRowId; 2]> { + self.data.endpoint.get(&self.base, edge, Some(revision)) + } + + fn provide_incoming_at( + &self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator { + self.lookup_incoming(node, Some(revision)) + } + + fn provide_outgoing_at( + &self, + node: NodeRowId, + revision: DeltaRevision, + ) -> impl Iterator { + self.lookup_outgoing(node, Some(revision)) + } +} diff --git a/libs/@local/graph/atlas/src/serve/delta/topology/tests.rs b/libs/@local/graph/atlas/src/serve/delta/topology/tests.rs new file mode 100644 index 00000000000..4fd2531475f --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/delta/topology/tests.rs @@ -0,0 +1,465 @@ +use hashql_core::id::IdVec; + +use super::{ + TopologyDelta, + provider::{NaiveTopologyProvider, VersionedTopologyProvider}, +}; +use crate::{ + identity::{EdgeRowId, NodeRowId}, + serve::{ + delta::{DeltaRevision, history::CAPACITY}, + world::topology::TopologyProvider, + }, +}; + +/// A fixed base topology with a small fitted edge set. +struct Origin { + edges: IdVec, +} + +impl Origin { + /// Builds a fitted topology of three nodes and three edges, including a self-loop. + fn new() -> Self { + Self { + edges: [[0, 1], [0, 1], [1, 1]] + .map(|pair| pair.map(NodeRowId::new)) + .into_iter() + .collect(), + } + } +} + +impl TopologyProvider for Origin { + fn provide_node_count(&self) -> usize { + 3 + } + + fn provide_edge_count(&self) -> usize { + self.edges.len() + } + + fn provide_endpoints(&self, edge: EdgeRowId) -> Option<[NodeRowId; 2]> { + self.edges.get(edge).copied() + } + + fn provide_incoming(&self, node: NodeRowId) -> impl Iterator { + self.edges + .iter_enumerated() + .filter_map(move |(edge, &[_source, target])| (target == node).then_some(edge)) + } + + fn provide_outgoing(&self, node: NodeRowId) -> impl Iterator { + self.edges + .iter_enumerated() + .filter_map(move |(edge, &[source, _target])| (source == node).then_some(edge)) + } +} + +/// Checks a node's ordered incoming and outgoing edge rows at `revision`. +/// +/// # Panics +/// +/// Panics if either ordered edge sequence differs from its expected rows. +#[track_caller] +fn assert_edges( + provider: &impl VersionedTopologyProvider, + node: u64, + revision: u64, + incoming: &[u64], + outgoing: &[u64], +) { + let node = NodeRowId::new(node); + let revision = DeltaRevision::new(revision); + assert_eq!( + provider + .provide_incoming_at(node, revision) + .collect::>(), + incoming + .iter() + .copied() + .map(EdgeRowId::new) + .collect::>(), + ); + assert_eq!( + provider + .provide_outgoing_at(node, revision) + .collect::>(), + outgoing + .iter() + .copied() + .map(EdgeRowId::new) + .collect::>(), + ); +} + +/// Delegates unchanged topology reads to the base. +/// +/// An empty delta over the base forwards adjacency and endpoint lookups unchanged, including for +/// a node and an edge outside the base's domain. +#[test] +fn adjacency_origin() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let data = TopologyDelta::default(); + let provider = data.bind(base); + assert_edges(&provider, 0, 0, &[], &[0, 1]); + assert_edges(&provider, 1, 0, &[0, 1, 2], &[2]); + assert_edges(&provider, 2, 0, &[], &[]); + assert_edges(&provider, 99, 0, &[], &[]); + assert_eq!( + provider.provide_endpoints_at(EdgeRowId::new(3), DeltaRevision::new(0)), + None + ); +} + +/// Tracks inherited edge withdrawal and revival by revision. +/// +/// Withdrawing an inherited edge hides it and its adjacency from the withdrawal revision onward, +/// while an earlier revision still resolves through the base. Re-insertion with new endpoints +/// restores adjacency from the new revision onward. +#[test] +fn inherited_withdrawal_revival() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + let edge = EdgeRowId::new(0); + assert!(data.withdraw(base, edge, DeltaRevision::new(2))); + { + let provider = data.bind(base); + assert_edges(&provider, 0, 1, &[], &[0, 1]); + assert_edges(&provider, 0, 2, &[], &[1]); + assert_edges(&provider, 1, 2, &[1, 2], &[2]); + assert_eq!( + provider.provide_endpoints_at(edge, DeltaRevision::new(2)), + None + ); + } + assert!(data.insert(base, edge, [NodeRowId::new(2); 2], DeltaRevision::new(4))); + let provider = data.bind(base); + assert_edges(&provider, 0, 3, &[], &[1]); + assert_edges(&provider, 0, 4, &[], &[0, 1]); + assert_edges(&provider, 1, 4, &[0, 1, 2], &[2]); + assert_edges(&provider, 2, 4, &[], &[]); + assert_eq!( + provider.provide_endpoints_at(edge, DeltaRevision::new(4)), + base.provide_endpoints(edge) + ); +} + +/// Tracks an added edge across birth, withdrawal and rebinding. +/// +/// An added edge is absent from adjacency before its birth revision, appears at birth, disappears +/// again after a later withdrawal, and a re-insertion with different endpoints restores adjacency +/// under the new endpoints from that revision onward. +#[test] +fn addition_birth_withdrawal_revival() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + let edge = EdgeRowId::new(3); + let endpoints = [NodeRowId::new(0), NodeRowId::new(3)]; + assert!(data.insert(base, edge, endpoints, DeltaRevision::new(2))); + { + let provider = data.bind(base); + assert_edges(&provider, 0, 1, &[], &[0, 1]); + assert_edges(&provider, 3, 1, &[], &[]); + assert_edges(&provider, 0, 2, &[], &[0, 1, 3]); + assert_edges(&provider, 3, 2, &[3], &[]); + assert_eq!( + provider.provide_endpoints_at(edge, DeltaRevision::new(1)), + None + ); + assert_eq!( + provider.provide_endpoints_at(edge, DeltaRevision::new(2)), + Some(endpoints) + ); + } + assert!(data.withdraw(base, edge, DeltaRevision::new(4))); + assert!(data.insert(base, edge, [NodeRowId::new(1); 2], DeltaRevision::new(6))); + let provider = data.bind(base); + assert_edges(&provider, 0, 4, &[], &[0, 1]); + assert_edges(&provider, 3, 4, &[], &[]); + assert_eq!( + provider.provide_endpoints_at(edge, DeltaRevision::new(4)), + None + ); + assert_edges(&provider, 0, 6, &[], &[0, 1, 3]); + assert_edges(&provider, 1, 6, &[0, 1, 2], &[2]); + assert_edges(&provider, 3, 6, &[3], &[]); + assert_eq!( + provider.provide_endpoints_at(edge, DeltaRevision::new(6)), + Some(endpoints) + ); +} + +/// Keeps out-of-order edge bindings independent. +/// +/// Binding a higher-numbered edge before a lower one does not disturb the lower edge's later +/// binding or either edge's adjacency at its own revision. +#[test] +fn binding_out_of_order() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + let endpoints = [NodeRowId::new(0), NodeRowId::new(3)]; + assert!(data.insert(base, EdgeRowId::new(5), endpoints, DeltaRevision::new(2))); + assert!(!data.withdraw(base, EdgeRowId::new(3), DeltaRevision::new(3))); + assert!(data.insert(base, EdgeRowId::new(3), endpoints, DeltaRevision::new(4))); + let provider = data.bind(base); + assert_edges(&provider, 0, 2, &[], &[0, 1, 5]); + assert_edges(&provider, 3, 2, &[5], &[]); + assert_edges(&provider, 0, 4, &[], &[0, 1, 3, 5]); + assert_edges(&provider, 3, 4, &[3, 5], &[]); + assert_eq!( + provider.provide_endpoints_at(EdgeRowId::new(3), DeltaRevision::new(3)), + None + ); + assert_eq!( + provider.provide_endpoints_at(EdgeRowId::new(4), DeltaRevision::new(4)), + None + ); + assert_eq!( + provider.provide_endpoints_at(EdgeRowId::new(5), DeltaRevision::new(4)), + Some(endpoints) + ); +} + +/// Maintains both adjacency directions for a self-loop across replay. +/// +/// A self-loop edge appears in both a node's incoming and outgoing lists. Repeating its insertion +/// or withdrawal at the same revision reports no change, and it can be withdrawn and revived. +#[test] +fn self_loop_replay() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + let edge = EdgeRowId::new(3); + let pair = [NodeRowId::new(3); 2]; + assert!(data.insert(base, edge, pair, DeltaRevision::new(1))); + assert!(!data.insert(base, edge, pair, DeltaRevision::new(2))); + assert_edges(&data.bind(base), 3, 2, &[3], &[3]); + assert!(data.withdraw(base, edge, DeltaRevision::new(3))); + assert!(!data.withdraw(base, edge, DeltaRevision::new(4))); + assert_edges(&data.bind(base), 3, 4, &[], &[]); + assert!(data.insert(base, edge, pair, DeltaRevision::new(5))); + assert_edges(&data.bind(base), 3, 5, &[3], &[3]); +} + +/// Preserves inherited and added edge origins through history rollover. +/// +/// Both an inherited and an added edge's visibility decisions survive past the retention +/// capacity, keeping adjacency correct at revisions before, during, and after the rollover. +#[test] +fn history_origin_rollover() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + let added = EdgeRowId::new(3); + let inherited = EdgeRowId::new(0); + assert!(data.insert(base, added, [NodeRowId::new(3); 2], DeltaRevision::new(2))); + for edge in [inherited, added] { + assert!(data.withdraw(base, edge, DeltaRevision::new(3))); + } + let capacity = u64::try_from(CAPACITY).expect("should fit the retention capacity"); + let end = 3 + 2 * capacity; + for offset in 1..=capacity { + let revision = 3 + 2 * offset; + for edge in [inherited, added] { + assert!(data.insert( + base, + edge, + [NodeRowId::new(3); 2], + DeltaRevision::new(revision - 1) + )); + assert!(data.withdraw(base, edge, DeltaRevision::new(revision))); + } + } + let provider = data.bind(base); + assert_edges(&provider, 0, 3, &[], &[0, 1]); + assert_edges(&provider, 0, end, &[], &[1]); + assert_edges(&provider, 3, 1, &[], &[]); + assert_edges(&provider, 3, 3, &[3], &[3]); + assert_edges(&provider, 3, end, &[], &[]); + assert_eq!( + provider.provide_endpoints_at(added, DeltaRevision::new(1)), + None + ); +} + +/// Keeps only the last edge decision at one revision. +/// +/// Repeated insert-then-withdraw decisions never affect earlier adjacency. +#[test] +fn history_same_revision() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + let edge = EdgeRowId::new(3); + let pair = [NodeRowId::new(3); 2]; + assert!(data.insert(base, edge, pair, DeltaRevision::new(2))); + assert!(data.withdraw(base, edge, DeltaRevision::new(2))); + assert_edges(&data.bind(base), 3, 2, &[], &[]); + assert!(data.insert(base, edge, pair, DeltaRevision::new(2))); + assert_edges(&data.bind(base), 3, 1, &[], &[]); + assert_edges(&data.bind(base), 3, 2, &[3], &[3]); +} + +/// Replaces all target topology state from the source. +/// +/// `clone_from` discards the target's own prior changes. +#[test] +fn clone_from_replacement() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut source = TopologyDelta::default(); + source.insert( + base, + EdgeRowId::new(3), + [NodeRowId::new(3); 2], + DeltaRevision::new(2), + ); + source.withdraw(base, EdgeRowId::new(0), DeltaRevision::new(4)); + let mut target = source.clone(); + target.insert( + base, + EdgeRowId::new(5), + [NodeRowId::new(4); 2], + DeltaRevision::new(5), + ); + assert_edges(&target.bind(base), 4, 5, &[5], &[5]); + target.clone_from(&source); + let provider = target.bind(base); + assert_edges(&provider, 4, 5, &[], &[]); + assert_edges(&provider, 3, 2, &[3], &[3]); + assert_edges(&provider, 0, 3, &[], &[0, 1]); + assert_edges(&provider, 0, 4, &[], &[1]); + assert_eq!( + provider.provide_endpoints_at(EdgeRowId::new(5), DeltaRevision::new(5)), + None + ); +} + +/// Composes current and historical topology across nested deltas. +/// +/// A delta composed over another delta resolves node and edge counts, adjacency, and endpoint +/// lookups through both layers, at both the current state and a historical revision that predates +/// the upper layer's own changes. +#[test] +fn nested_provider_revisions() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut lower_data = TopologyDelta::default(); + let edge = EdgeRowId::new(3); + let pair = [NodeRowId::new(3); 2]; + lower_data.insert(base, edge, pair, DeltaRevision::new(2)); + lower_data.withdraw(base, edge, DeltaRevision::new(4)); + let lower = lower_data.bind(base); + let mut upper_data = TopologyDelta::default(); + upper_data.withdraw(&lower, EdgeRowId::new(0), DeltaRevision::new(3)); + upper_data.insert( + &lower, + EdgeRowId::new(4), + [NodeRowId::new(3), NodeRowId::new(4)], + DeltaRevision::new(5), + ); + let upper = upper_data.bind(&lower); + + assert_eq!(upper.provide_node_count(), 5); + assert_eq!(upper.provide_edge_count(), 5); + assert_edges(&upper, 0, 2, &[], &[0, 1]); + assert_edges(&upper, 0, 3, &[], &[1]); + assert_edges(&upper, 3, 1, &[], &[]); + assert_edges(&upper, 3, 2, &[3], &[3]); + assert_edges(&upper, 3, 4, &[], &[]); + assert_edges(&upper, 3, 5, &[], &[4]); + assert_edges(&upper, 4, 5, &[4], &[]); + assert_eq!( + upper.provide_endpoints_at(edge, DeltaRevision::new(2)), + Some(pair) + ); + assert_eq!(upper.provide_endpoints(edge), None); + assert_eq!(upper.provide_incoming(NodeRowId::new(3)).count(), 0); + assert_eq!( + upper + .provide_outgoing(NodeRowId::new(3)) + .collect::>(), + [EdgeRowId::new(4)] + ); +} + +/// Preserves lower-layer history after upper-layer rollover. +/// +/// An upper delta's decisions for a lower-layer edge survive past the retention capacity, keeping +/// the lower layer's own historical adjacency resolvable while the upper layer's current state +/// remains withdrawn. +#[test] +fn nested_history_rollover() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut lower_data = TopologyDelta::default(); + let edge = EdgeRowId::new(3); + let pair = [NodeRowId::new(3); 2]; + lower_data.insert(base, edge, pair, DeltaRevision::new(2)); + lower_data.withdraw(base, edge, DeltaRevision::new(4)); + let lower = lower_data.bind(base); + let mut upper_data = TopologyDelta::default(); + assert!(upper_data.withdraw(&lower, edge, DeltaRevision::new(3))); + let capacity = u64::try_from(CAPACITY).expect("should fit the retention capacity"); + for offset in 1..=capacity { + let revision = 3 + 2 * offset; + assert!(upper_data.insert(&lower, edge, pair, DeltaRevision::new(revision - 1))); + assert!(upper_data.withdraw(&lower, edge, DeltaRevision::new(revision))); + } + let upper = upper_data.bind(&lower); + assert_edges(&upper, 3, 1, &[], &[]); + assert_edges(&upper, 3, 2, &[3], &[3]); + assert_edges(&upper, 3, 3, &[3], &[3]); + assert_edges(&upper, 3, 4, &[], &[]); + assert_eq!( + upper.provide_endpoints_at(edge, DeltaRevision::new(1)), + None + ); + assert_eq!( + upper.provide_endpoints_at(edge, DeltaRevision::new(2)), + Some(pair) + ); + assert_eq!(upper.provide_endpoints(edge), None); +} + +/// Extends row domains without fabricating topology for reserved rows. +/// +/// Reserving a node and an edge extends both domains immediately while leaving the reserved rows +/// unbound: no adjacency and no endpoints, at the current state and at an earlier revision. +#[test] +fn reserved_rows_unbound() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + data.reserve_node(base, NodeRowId::new(5)); + data.reserve_edge(base, EdgeRowId::new(6)); + let provider = data.bind(base); + assert_eq!(provider.provide_node_count(), 6); + assert_eq!(provider.provide_edge_count(), 7); + assert_edges(&provider, 5, 0, &[], &[]); + assert_eq!(provider.provide_endpoints(EdgeRowId::new(6)), None); + assert_eq!( + provider.provide_endpoints_at(EdgeRowId::new(6), DeltaRevision::new(0)), + None + ); +} + +/// Panics when an insertion revision precedes a prior withdrawal. +#[test] +#[should_panic(expected = "history revisions must be nondecreasing")] +fn history_decreasing_revision() { + let origin = Origin::new(); + let base = NaiveTopologyProvider::from_ref(&origin); + let mut data = TopologyDelta::default(); + data.withdraw(base, EdgeRowId::new(0), DeltaRevision::new(4)); + data.insert( + base, + EdgeRowId::new(0), + [NodeRowId::new(2); 2], + DeltaRevision::new(3), + ); +} diff --git a/libs/@local/graph/atlas/src/serve/density/mod.rs b/libs/@local/graph/atlas/src/serve/density/mod.rs new file mode 100644 index 00000000000..9d6b9e1324e --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/density/mod.rs @@ -0,0 +1,303 @@ +//! Occupancy-based delivery cuts that target a cell-count band within the schedule's key width. +//! +//! For a view's multiset of [`MortonKey`] values V and integer depth 0 ≤ d ≤ 32, C(d, V) counts +//! distinct leading 2d-bit prefixes. The distinct-key count is Q(V) = C(32, V), and the saturation +//! depth is dₛₐₜ(V) = min { d : C(d, V) = Q(V) }. Empty views have C(d, V) = 0 at every depth. +//! Nonempty views have C(0, V) = 1. Empty and single-distinct-key views both have dₛₐₜ(V) = 0. +//! +//! # Properties +//! +//! For every view, C(d, V) is nondecreasing in d and equals Q(V) from saturation through depth 32. +//! Duplicate keys and input order leave the profile unchanged. +//! +//! A policy has integer band bounds 1 ≤ L ≤ U ≤ 2⁶⁴ − 1. Its schedule has integer span exponent 0 ≤ +//! s ≤ 63 and deepest tile zoom 0 ≤ z ≤ 32. Construction requires z > 0 and z + s ≤ 32. The offset +//! ceiling is h = 32 − (z + s). An offset 0 ≤ k ≤ h selects the view cut at depth s + k. +//! +//! For an integer count 0 ≤ c ≤ 2⁶⁴ − 1, the band distance is δ(c) = max(L − c, 0, c − U). +//! Resolution chooses the least integer k minimizing δ(C(s + k, V)) over 0 ≤ k ≤ min(max(dₛₐₜ(V) − +//! s, 0), h). Counts and distances use exact integer arithmetic, and equal distances select the +//! coarser offset. When dₛₐₜ(V) < s, the result is k = 0 and the cut at depth s is deeper than +//! saturation. +//! +//! For every carried integer offset 0 ≤ k ≤ 32, rebinding returns min(k, resolve(V)) under the new +//! policy and view. + +use core::{error::Error, fmt, num::NonZero}; + +use hashql_core::id::{Id as _, IdArray}; + +use crate::{ + math::Log2, + morton::{Depth, MortonKey, Zoom}, +}; + +#[cfg(test)] +mod tests; + +/// The inclusive occupied-cell target for a scope's delivery cut. +/// +/// By default, the band runs from 2,000 through 4,000 occupied cells. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) struct DensityBand { + /// The least count inside the band, L. + lower: NonZero, + /// The greatest count inside the band, U. + upper: NonZero, +} + +impl DensityBand { + /// Validates a configured band. + /// + /// Returns [`None`] when `upper` lies below `lower`, a band that admits no count. + /// + /// # Example + /// + /// This example uses a test-only constructor on a crate-private type. + /// + /// ```ignore + /// use core::num::NonZero; + /// + /// let band = DensityBand::new( + /// NonZero::new(2_000).expect("2,000 is positive"), + /// NonZero::new(4_000).expect("4,000 is positive"), + /// ) + /// .expect("2,000 ≤ 4,000"); + /// + /// assert_eq!(band.distance(3_000), 0); + /// assert_eq!(band.distance(1_500), 500); + /// assert_eq!(band.distance(4_500), 500); + /// ``` + #[must_use] + #[cfg(test)] // The density and manifest tests configure bands directly. + pub(crate) const fn new(lower: NonZero, upper: NonZero) -> Option { + if upper.get() < lower.get() { + return None; + } + + Some(Self { lower, upper }) + } + + /// Returns `count`'s distance to the band, zero inside it. + #[must_use] + pub(crate) const fn distance(self, count: u64) -> u64 { + if count < self.lower.get() { + return self.lower.get() - count; + } + + if count > self.upper.get() { + return count - self.upper.get(); + } + + 0 + } +} + +const impl Default for DensityBand { + fn default() -> Self { + Self { + lower: NonZero::new(2_000).expect("2,000 is positive"), + upper: NonZero::new(4_000).expect("4,000 is positive"), + } + } +} + +/// A recorded schedule that leaves the density policy no delivery-cut offset to resolve. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) enum DensityPolicyError { + /// The deepest bucket `max_tile_depth + span` exceeds the Morton key width. + Schedule { + /// The base-2 exponent of the cells per tile axis of the delivery cut. + span: Log2, + /// The deepest tile zoom the schedule names. + max_tile_depth: Zoom, + }, + /// The deepest zoom is the root, whose one catch-all tile no density policy deepens. + TerminalRoot, +} + +impl fmt::Display for DensityPolicyError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Schedule { + span, + max_tile_depth, + } => write!( + fmt, + "the schedule's deepest bucket {max_tile_depth} + {span} exceeds the 32 \ + subdivisions a Morton key resolves" + ), + Self::TerminalRoot => fmt.write_str( + "a generation whose deepest zoom is its root serves one catch-all tile, which no \ + density policy deepens", + ), + } + } +} + +impl Error for DensityPolicyError {} + +/// A band and schedule bound for choosing a scope's delivery cut. +/// +/// The band, span exponent and offset ceiling determine resolution from the view's occupied-cell +/// profile. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) struct DensityPolicy { + /// The occupied-cell target. + band: DensityBand, + /// The span exponent s, the cells per tile axis of the delivery cut as a base-2 log. + span: Log2, + /// The offset ceiling h = 32 − (z + s). + ceiling: Zoom, +} + +impl DensityPolicy { + /// Validates the schedule's room for density offsets. + /// + /// # Errors + /// + /// Returns [`DensityPolicyError::TerminalRoot`] when `max_tile_depth` is zero, then + /// [`DensityPolicyError::Schedule`] when `max_tile_depth + span` exceeds 32. + pub(crate) fn new( + band: DensityBand, + span: Log2, + max_tile_depth: Zoom, + ) -> Result { + if max_tile_depth == Zoom::MIN { + return Err(DensityPolicyError::TerminalRoot); + } + + let Some(ceiling) = max_tile_depth.depth(span).map(Depth::ceiling) else { + return Err(DensityPolicyError::Schedule { + span, + max_tile_depth, + }); + }; + + Ok(Self { + band, + span, + ceiling, + }) + } + + /// Chooses the coarsest offset minimizing distance to the target band. + /// + /// # Complexity + /// + /// Finds saturation with [`ViewOccupancy::saturation_depth`] and examines at most 33 candidate + /// offsets, using constant additional space. + #[must_use] + pub(crate) fn resolve(self, occupancy: &ViewOccupancy) -> Zoom { + // Counts equal the distinct-key count from saturation onward. Later offsets have the + // same distance, and the coarser tie-break retains the first. The saturation cap therefore + // omits only candidates that cannot change the result. Saturation below the span admits + // only offset zero, whose cut is already deeper than saturation. + let saturation = occupancy.saturation_depth().zoom(self.span); + let limit = saturation.min(self.ceiling); + + let mut resolved = Zoom::MIN; + let mut distance = u64::MAX; + + // Occupancy can plateau before increasing again: counts 2, 2, 4 reach band [3, 4] only + // at the last offset. A non-improving offset alone therefore cannot terminate the search. + for offset in Zoom::MIN..=limit { + let depth = offset.saturating_depth(self.span); + + let candidate = self.band.distance(occupancy.occupied_cells(depth)); + if candidate < distance { + distance = candidate; + resolved = offset; + } + } + + resolved + } + + /// Returns the coarser of `zoom` and [`resolve`](Self::resolve)'s offset for `occupancy`. + #[must_use] + pub(crate) fn rebind(self, zoom: Zoom, occupancy: &ViewOccupancy) -> Zoom { + zoom.min(self.resolve(occupancy)) + } +} + +/// Distinct occupied-cell counts at every Morton depth. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ViewOccupancy { + /// C(d, V) at every depth d. + occupied: IdArray, +} + +impl ViewOccupancy { + /// Builds the occupancy profile, sorting `keys` in ascending order in place. + /// + /// # Complexity + /// + /// For n keys, sorting takes O(n log n) worst-case time and allocates no heap storage. + /// The profile pass takes O(n + 33) time with a 33-entry result and a temporary 33-entry + /// separation table, excluding the input storage. + #[must_use] + pub(crate) fn of(keys: &mut [MortonKey]) -> Self { + keys.sort_unstable(); + + let mut occupied = IdArray::from_elem(0_u64); + if keys.is_empty() { + return Self { occupied }; + } + + let mut separations: IdArray = + IdArray::from_elem(0_u64); + for &[earlier, later] in keys.array_windows::<2>() { + if earlier == later { + continue; + } + + // Adjacent distinct keys first occupy separate cells one depth below their shared + // prefix. + let depth = earlier.shared_depth(later).plus(1); + separations[depth] += 1; + } + + // Sorted prefixes form contiguous runs. Each adjacent separation starts one more occupied + // cell at its depth and every deeper depth. + let mut cells = 1; + for (count, separations) in occupied.iter_mut().zip(separations) { + cells += separations; + *count = cells; + } + + Self { occupied } + } + + /// Counts the distinct depth-`depth` cells the view occupies: `C(depth, V)`. + /// + /// Zero for an empty view. One for every other view at [`Depth::MIN`], the whole domain. + #[must_use] + pub(crate) const fn occupied_cells(&self, depth: Depth) -> u64 { + self.occupied[depth] + } + + /// Counts the distinct complete keys the view carries: `Q(V) = C(32, V)`. + #[must_use] + pub(crate) const fn distinct_keys(&self) -> u64 { + self.occupied_cells(Depth::MAX) + } + + /// Returns the coarsest depth at which every distinct key occupies its own cell: dₛₐₜ(V). + /// + /// Returns [`Depth::MIN`] for empty and single-distinct-key views. + /// + /// # Complexity + /// + /// Reads the distinct-key count, then scans at most 33 profile entries, using constant + /// additional space. + #[must_use] + pub(crate) fn saturation_depth(&self) -> Depth { + let saturated = self.distinct_keys(); + + // Depth::MAX always reaches the distinct-key count. + Depth::all() + .find(|&depth| self.occupied_cells(depth) == saturated) + .unwrap_or(Depth::MAX) + } +} diff --git a/libs/@local/graph/atlas/src/serve/density/tests.rs b/libs/@local/graph/atlas/src/serve/density/tests.rs new file mode 100644 index 00000000000..af0c96c65bd --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/density/tests.rs @@ -0,0 +1,407 @@ +use core::num::NonZero; +use std::collections::HashSet; + +use hashql_core::id::Id as _; +use proptest::{prop_assert, prop_assert_eq, property_test}; + +use super::{DensityBand, DensityPolicy, DensityPolicyError, ViewOccupancy}; +use crate::{ + math::{Log2, nz}, + morton::{Depth, MortonCell, MortonKey, Zoom}, +}; + +/// The fixtures' span exponent. +/// +/// A view's cut at offset `k` is depth `1 + k`. +const SPAN: Log2 = Log2::new(1).unwrap(); + +/// The fixtures' deepest served zoom. +const MAX_TILE_DEPTH: Zoom = Zoom::new(4).unwrap(); + +/// The fixture schedule's offset ceiling: 32 − (4 + 1) = 27. +const CEILING: Zoom = Zoom::new(27).unwrap(); + +/// Builds a density policy over the band `[lower, upper]` with the fixture span and tile depth. +/// +/// # Panics +/// +/// Panics if `upper` lies below `lower`. +fn policy(lower: NonZero, upper: NonZero) -> DensityPolicy { + let band = DensityBand::new(lower, upper).expect("the fixture band is ordered"); + + DensityPolicy::new(band, SPAN, MAX_TILE_DEPTH).expect("the fixture policy is admissible") +} + +/// Builds a view with one additional occupied cell per depth through depth 24. +/// +/// Key i has one set bit at position 64 − 2i for 1 ≤ i ≤ 24. Together with the zero key, these give +/// exactly one adjacent separation at every depth from 1 through 24: C(d, V) = 1 + d for 0 ≤ d ≤ 24 +/// and Q(V) = 25. +fn deep_view() -> ViewOccupancy { + let mut keys = vec![MortonKey::from_bits(0)]; + keys.extend((1..=24_u32).map(|index| MortonKey::from_bits(1_u64 << (64 - 2 * index)))); + + ViewOccupancy::of(&mut keys) +} + +/// Returns the corner key of one cell of the depth's grid. +/// +/// # Panics +/// +/// Panics if `depth` exceeds the key width or `(x, y)` lies outside the depth's grid. +fn key(depth: u8, x: u32, y: u32) -> MortonKey { + MortonCell::new( + Depth::try_new(depth).expect("the fixture depth lies within the key width"), + x, + y, + ) + .expect("the fixture cell lies on the depth's grid") + .min_key() +} + +/// Builds a view whose occupancy plateaus once and then splits. +/// +/// The depth-3 cells `(0,0)` and `(1,0)` share one depth-1 quadrant, while `(4,0)` and `(5,0)` +/// share another. Each pair shares a depth-2 cell and separates at depth 3. The counts are C(1, V) +/// = 2, C(2, V) = 2 and C(3, V) = Q(V) = 4, with saturation at depth 3. +fn plateau_view() -> ViewOccupancy { + ViewOccupancy::of(&mut [key(3, 0, 0), key(3, 1, 0), key(3, 4, 0), key(3, 5, 0)]) +} + +/// [`DensityBand::new`] accepts ordered and single-count bands and refuses inverted ones. +#[test] +fn band_inverted_bounds() { + let lower = nz!(2_000); + let upper = nz!(4_000); + + assert!(DensityBand::new(lower, upper).is_some()); + assert_eq!(DensityBand::new(upper, lower), None); + assert!( + DensityBand::new(lower, lower).is_some(), + "a single-count band is ordered" + ); +} + +/// `distance` is zero at both band endpoints and one immediately outside either. +#[test] +fn band_endpoints() { + let band = DensityBand::new(nz!(2_000), nz!(4_000)).expect("the fixture band is ordered"); + + assert_eq!(band.distance(2_000), 0); + assert_eq!(band.distance(4_000), 0); + assert_eq!(band.distance(1_999), 1); + assert_eq!(band.distance(4_001), 1); +} + +/// `distance` is zero inside the band and the gap to the nearer endpoint outside it. +#[test] +fn band_distance_gaps() { + let band = DensityBand::new(nz!(2_000), nz!(4_000)).expect("the fixture band is ordered"); + + assert_eq!(band.distance(3_000), 0); + assert_eq!(band.distance(1_500), 500); + assert_eq!(band.distance(4_500), 500); +} + +/// The schedule ceiling limits resolution before the view saturates. +/// +/// Span 6 and deepest zoom 18 leave ceiling 32 − (18 + 6) = 8. The deepest bucket is 18 + 6 + 8 = +/// 32, while the view cut is 6 + 8 = 14. The deep view saturates at depth 24, giving saturation +/// offset 24 − 6 = 18. Every candidate count is below the band and increases with depth, making +/// offset 8 the unique minimum within the ceiling. +#[test] +fn resolve_key_width_ceiling() { + let policy = DensityPolicy::new( + DensityBand::new(nz!(100), nz!(200)).expect("the fixture band is ordered"), + Log2::new(6).expect("the fixture span lies below the shift width"), + Zoom::new(18).expect("the fixture zoom lies within the key width"), + ) + .expect("a span-6 schedule serving 18 zooms is admissible"); + let view = deep_view(); + + assert_eq!( + view.saturation_depth(), + Depth::new(24), + "k_sat is 24 - 6 = 18" + ); + assert_eq!( + view.occupied_cells(Depth::new(14)), + 15, + "the cut the ceiling gives" + ); + assert_eq!( + view.occupied_cells(Depth::new(24)), + 25, + "the cut k_sat would give" + ); + + assert_eq!( + policy.resolve(&view).get(), + 8, + "the key-width ceiling limits the offset before saturation" + ); +} + +/// [`DensityPolicy::new`] refuses a root-only schedule and one beyond the key width. +/// +/// It returns [`DensityPolicyError::TerminalRoot`] at the minimum zoom and +/// [`DensityPolicyError::Schedule`] when the span and maximum depth exceed the key width. +#[test] +fn policy_inadmissible_schedules() { + let band = DensityBand::new(nz!(2_000), nz!(4_000)).expect("the fixture band is ordered"); + let span = Log2::new(6).expect("the fixture span lies below the shift width"); + let max_tile_depth = Zoom::new(30).expect("the fixture zoom lies within the key width"); + + assert_eq!( + DensityPolicy::new(band, span, Zoom::MIN), + Err(DensityPolicyError::TerminalRoot) + ); + assert_eq!( + DensityPolicy::new(band, span, max_tile_depth), + Err(DensityPolicyError::Schedule { + span, + max_tile_depth + }) + ); +} + +#[test] +fn occupancy_duplicate_keys() { + let anchor = key(3, 2, 1); + let view = ViewOccupancy::of(&mut [anchor, anchor, anchor]); + + assert_ne!(view.occupied_cells(Depth::MIN), 0); + assert_eq!(view.distinct_keys(), 1); + assert_eq!(view.occupied_cells(Depth::MAX), 1); + assert_eq!(view.saturation_depth(), Depth::MIN); +} + +/// An empty view occupies no cell at any depth and has no distinct keys. +#[test] +fn occupancy_empty_view() { + let view = ViewOccupancy::of(&mut []); + + assert_eq!(view.occupied_cells(Depth::MIN), 0); + assert_eq!(view.distinct_keys(), 0); + for depth in Depth::all() { + assert_eq!(view.occupied_cells(depth), 0, "depth {}", depth.get()); + } +} + +/// A view whose occupancy plateaus reaches its four distinct keys first at depth three. +/// +/// Depth three is its saturation depth. +#[test] +fn occupancy_plateau_saturation() { + let view = plateau_view(); + + assert_eq!(view.occupied_cells(Depth::new(0)), 1); + assert_eq!(view.occupied_cells(Depth::new(1)), 2); + assert_eq!(view.occupied_cells(Depth::new(2)), 2); + assert_eq!(view.occupied_cells(Depth::new(3)), 4); + assert_eq!(view.occupied_cells(Depth::new(4)), 4); + assert_eq!(view.distinct_keys(), 4); + assert_eq!(view.saturation_depth(), Depth::new(3)); +} + +/// Resolving an empty view yields the minimum zoom. +#[test] +fn resolve_empty_view() { + assert_eq!( + policy(nz!(2), nz!(4)).resolve(&ViewOccupancy::of(&mut [])), + Zoom::MIN + ); +} + +/// Saturation at depth 0 admits only offset 0, whose cut has depth 1 under this span. +#[test] +fn resolve_duplicate_keys() { + let anchor = key(3, 2, 1); + + assert_eq!( + policy(nz!(2), nz!(4)).resolve(&ViewOccupancy::of(&mut [anchor, anchor])), + Zoom::MIN + ); +} + +/// Offsets 0 and 1 count 2 cells, and offset 2 counts 4. All are in band. +#[test] +fn resolve_in_band_tie() { + assert_eq!(policy(nz!(2), nz!(4)).resolve(&plateau_view()).get(), 0); +} + +/// Offsets 0 and 1 count 2 cells, while offset 2 counts 4 and is the only in-band cut. +#[test] +fn resolve_plateau() { + assert_eq!(policy(nz!(3), nz!(4)).resolve(&plateau_view()).get(), 2); +} + +/// Band `[3, 3]` puts offsets 0 and 1 one cell below and offset 2 one cell above. +#[test] +fn resolve_equal_distance() { + assert_eq!(policy(nz!(3), nz!(3)).resolve(&plateau_view()).get(), 0); +} + +/// Every candidate count is below `[10, 20]`, with the closest count at offset 2. +#[test] +fn resolve_below_band() { + assert_eq!(policy(nz!(10), nz!(20)).resolve(&plateau_view()).get(), 2); +} + +/// Counts never decrease with depth. Offsets 0 and 1 tie for the smallest excess over `[1, 1]`. +#[test] +fn resolve_above_band() { + assert_eq!(policy(nz!(1), nz!(1)).resolve(&plateau_view()).get(), 0); +} + +/// The plateau view's saturation offset is below the schedule ceiling. +/// +/// Saturation at depth 3 with span 1 limits the offset to 3 − 1 = 2, below [`CEILING`] = 27. The +/// band `[10, 20]` exceeds every count in this view. +#[test] +fn resolve_saturation_above_span() { + let view = plateau_view(); + + assert_eq!(view.saturation_depth(), Depth::new(3)); + assert_eq!( + view.occupied_cells(Depth::new(5)), + view.occupied_cells(Depth::new(3)) + ); + assert_eq!(policy(nz!(10), nz!(20)).resolve(&view).get(), 2); +} + +/// The profile equals a direct census of distinct prefixes at every depth. +/// +/// `occupied_cells` at each depth equals the count of distinct key prefixes at that depth, +/// `distinct_keys` equals the set size, and the saturation depth is the coarsest depth reaching +/// it. +#[property_test] +fn occupied_cells_prefix_census( + #[strategy = proptest::collection::vec(0_u64..1_u64 << 12, 0..24_usize)] bits: Vec, +) { + // Shifting 12 bits left by 40 confines varying bits to positions 40 through 51. Distinct keys + // can separate only at depths 7 through 12. + let keys: Vec = bits + .iter() + .map(|&bits| MortonKey::from_bits(bits << 40)) + .collect(); + let view = ViewOccupancy::of(&mut keys.clone()); + + let distinct: HashSet = keys.iter().map(|key| key.to_bits()).collect(); + prop_assert_eq!(view.distinct_keys(), distinct.len() as u64); + prop_assert_eq!(view.occupied_cells(Depth::MIN) == 0, keys.is_empty()); + + let mut previous = 0; + for depth in Depth::all() { + let census: HashSet = keys.iter().map(|key| key.prefix(depth)).collect(); + let expected = if keys.is_empty() { + 0 + } else { + census.len() as u64 + }; + + prop_assert_eq!( + view.occupied_cells(depth), + expected, + "depth {}", + depth.get() + ); + prop_assert!( + view.occupied_cells(depth) >= previous, + "occupancy fell at depth {}", + depth.get() + ); + previous = view.occupied_cells(depth); + + // The saturation depth is the coarsest depth reaching the distinct-key count. + prop_assert_eq!( + depth >= view.saturation_depth(), + view.occupied_cells(depth) == view.distinct_keys(), + "depth {} against saturation {}", + depth.get(), + view.saturation_depth().get() + ); + } +} + +/// `resolve` is order-invariant, within the ceiling and nearest the band. +/// +/// `resolve` is invariant under key order, never exceeds the ceiling, and resolves to a zoom whose +/// occupancy is nearest the band. +#[property_test] +fn resolve_order_invariance_argmin( + #[strategy = proptest::collection::vec(0_u64..1_u64 << 8, 1..16_usize)] bits: Vec, +) { + let policy = policy(nz!(3), nz!(5)); + + let mut keys: Vec = bits + .iter() + .map(|&bits| MortonKey::from_bits(bits << 48)) + .collect(); + let mut reversed: Vec = keys.iter().rev().copied().collect(); + + let view = ViewOccupancy::of(&mut keys); + let resolved = policy.resolve(&view); + prop_assert_eq!(resolved, policy.resolve(&ViewOccupancy::of(&mut reversed))); + + let cut = resolved.saturating_depth(SPAN); + let distance = policy.band.distance(view.occupied_cells(cut)); + prop_assert!(resolved <= CEILING); + for offset in Zoom::MIN..=CEILING { + if offset > view.saturation_depth().zoom(SPAN) { + continue; + } + + let candidate = policy + .band + .distance(view.occupied_cells(offset.saturating_depth(SPAN))); + prop_assert!( + distance < candidate || (distance == candidate && resolved <= offset), + "offset {} beats the resolved {}", + offset, + resolved + ); + } +} + +/// The plateau view reaches 4 cells at offset 2, below the band `[20, 20]`. +/// +/// The deep view reaches the band exactly at offset 18: C(19, V) = 20. Rebinding retains offset 2 +/// instead of adding 16 subdivisions. +#[test] +fn rebind_deeper_view() { + let policy = policy(nz!(20), nz!(20)); + let carried = policy.resolve(&plateau_view()); + assert_eq!(carried.get(), 2, "the plateau view's own resolution"); + assert_eq!( + policy.resolve(&deep_view()).get(), + 18, + "the deep view's own resolution, which the re-bind must not adopt" + ); + + assert_eq!(policy.rebind(carried, &deep_view()), carried); +} + +/// The deep view resolves offset 18, while the plateau view resolves offset 2. +#[test] +fn rebind_coarser_view() { + let policy = policy(nz!(20), nz!(20)); + let carried = policy.resolve(&deep_view()); + + assert_eq!( + policy.rebind(carried, &plateau_view()).get(), + 2, + "the deep session kept its cut over a view the band serves shallower" + ); +} + +/// Rebinding from a deep resolution onto an empty view yields the minimum zoom. +#[test] +fn rebind_empty_view() { + let policy = policy(nz!(20), nz!(20)); + + assert_eq!( + policy.rebind(policy.resolve(&deep_view()), &ViewOccupancy::of(&mut [])), + Zoom::MIN + ); +} diff --git a/libs/@local/graph/atlas/src/serve/intern.rs b/libs/@local/graph/atlas/src/serve/intern.rs new file mode 100644 index 00000000000..f9d793e2e27 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/intern.rs @@ -0,0 +1,103 @@ +//! Interning of distinct values into one table addressed by insertion index. +//! +//! An [`InternTable`] stores each distinct value once and answers both directions: the +//! [`TableIndex`] of a value, and the value at an index. + +use core::hash::{BuildHasher as _, Hash}; + +use hashbrown::HashTable; +use hashql_core::{ + collections::FastHasher, + id::{Id as _, IdSlice, IdVec, newtype}, +}; +use moka::Equivalent; + +newtype! { + /// A reference to one interned value by its position in an [`InternTable`]'s insertion order. + #[id(const)] + pub(crate) struct TableIndex(u32) +} + +/// A table of distinct values keyed by [`TableIndex`]. +/// +/// Each distinct value interns once. [`intern`](Self::intern) and [`index_of`](Self::index_of) +/// find a value's index, and [`entries`](Self::entries) reads values by index. +#[derive(Debug)] +pub(crate) struct InternTable { + /// The hasher every lookup and insertion shares. + hasher: FastHasher, + /// Every interned value, in insertion order. + table: IdVec, T>, + /// The index of every interned value, found through the value's hash. + reverse: HashTable>, +} + +impl InternTable +where + T: 'static, +{ + /// Creates an empty table. + pub(crate) fn new() -> Self { + Self { + hasher: FastHasher::default(), + table: IdVec::new(), + reverse: HashTable::new(), + } + } + + /// Returns `value`'s table index, interning it if this is its first occurrence. + /// + /// # Panics + /// + /// Panics when a new distinct value's insertion index, the current table length, lies outside + /// the range [`TableIndex`] represents. + pub(crate) fn intern(&mut self, value: T) -> TableIndex + where + T: Hash + PartialEq, + { + let hash = self.hasher.hash_one(&value); + + if let Some(&value) = self.reverse.find(hash, |&index| self.table[index] == value) { + value + } else { + let index = TableIndex::from_usize(self.table.len()); + + self.table.push(value); + self.reverse.insert_unique(hash, index, |&index| { + self.hasher.hash_one(&self.table[index]) + }); + + index + } + } + + /// Returns the table index of a value equivalent to `value`, if the table holds one. + /// + /// The lookup hashes `value` as `K` and compares it against interned `T` values, which + /// [`Equivalent`]'s contract supports: a key hashes as the value it compares equal to. + pub(super) fn index_of(&self, value: &K) -> Option> + where + K: Equivalent + Hash + ?Sized, + { + let hash = self.hasher.hash_one(value); + + self.reverse + .find(hash, |&index| value.equivalent(&self.table[index])) + .copied() + } + + /// Borrows every interned value, in insertion order. + pub(super) const fn entries(&self) -> &IdSlice, T> { + &self.table + } + + /// Returns the number of distinct interned values. + pub(super) const fn len(&self) -> usize { + self.table.len() + } + + /// Returns whether no value has been interned yet. + pub(super) const fn is_empty(&self) -> bool { + self.table.is_empty() + } +} diff --git a/libs/@local/graph/atlas/src/serve/mod.rs b/libs/@local/graph/atlas/src/serve/mod.rs new file mode 100644 index 00000000000..d4dc6a4552a --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/mod.rs @@ -0,0 +1,28 @@ +//! Request-coherent serving over fitted generations and live deltas. +//! +//! The runtime manager opens fitted generations and owns any configured feed execution, retirement +//! and removal. Without feed options or temporal axes, a generation instead exposes a static +//! publication. An active feed can lag or fail. Each scene-backed delivery request +//! captures a coherent world-and-delta epoch and obtains a cached or newly resolved visibility +//! scope before constructing a `scene::Scene`. A cached mask and schedule may predate the +//! request's epoch within the same delta lifetime. Scene geometry, identity and topology lookups +//! still use only the request's captured publication. +//! +//! Authority validation binds the generation and [`delta::DeltaId`], not a +//! [`delta::DeltaRevision`]. A `runtime::registry::Observation` captures the immutable revision +//! used for data reads. Authority-token expiry, visibility-cache age and +//! retained-generation admission are distinct checks even when the host derives them from one +//! maximum duration. Issuing or renewing a token does not force a permission-store refresh. A +//! request using the new token may therefore reuse a soft-stale scope while its detached refresh +//! runs. + +pub(crate) mod codec; +pub(crate) mod delta; +pub(crate) mod density; +mod intern; +mod schedule; +pub(crate) mod secret; +#[cfg(test)] +pub(crate) mod tests; +pub(crate) mod visibility; +mod world; diff --git a/libs/@local/graph/atlas/src/serve/schedule/bucket.rs b/libs/@local/graph/atlas/src/serve/schedule/bucket.rs new file mode 100644 index 00000000000..153345b83a3 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/bucket.rs @@ -0,0 +1,102 @@ +//! The recorded bucket cuts, proven against the Morton key width. + +use super::{ScheduleError, ScheduleWidthError}; +use crate::{ + math::Log2, + morton::{Depth, Zoom}, + salt::lod::stage::LodConfig, +}; + +/// A generation's recorded schedule, proven within the Morton key width. +/// +/// Validation at [`new`](Self::new) makes [`cut`](Self::cut) the exact sum `zoom + span` at every +/// served zoom, and [`deepest`](Self::deepest) is the cut at the deepest one. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) struct BucketSchedule(LodConfig); + +impl BucketSchedule { + /// Validates the recorded schedule against the Morton key width. + /// + /// # Errors + /// + /// Returns [`ScheduleError`] when `max_tile_depth + span` exceeds the key width. + pub(crate) const fn new(config: LodConfig) -> Result { + if config.deepest().is_none() { + return Err(ScheduleError { + span: config.span, + max_tile_depth: config.max_tile_depth, + }); + } + + Ok(Self(config)) + } + + /// Deepens the delivery cut by `offset` zoom levels. + /// + /// The offset adds to the span exponent and deepens every served zoom's cut by `offset` without + /// changing which zooms the schedule serves. For a fixed cell and bucket assignment, cumulative + /// delivery includes the rows delivered without the offset. Zero offset leaves the schedule + /// unchanged. Individual increments need not contain their unshifted counterparts. + /// + /// # Errors + /// + /// Returns [`ScheduleWidthError`] when the deepened cut leaves the Morton key width. + pub(super) fn offset(self, offset: Zoom) -> Result { + self.deepest() + .checked_add(offset.into()) + .ok_or(ScheduleWidthError { + schedule: self, + offset, + })?; + + // `deepest + offset` fits the key width above, and `span` is at most `deepest`. + let span = self + .span() + .checked_add(Log2::from(offset)) + .expect("the validated cut should fit within the key width"); + + Ok(Self(LodConfig { span, ..self.0 })) + } + + /// Returns the deepest tile zoom the schedule serves. + pub(crate) const fn max_tile_depth(self) -> Zoom { + self.0.max_tile_depth + } + + /// Returns the base-2 exponent of the cells per tile axis of the delivery cut. + /// + /// A tile at zoom `z` cuts at depth `z + span`, sampling a `2^span` by `2^span` grid. + pub(crate) const fn span(self) -> Log2 { + self.0.span + } + + /// Returns the delivery cut of zoom `z`. + /// + /// # Panics + /// + /// Panics when `z` exceeds [`max_tile_depth`](Self::max_tile_depth). + pub(crate) const fn cut(self, z: Zoom) -> Depth { + assert!( + z.get() <= self.0.max_tile_depth.get(), + "the schedule serves zooms 0..=max_tile_depth", + ); + + z.saturating_depth(self.0.span) + } + + /// Returns the deepest served bucket, the catch-all. + /// + /// Every recorded bucket is at most this depth: the cascade assigns depths up to it and gives + /// it to the rows no pass claims ([`buckets`](crate::salt::lod::cascade::buckets)). + pub(crate) const fn deepest(self) -> Depth { + self.cut(self.0.max_tile_depth) + } + + /// Returns the first zoom whose cumulative schedule delivers `bucket`. + /// + /// Bucket `b` enters the schedule at zoom `b - span`, clamped to the root for the buckets + /// the root itself spans. + pub(crate) const fn first_zoom(self, bucket: Depth) -> Zoom { + bucket.first_zoom(self.span()) + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/column.rs b/libs/@local/graph/atlas/src/serve/schedule/column.rs new file mode 100644 index 00000000000..6741d27bd34 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/column.rs @@ -0,0 +1,336 @@ +//! Natural bucket columns with stable-row lookup. +//! +//! [`BucketColumn`] indexes both complete visible cascades and extension rows assigned against a +//! base key set. Delivery clamps natural buckets into the requested catch-all. + +use core::{cmp::Ordering, slice}; + +use hashql_core::{ + heap::CollectIn as _, + id::{Id as _, IdArray}, +}; + +use crate::{ + allocator::{HeapMemoryUsage, MemoryUsage, MemoryUsageAllocator}, + identity::NodeRowId, + math::{Bounds2, Log2, Vec2}, + morton::{Depth, MortonCell, MortonKey}, + salt::lod::{cascade, stage::WIRE_FRAME}, + serve::{ + delta::epoch::Epoch, + visibility::VisibilityMask, + world::{Layout, node_importance::NodePriority}, + }, +}; + +/// One row as the cascade sees it: where it sits and how it ranks. +/// +/// The cascade orders by key and breaks ties by priority. Capture reduces each row to those two +/// quantities once, and they are never re-derived from the layout. +#[derive(Debug, Copy, Clone)] +pub(super) struct ScheduleNode { + /// The row this schedules. + pub node: NodeRowId, + /// The row's quantized position, in wire-frame Morton order. + pub key: MortonKey, + /// The row's importance, which decides which of two rows in a cell is delivered first. + pub priority: NodePriority, +} + +impl ScheduleNode { + /// Quantizes `position` into the wire frame and pairs it with `priority`. + pub(super) fn new(node: NodeRowId, position: Vec2, priority: NodePriority) -> Self { + let [x, y] = WIRE_FRAME.quantize(position); + + Self { + node, + key: MortonKey::new(x, y), + priority, + } + } + + /// Captures `node` when the mask admits it and the epoch places it. + /// + /// Returns [`None`] for a withheld row and for one with no placement at this epoch. An + /// admitted row that has not been placed yet therefore contributes to neither buckets nor + /// bounds. + /// + /// # Panics + /// + /// Panics when an admitted row requires reading a `layout` that does not belong to `epoch`'s + /// world, or when a placed row has no allocated priority. The layout's count checks rule out + /// the missing priority for a base row, and slot allocation rules it out for an added row. + fn visible( + layout: &Layout, + epoch: &Epoch, + mask: &VisibilityMask, + node: NodeRowId, + ) -> Option<(Self, Vec2)> { + mask.visible_node(node)?; + + let position = layout.position(epoch, node)?; + let priority = layout + .priority(epoch, node) + .expect("a placed node should have an allocated priority"); + + Some((Self::new(node, position, priority), position)) + } + + /// Captures every admitted, placed row of `nodes` with their tight extent. + /// + /// The extent covers exactly the captured rows and is [`None`] for an empty capture. + /// + /// # Panics + /// + /// Panics when an admitted row requires reading a `layout` that does not belong to `epoch`'s + /// world, a placed row lacks a priority, or a captured coordinate is non-finite. A non-finite + /// coordinate leaves the extent undefined over a non-empty capture. + pub(super) fn collect( + layout: &Layout, + epoch: &Epoch, + mask: &VisibilityMask, + nodes: impl IntoIterator, + ) -> (Vec, Option) { + let mut rows = Vec::new(); + let bounds = Bounds2::from_points(nodes.into_iter().filter_map(|node| { + let (row, position) = Self::visible(layout, epoch, mask, node)?; + rows.push(row); + Some(position) + })); + assert!( + rows.is_empty() || bounds.is_some(), + "visible placements must have finite coordinates" + ); + (rows, bounds) + } +} + +/// A captured row beside the bucket the cascade assigned it. +#[derive(Debug, Copy, Clone)] +pub(super) struct BucketedNode { + /// The depth at which this row first becomes deliverable. + pub bucket: Depth, + /// The captured row. + pub row: ScheduleNode, +} + +/// One bucket's rows within a cell. +/// +/// A bucket below the catch-all is already a contiguous, ordered slice of the column. The +/// catch-all is the union of every deeper bucket, which has to be gathered and reordered. The +/// two cases therefore differ in whether delivery can borrow or must allocate. +enum BucketRun<'schedule> { + /// A borrowed run of one stored bucket. + Slice(slice::Iter<'schedule, BucketedNode>), + /// A reordered union of every bucket at or below the catch-all. + Gathered(alloc::vec::IntoIter), +} + +impl Iterator for BucketRun<'_> { + type Item = ScheduleNode; + + fn next(&mut self) -> Option { + match self { + Self::Slice(slots) => slots.next().map(|slot| slot.row), + Self::Gathered(rows) => rows.next(), + } + } +} + +/// Rows sorted by delivery bucket, with the two indexes delivery reads them through. +/// +/// Delivery asks two questions of a schedule: which rows fall in this bucket and this cell, and +/// which bucket does this row belong to. The first is answered by the bucket-major ordering of +/// `slots` with the per-bucket ranges beside it, and the second by a separate row-ordered index, +/// because one ordering cannot serve both searches. +#[derive(Debug)] +pub(super) struct BucketColumn { + /// Every captured row beside its bucket, in bucket, key and priority order. + slots: Box<[BucketedNode], MemoryUsageAllocator>, + /// The range of `slots` each bucket occupies, empty for an unoccupied bucket. + buckets: IdArray, { Depth::MAX.as_usize() + 1 }>, + /// Every captured row's bucket, in row order. + by_node: Box<[(NodeRowId, Depth)], MemoryUsageAllocator>, + /// The heap bytes `slots` and `by_node` hold. + memory_usage: MemoryUsage, +} + +impl BucketColumn { + /// Assigns natural buckets against the rows and an optional better-ranked key set. + /// + /// `shared` returns the deepest prefix shared with any external predecessor. Every external + /// predecessor must outrank every input row: a row's natural bucket is one past the deepest + /// grid it shares with a better-ranked row, and the external keys enter that rule only as + /// better-ranked rows. One past the deepest prefix saturates at [`Depth::MAX`]: a key sharing + /// its complete prefix with a predecessor takes that depth, because no deeper level exists to + /// separate them. + pub(super) fn new( + mut rows: Vec, + shared: impl Fn(MortonKey) -> Option, + ) -> Self { + rows.sort_unstable_by_key(|row| (row.key, row.priority)); + let buckets = cascade::separation_buckets(&rows, |row| row.key, |row| row.priority); + + let alloc = MemoryUsageAllocator::global(); + let memory_usage = alloc.memory_usage(); + let mut slots: Vec<_, _> = rows + .into_iter() + .zip(buckets.iter().copied()) + .map(|(row, bucket)| { + let predecessor = + shared(row.key).map_or(Depth::MIN, |depth| depth.saturating_add(Log2::ONE)); + BucketedNode { + bucket: bucket.max(predecessor), + row, + } + }) + .collect_in(alloc.clone()); + slots.sort_unstable_by_key(|slot| (slot.bucket, slot.row.key, slot.row.priority)); + + let mut counts = IdArray::::from_elem(0); + for slot in &slots { + counts[slot.bucket] += 1; + } + + let mut start = 0; + let buckets = counts.map(|count| { + let range = start..start + count; + start = range.end; + core::range::Range::from(range) + }); + + let mut by_node: Vec<_, _> = slots + .iter() + .map(|slot| (slot.row.node, slot.bucket)) + .collect_in(alloc); + by_node.sort_unstable_by_key(|&(node, _)| node); + + Self { + slots: slots.into_boxed_slice(), + buckets, + by_node: by_node.into_boxed_slice(), + memory_usage, + } + } + + /// Iterates every captured key, in bucket-major order. + pub(super) fn keys(&self) -> impl Iterator { + self.slots.iter().map(|slot| slot.row.key) + } + + /// Returns the bucket `node` was assigned, absent when the column does not hold it. + /// + /// # Complexity + /// + /// O(log n) for n captured rows. + pub(super) fn bucket_of(&self, node: NodeRowId) -> Option { + let index = self + .by_node + .binary_search_by_key(&node, |&(node, _)| node) + .ok()?; + + Some(self.by_node[index].1) + } + + /// Borrows every slot assigned to `bucket`, in key and priority order. + fn bucket_slots(&self, bucket: Depth) -> &[BucketedNode] { + &self.slots[self.buckets[bucket]] + } + + /// Borrows the slots of `bucket` whose keys lie inside `cell`. + /// + /// A Morton cell is a contiguous key range. The cell's rows are therefore a contiguous window + /// of the bucket's already key-ordered slots, and two binary searches locate it. + fn cell_slots(&self, bucket: Depth, cell: MortonCell) -> &[BucketedNode] { + let slots = self.bucket_slots(bucket); + + let start = slots.partition_point(|slot| slot.row.key < cell.min_key()); + let count = slots[start..].partition_point(|slot| slot.row.key <= cell.max_key()); + &slots[start..start + count] + } + + /// Iterates the rows `bucket` delivers within `cell`, given the catch-all at `deepest`. + /// + /// At `deepest` the run is the union of every bucket from there down, reordered by key and + /// priority, because the catch-all delivers everything the earlier cuts left behind. Past + /// `deepest` nothing is delivered. Rows arrive in key then priority order in every case. + pub(super) fn run( + &self, + bucket: Depth, + cell: MortonCell, + deepest: Depth, + ) -> impl Iterator { + match bucket.cmp(&deepest) { + Ordering::Less => BucketRun::Slice(self.cell_slots(bucket, cell).iter()), + Ordering::Equal => { + let mut gathered = Vec::new(); + + for natural in bucket..=Depth::MAX { + gathered.extend(self.cell_slots(natural, cell).iter().map(|slot| slot.row)); + } + + gathered.sort_unstable_by_key(|row| (row.key, row.priority)); + BucketRun::Gathered(gathered.into_iter()) + } + Ordering::Greater => BucketRun::Slice([].iter()), + } + } + + /// Counts the rows delivered by the cumulative schedule through `bucket`. + /// + /// The caller passes `bucket <= deepest`. At `deepest` the count is every captured row, because + /// the catch-all absorbs the buckets beyond it. + pub(super) fn delivered_through(&self, bucket: Depth, deepest: Depth) -> usize { + if bucket == deepest { + self.slots.len() + } else { + self.buckets[bucket].end + } + } + + /// Returns the deepest bucket holding a row, absent for an empty column. + pub(super) fn deepest_occupied(&self) -> Option { + self.slots.last().map(|slot| slot.bucket) + } + + /// Returns whether `cell` holds a row in a bucket past `cut`, one a deeper zoom still delivers. + /// + /// # Panics + /// + /// Panics when `cut` is [`Depth::MAX`], which no bucket lies past. + pub(super) fn occupied_past(&self, cut: Depth, cell: MortonCell) -> bool { + (cut.plus(1)..=Depth::MAX).any(|bucket| !self.cell_slots(bucket, cell).is_empty()) + } + + /// Returns the deepest prefix `key` shares with any captured key. + /// + /// An extension row assigned against this column starts no earlier than one level below the + /// returned depth, capped at [`Depth::MAX`]. Below that cap, the extension's key prefix differs + /// from every captured key's prefix at its assigned depth. Equal complete keys take the maximum + /// depth, where another subdivision cannot separate them. Returns [`None`] for an empty column. + /// + /// Each bucket's slots sort by key. For every depth, the keys sharing that depth's prefix with + /// `key` form one contiguous run of a sorted bucket, and a non-empty run holds one of the two + /// keys adjacent to `key`'s insertion point. The deepest shared prefix in the bucket is + /// therefore attained at one of those two keys. + pub(super) fn shared_depth(&self, key: MortonKey) -> Option { + (Depth::MIN..=Depth::MAX) + .filter_map(|bucket| { + let slots = self.bucket_slots(bucket); + let at = slots.partition_point(|slot| slot.row.key < key); + + [at.checked_sub(1), (at < slots.len()).then_some(at)] + .into_iter() + .flatten() + .map(|index| key.shared_depth(slots[index].row.key)) + .max() + }) + .max() + } +} + +impl HeapMemoryUsage for BucketColumn { + fn heap_memory_usage(&self) -> u64 { + self.memory_usage.get() as u64 + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/cut/mod.rs b/libs/@local/graph/atlas/src/serve/schedule/cut/mod.rs new file mode 100644 index 00000000000..10324537079 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/cut/mod.rs @@ -0,0 +1,341 @@ +//! Delivery queries at one resolved density offset. +//! +//! [`DeliverySchedule`] reads recorded or natural buckets through a [`BucketSchedule`]. Scoped +//! delivery combines every remaining natural bucket into the deepest cut, ordered by key and +//! priority. + +use core::{error::Error, fmt}; + +use hashql_core::id::Id as _; + +use super::{ + BucketSchedule, + column::{BucketColumn, ScheduleNode}, + scope::ScopeSchedule, +}; +use crate::{ + identity::NodeRowId, + morton::{Depth, MortonCell, MortonKey, Zoom}, + serve::world::{Layout, World}, +}; + +#[cfg(test)] +mod tests; + +/// A density offset that puts the deepest cut beyond the Morton key width. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) struct ScheduleWidthError { + /// The recorded schedule the offset was applied to. + pub schedule: BucketSchedule, + /// The requested density offset. + pub offset: Zoom, +} + +impl fmt::Display for ScheduleWidthError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + fmt, + "delivery-cut offset {} puts the deepest bucket {} + {} + {} beyond the {} \ + subdivisions a Morton key resolves", + self.offset, + self.schedule.max_tile_depth(), + self.schedule.span(), + self.offset, + Depth::MAX + ) + } +} + +impl Error for ScheduleWidthError {} + +/// Stable node rows and their per-bucket delivered counts. +#[derive(Debug, PartialEq, Eq)] +pub(crate) struct DeliveredNodes { + /// The delivered rows, bucket-major and key-ordered within each bucket. + pub rows: Vec, + /// The bucket the first run belongs to. + pub first_bucket: Depth, + /// How many rows each consecutive bucket from [`first_bucket`](Self::first_bucket) delivered. + /// + /// The counts partition [`rows`](Self::rows). An empty bucket keeps its zero. A run index is + /// therefore also a bucket offset. + pub runs: Vec, +} + +/// The origin of a schedule's buckets: the generation's record or a scope's own cascade. +#[derive(Debug, Copy, Clone)] +enum ScheduleSource<'schedule> { + /// The generation's recorded base assignments, shared by every corpus reader. + Corpus(&'schedule Layout), + /// A cascade assigned over one actor's visible rows. + Scope(&'schedule ScopeSchedule), +} + +/// Recorded base buckets or a visible cascade at one validated delivery cut. +/// +/// Corpus delivery preserves the base assignments. Scoped delivery clamps natural buckets into +/// its deepest cut. +#[derive(Debug, Copy, Clone)] +pub(crate) struct DeliverySchedule<'schedule> { + /// The origin of the scheduled buckets. + source: ScheduleSource<'schedule>, + /// Rows assigned against the source's keys, every one ranking below every scheduled row. + extension: Option<&'schedule BucketColumn>, + /// The bucket cuts after the density offset. + buckets: BucketSchedule, + /// The applied density offset, zero for corpus delivery. + offset: Zoom, +} + +impl<'schedule> DeliverySchedule<'schedule> { + /// Reads the generation's recorded buckets without a density offset. + /// + /// Base withdrawals leave the recorded delivery and aggregates unchanged. + pub(crate) const fn corpus(world: &'schedule World) -> Self { + Self { + source: ScheduleSource::Corpus(&world.layout), + extension: None, + buckets: world.schedule(), + offset: Zoom::MIN, + } + } + + /// Reads a visible cascade at the cut `offset` deepens. + /// + /// # Errors + /// + /// Returns [`ScheduleWidthError`] when the deepened cut leaves the Morton key width. + pub(super) fn bind( + schedule: &'schedule ScopeSchedule, + buckets: BucketSchedule, + offset: Zoom, + ) -> Result { + Ok(Self { + source: ScheduleSource::Scope(schedule), + extension: None, + buckets: buckets.offset(offset)?, + offset, + }) + } + + /// Adds rows assigned against this schedule's keys, delivered in the same key order. + /// + /// The extension holds rows the recorded assignment does not cover, and every extension row + /// ranks below every scheduled row, which is what lets delivery merge the two by key alone. + pub(super) const fn with_extension(mut self, extension: &'schedule BucketColumn) -> Self { + self.extension = Some(extension); + self + } + + /// Returns the bucket cuts after applying the density offset. + pub(crate) const fn buckets(&self) -> BucketSchedule { + self.buckets + } + + /// Returns the applied density offset, zero for corpus delivery. + pub(crate) const fn offset(&self) -> Zoom { + self.offset + } + + /// Returns the catch-all bucket, the deepest this schedule delivers into. + pub(crate) const fn deepest(&self) -> Depth { + self.buckets.deepest() + } + + /// Returns the cumulative delivery cut at a served zoom. + /// + /// # Panics + /// + /// Panics beyond the generation's deepest served zoom. + pub(crate) const fn cut_of(&self, zoom: Zoom) -> Depth { + self.buckets.cut(zoom) + } + + /// Appends one bucket's rows within `cell` to `rows`, returning how many it added. + fn run(&self, bucket: Depth, cell: MortonCell, rows: &mut Vec) -> usize { + let start = rows.len(); + let extension = self + .extension + .into_iter() + .flat_map(|column| column.run(bucket, cell, self.deepest())); + + match self.source { + ScheduleSource::Corpus(layout) => { + Self::merge(layout.base_run(bucket, cell), extension, rows); + } + ScheduleSource::Scope(schedule) => Self::merge( + schedule + .column() + .run(bucket, cell, self.deepest()) + .map(|row| (row.key, row.node)), + extension, + rows, + ), + } + + rows.len() - start + } + + /// Merges an extension run into a scheduled run, preserving key order. + /// + /// Interleaves the key-ordered inputs in one pass. + fn merge( + scheduled: impl IntoIterator, + extension: impl IntoIterator, + rows: &mut Vec, + ) { + let mut extension = extension.into_iter().peekable(); + + for (key, node) in scheduled { + // An equal key delivers the scheduled row first: every extension row ranks below every + // scheduled row. + while let Some(row) = extension.next_if(|row| row.key < key) { + rows.push(row.node); + } + + rows.push(node); + } + + rows.extend(extension.map(|row| row.node)); + } + + /// Counts rows delivered by the root's cumulative schedule. + pub(crate) fn root_delivered(&self) -> usize { + let cut = self.cut_of(Zoom::MIN); + let count = match self.source { + ScheduleSource::Corpus(layout) => layout.base_count_through(cut), + ScheduleSource::Scope(schedule) => { + schedule.column().delivered_through(cut, self.deepest()) + } + }; + + count + + self + .extension + .map_or(0, |column| column.delivered_through(cut, self.deepest())) + } + + /// Returns the deepest occupied delivery bucket, zero for an empty view. + pub(crate) fn min_resolution(&self) -> Depth { + let depth = match self.source { + ScheduleSource::Corpus(layout) => layout.base_deepest_occupied().unwrap_or(Depth::MIN), + ScheduleSource::Scope(schedule) => schedule + .column() + .deepest_occupied() + .map_or(Depth::MIN, |bucket| bucket.min(self.deepest())), + }; + + let extension = self + .extension + .and_then(BucketColumn::deepest_occupied) + .map_or(Depth::MIN, |bucket| bucket.min(self.deepest())); + depth.max(extension) + } + + /// Returns the Morton-child mask for rows beyond this zoom's cumulative cut. + /// + /// Bit `i` names child `i` in Morton order. At the deepest served zoom the mask is zero. + /// + /// # Panics + /// + /// Panics beyond the generation's deepest served zoom. + pub(crate) fn children(&self, zoom: Zoom, cell: MortonCell) -> u8 { + let cut = self.cut_of(zoom); + if cut == self.deepest() { + return 0; + } + + let Some(children) = cell.children() else { + return 0; + }; + + let mut bits = 0; + for (index, child) in children.into_iter().enumerate() { + let occupied = match self.source { + ScheduleSource::Corpus(layout) => { + (cut.plus(1)..=self.deepest()).any(|bucket| layout.base_occupied(bucket, child)) + } + ScheduleSource::Scope(schedule) => schedule.column().occupied_past(cut, child), + } || self + .extension + .is_some_and(|column| column.occupied_past(cut, child)); + + if occupied { + bits |= 1 << index; + } + } + + bits + } + + /// Returns a row's delivery bucket, absent when the schedule does not contain it. + /// + /// # Panics + /// + /// A corpus schedule panics when a base row's recorded position lies beyond the Morton order's + /// count, the condition [`Layout::base_bucket_of`] states. + pub(crate) fn bucket_of(&self, node: NodeRowId) -> Option { + match self.source { + ScheduleSource::Corpus(layout) => layout.base_bucket_of(node), + ScheduleSource::Scope(schedule) => schedule + .column() + .bucket_of(node) + .map(|bucket| bucket.min(self.deepest())), + } + .or_else(|| { + self.extension + .and_then(|column| column.bucket_of(node)) + .map(|bucket| bucket.min(self.deepest())) + }) + } + + /// Returns the first served zoom whose cumulative schedule delivers `node`. + /// + /// Absent when the schedule does not contain the row. + /// + /// # Panics + /// + /// A corpus schedule panics under the condition [`bucket_of`](Self::bucket_of) states. + pub(crate) fn first_zoom(&self, node: NodeRowId) -> Option { + self.bucket_of(node) + .map(|bucket| self.buckets.first_zoom(bucket)) + } + + /// Gathers only rows newly delivered at this zoom, in bucket-major order. + /// + /// The root includes its cumulative buckets. Deeper zooms include only their cut bucket. Runs + /// retain empty buckets. + /// + /// # Panics + /// + /// Panics beyond the generation's deepest served zoom. + pub(crate) fn delta(&self, zoom: Zoom, cell: MortonCell) -> DeliveredNodes { + let cut = self.cut_of(zoom); + let first = if zoom == Zoom::MIN { Depth::MIN } else { cut }; + + self.gather(first, cut, cell) + } + + /// Gathers the cumulative buckets at this zoom, in bucket-major order. + /// + /// # Panics + /// + /// Panics beyond the generation's deepest served zoom. + pub(crate) fn total(&self, zoom: Zoom, cell: MortonCell) -> DeliveredNodes { + self.gather(Depth::MIN, self.cut_of(zoom), cell) + } + + /// Collects buckets `first..=last` within `cell` into one bucket-major delivery. + fn gather(&self, first: Depth, last: Depth, cell: MortonCell) -> DeliveredNodes { + let mut rows = Vec::new(); + let runs = (first..=last) + .map(|bucket| self.run(bucket, cell, &mut rows)) + .collect(); + + DeliveredNodes { + rows, + first_bucket: first, + runs, + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/cut/tests.rs b/libs/@local/graph/atlas/src/serve/schedule/cut/tests.rs new file mode 100644 index 00000000000..9cd95c3f841 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/cut/tests.rs @@ -0,0 +1,193 @@ +//! Cases covering delivery at one cut: what each zoom adds, and what the catch-all absorbs. + +use hashql_core::id::{Id as _, IdSlice}; + +use super::{DeliveredNodes, DeliverySchedule}; +use crate::{ + file::{WriteInto as _, array::SizedColumn, generation::Generation, morton::read::MortonFile}, + identity::{BasePosition, Column, NodeRowId}, + math::Vec2, + morton::{Depth, MortonCell, MortonKey, Zoom}, + serve::{ + tests::fixture::{NODES, TamperFixture, secret}, + world::World, + }, +}; + +/// Reads the generation's recorded assignment straight from its artifacts. +/// +/// The cases compare delivery against this independent reading rather than against the schedule +/// that produced it. +/// +/// # Panics +/// +/// Panics if the Morton or row-of-position artifact fails to open, if a Morton position does not +/// fit [`BasePosition`], or if the row-of-position column is shorter than the Morton domain: every +/// position in `0..morton.count()` converts and then indexes it. The generation is any published +/// one rather than one [`World::open`] accepted, and the Morton file's own validation checks its +/// fenceposts' structure and the file length, not the position range. +fn records(generation: &Generation) -> impl IntoIterator { + let files = &generation.repository().files; + let morton: MortonFile = files + .morton + .open(generation) + .expect("should open the recorded keys"); + let rows: Column = files + .row_of_position + .open(generation) + .expect("should open the row permutation"); + + (0..morton.count()) + .map(|index| { + let position = BasePosition::from_u64(index); + ( + morton.bucket_of(position), + morton.code(position), + rows.view()[position], + ) + }) + .collect::>() +} + +/// Asserts that one cell's delivery at `zoom` matches the recorded assignment. +/// +/// Asserts cumulative delivery through the cut, the increment at this zoom and the child mask for +/// cells with deeper rows. +/// +/// # Panics +/// +/// Panics when `zoom` lies beyond the schedule's deepest served zoom, or when the cumulative +/// delivery through the cut, the incremental delivery at `zoom`, or the child mask of `cell` +/// disagrees with `records`. +#[track_caller] +fn assert_delivery( + schedule: DeliverySchedule<'_>, + records: &[(Depth, MortonKey, NodeRowId)], + zoom: Zoom, + cell: MortonCell, +) { + let cut = schedule.cut_of(zoom); + let first_delta = if zoom == Zoom::MIN { Depth::MIN } else { cut }; + for (actual, first) in [ + (schedule.total(zoom, cell), Depth::MIN), + (schedule.delta(zoom, cell), first_delta), + ] { + let rows = records + .iter() + .filter(|&&(bucket, key, _)| first <= bucket && bucket <= cut && cell.contains(key)) + .map(|&(_, _, node)| node) + .collect(); + let runs = (first..=cut) + .map(|bucket| { + records + .iter() + .filter(|&&(held, key, _)| held == bucket && cell.contains(key)) + .count() + }) + .collect(); + assert_eq!( + actual, + DeliveredNodes { + rows, + first_bucket: first, + runs + } + ); + } + + let mut children = 0; + if cut < schedule.deepest() + && let Some(cells) = cell.children() + { + for (index, child) in cells.into_iter().enumerate() { + if records + .iter() + .any(|&(bucket, key, _)| bucket > cut && child.contains(key)) + { + children |= 1 << index; + } + } + } + assert_eq!(schedule.children(zoom, cell), children); +} + +/// Corpus delivery reproduces the recorded assignment at every served zoom. +/// +/// The case checks every recorded key's own cell. +#[test] +fn corpus_recorded_delivery() { + let fixture = TamperFixture::publish("corpus-recorded-delivery"); + let world = + World::open(fixture.generation().clone(), &secret()).expect("should open the world"); + let records: Vec<_> = records(fixture.generation()).into_iter().collect(); + let schedule = DeliverySchedule::corpus(&world); + let buckets = world.schedule(); + + assert_eq!(schedule.deepest(), buckets.deepest()); + assert_eq!( + schedule.root_delivered(), + records + .iter() + .filter(|&&(bucket, _, _)| bucket <= buckets.cut(Zoom::MIN)) + .count() + ); + assert_eq!( + schedule.min_resolution(), + records + .iter() + .map(|&(bucket, _, _)| bucket) + .max() + .expect("should contain base rows") + ); + assert_eq!(schedule.bucket_of(NodeRowId::MAX), None); + assert_eq!(schedule.first_zoom(NodeRowId::MAX), None); + + for &(bucket, key, node) in &records { + assert_eq!(schedule.bucket_of(node), Some(bucket)); + assert_eq!(schedule.first_zoom(node), Some(buckets.first_zoom(bucket))); + for zoom in 0..=buckets.max_tile_depth().get() { + let zoom = Zoom::new(zoom).expect("should lie in the served zoom range"); + assert_delivery(schedule, &records, zoom, key.cell(Depth::from_zoom(zoom))); + } + } +} + +/// Corpus delivery reads the recorded keys after zeroing the coordinate artifact. +/// +/// The generation records the assignment rather than recomputing it from placements. +#[test] +fn corpus_recorded_keys() { + let fixture = TamperFixture::publish("corpus-recorded-keys"); + let records: Vec<_> = records(fixture.generation()).into_iter().collect(); + assert!( + records + .iter() + .any(|&(_, key, _)| key != MortonKey::new(0, 0)), + "should exercise nonzero recorded keys" + ); + let name = fixture + .generation() + .repository() + .files + .wire_coordinates + .name(); + let changed = fixture.tamper(&name, |path| { + let positions = + vec![Vec2::ZERO; usize::try_from(NODES).expect("should fit the node count")]; + let replacement = path.with_extension("replacement"); + let mut file = + std::fs::File::create(&replacement).expect("should create replacement coordinates"); + SizedColumn::new(IdSlice::::from_raw(&positions)) + .write_into(&mut file) + .expect("should write replacement coordinates"); + drop(file); + std::fs::rename(replacement, path).expect("should replace the coordinate artifact"); + }); + let world = World::open(changed, &secret()).expect("should open structurally valid artifacts"); + let schedule = DeliverySchedule::corpus(&world); + let zoom = world.schedule().max_tile_depth(); + + for &(_, key, _) in &records { + assert_delivery(schedule, &records, zoom, key.cell(Depth::from_zoom(zoom))); + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/mod.rs b/libs/@local/graph/atlas/src/serve/schedule/mod.rs new file mode 100644 index 00000000000..6927ae1efd2 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/mod.rs @@ -0,0 +1,54 @@ +//! Delivery schedules for recorded generations and visible node sets. +//! +//! [`BucketSchedule`] validates the served zoom range and bucket cuts against the Morton key width. +//! [`ScopeSchedule`] assigns natural [first-occupant buckets](crate::salt::lod::cascade) over +//! visible rows. [`ViewSchedule`] chooses, from the visibility declaration, between the recorded +//! base assignment, the shared base cascade and a scope's own cascade. [`DeliverySchedule`] reads +//! either source at one density offset, and scoped delivery combines every natural bucket past the +//! deepest cut into that catch-all. + +use core::{error::Error, fmt}; + +use crate::{math::Log2, morton::Zoom}; + +mod bucket; +mod column; +mod cut; +mod scope; +mod view; + +pub(crate) use self::{ + bucket::BucketSchedule, + cut::{DeliveredNodes, DeliverySchedule, ScheduleWidthError}, + scope::ScopeSchedule, + view::ViewSchedule, +}; + +/// The bit width of one Morton key axis. +const AXIS_BITS: u8 = 32; + +/// The recorded schedule exceeds the Morton key width. +/// +/// `max_tile_depth + span` lies beyond the 32 subdivisions a Morton key axis resolves, the +/// inequality [`LodConfig::deepest`](crate::salt::lod::stage::LodConfig::deepest) checks. +#[derive(Debug)] +pub(crate) struct ScheduleError { + /// The base-2 exponent of the cells per tile axis of the delivery cut. + span: Log2, + /// The deepest tile zoom the schedule names. + max_tile_depth: Zoom, +} + +impl fmt::Display for ScheduleError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + fmt, + "the recorded schedule needs {} + {} subdivisions where a Morton key axis resolves \ + {AXIS_BITS}", + self.max_tile_depth.get(), + self.span.get(), + ) + } +} + +impl Error for ScheduleError {} diff --git a/libs/@local/graph/atlas/src/serve/schedule/scope/mod.rs b/libs/@local/graph/atlas/src/serve/schedule/scope/mod.rs new file mode 100644 index 00000000000..8c4c6c22e53 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/scope/mod.rs @@ -0,0 +1,116 @@ +//! Natural delivery buckets over a captured visible node set. +//! +//! [`ScopeSchedule`] applies the [first-occupant cascade](crate::salt::lod::cascade) to the visible +//! rows under their [`NodePriority`](crate::serve::world::node_importance::NodePriority) order. +//! Hidden rows contribute to neither bucket assignment nor delivery counts. Natural buckets use the +//! complete key width, allowing every admissible density offset to share one schedule. + +use hashql_core::id::Id as _; + +use super::{ + BucketSchedule, + column::{BucketColumn, ScheduleNode}, + cut::{DeliverySchedule, ScheduleWidthError}, +}; +use crate::{ + allocator::HeapMemoryUsage, + identity::NodeRowId, + math::Bounds2, + morton::Zoom, + serve::{ + delta::epoch::Epoch, + visibility::VisibilityMask, + world::{Layout, layout::LayoutProvider, node_importance::ImportanceProvider as _}, + }, +}; + +#[cfg(test)] +mod tests; + +/// A first-occupant cascade over one visible node set. +/// +/// Rows sort by natural bucket, then by Morton key and priority. Construction captures the keys and +/// priorities at one [`Epoch`]. Delivery addresses the same stable [`NodeRowId`] domain as the +/// layout and identity providers. +#[derive(Debug)] +pub(crate) struct ScopeSchedule { + /// The captured rows in natural-bucket order, with the indexes delivery reads them through. + column: BucketColumn, +} + +impl ScopeSchedule { + /// Builds the captured visible cascade and its tight wire-frame extent. + /// + /// # Panics + /// + /// Panics if `layout` does not belong to the epoch's world or a visible placement is + /// non-finite. + pub(crate) fn of( + layout: &Layout, + epoch: &Epoch, + mask: &VisibilityMask, + ) -> (Self, Option) { + let (rows, bounds) = ScheduleNode::collect( + layout, + epoch, + mask, + (0..layout.node_count(epoch)).map(NodeRowId::from_usize), + ); + + (Self::over(rows), bounds) + } + + /// Builds the complete base cascade [saturated scopes](super::ViewSchedule::of) share. + /// + /// # Panics + /// + /// Panics if a base node has no position or priority. + pub(crate) fn from_base(layout: &Layout) -> Self { + let rows = (0..LayoutProvider::provide_node_count(layout)) + .map(|index| { + let node = NodeRowId::from_usize(index); + let position = layout + .provide_position(node) + .expect("should resolve the base node's position"); + let priority = layout + .provide_priority(node) + .expect("should resolve the base node's priority"); + + ScheduleNode::new(node, position, priority) + }) + .collect(); + + Self::over(rows) + } + + /// Borrows the assigned bucket column delivery reads from. + pub(super) const fn column(&self) -> &BucketColumn { + &self.column + } + + /// Assigns the cascade over `rows`, without an external key set to rank against. + fn over(rows: Vec) -> Self { + Self { + column: BucketColumn::new(rows, |_| None), + } + } + + /// Binds a density offset while preserving the generation's served zoom range. + /// + /// # Errors + /// + /// Returns [`ScheduleWidthError`] when the offset puts the deepest cut beyond the key width. + pub(crate) fn cut( + &self, + buckets: BucketSchedule, + offset: Zoom, + ) -> Result, ScheduleWidthError> { + DeliverySchedule::bind(self, buckets, offset) + } +} + +impl HeapMemoryUsage for ScopeSchedule { + fn heap_memory_usage(&self) -> u64 { + self.column.heap_memory_usage() + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/scope/tests.rs b/libs/@local/graph/atlas/src/serve/schedule/scope/tests.rs new file mode 100644 index 00000000000..4b09c247b57 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/scope/tests.rs @@ -0,0 +1,430 @@ +//! Cases covering the visible cascade: bucket assignment, delivery order and the density offset. + +use core::iter; + +use hashql_core::id::Id as _; +use proptest::{ + arbitrary::any, collection, prop_assert_eq, prop_oneof, property_test, strategy::Just, +}; +use uuid::Uuid; + +use super::{ScheduleNode, ScopeSchedule}; +use crate::{ + identity::{ImportanceRank, NodeRowId}, + math::Log2, + morton::{Depth, MortonCell, MortonKey, Zoom}, + postgres::id::ArchivedEntityId, + salt::lod::stage::LodConfig, + serve::{ + schedule::{BucketSchedule, DeliveredNodes, column::BucketColumn}, + world::node_importance::NodePriority, + }, +}; + +/// Builds a recorded schedule over zooms `0..=zoom` whose tiles have `2^span` cells per axis. +/// +/// `span` is the base-2 exponent of the cell count per tile axis, the schedule's [`Log2`] span. +/// +/// # Panics +/// +/// Panics if `span` reaches the shift width, `zoom` exceeds the key width, or `span + zoom` does. +fn grid(span: u8, zoom: u8) -> BucketSchedule { + BucketSchedule::new(LodConfig { + span: Log2::new(span).expect("should fit the exponent domain"), + max_tile_depth: Zoom::new(zoom).expect("should fit the zoom domain"), + }) + .expect("should fit the key width") +} + +/// Captured rows at `keys`, alternating the two priority forms. +/// +/// Alternating rank and identity priorities keeps the cases from passing under an implementation +/// that orders one form only, and the identity form's three webs exercise its tie-breaking. +#[expect( + clippy::integer_division_remainder_used, + reason = "fixture identities partition into three webs by index residue" +)] +fn rows(keys: &[u64]) -> Vec { + keys.iter() + .enumerate() + .map(|(index, &key)| { + let priority = if index.is_multiple_of(2) { + NodePriority::Rank(ImportanceRank::from_usize(keys.len() - index)) + } else { + NodePriority::Identity(ArchivedEntityId { + web_id: Uuid::from_u128((index % 3) as u128).into(), + entity_uuid: Uuid::from_u128(index as u128).into(), + }) + }; + ScheduleNode { + node: NodeRowId::from_u64(u64::from(u32::MAX) + index as u64), + key: MortonKey::from_bits(key), + priority, + } + }) + .collect() +} + +/// Returns the leading `depth` levels of `key`, two bits per level. +fn prefix(key: MortonKey, depth: u8) -> u64 { + if depth == 0 { + 0 + } else { + key.to_bits() >> (64 - 2 * u32::from(depth)) + } +} + +/// Computes the natural bucket of each row: the first depth without a better-ranked occupant. +/// +/// Rows whose complete keys are equal never separate, and take the deepest level, 32. +fn natural(rows: &[ScheduleNode]) -> Vec { + rows.iter() + .map(|row| { + (0..=32) + .find(|&depth| { + !rows.iter().any(|other| { + other.priority < row.priority + && prefix(other.key, depth) == prefix(row.key, depth) + }) + }) + .unwrap_or(32) + }) + .collect() +} + +/// Computes the delivery the cases expect over `interval` from the rows rather than the column. +fn delivery( + rows: &[ScheduleNode], + natural: &[u8], + deepest: u8, + interval: core::ops::RangeInclusive, + cell: MortonCell, +) -> DeliveredNodes { + let first_bucket = Depth::new(*interval.start()); + let mut ordered: Vec<_> = rows + .iter() + .zip(natural) + .filter(|&(row, &bucket)| interval.contains(&bucket.min(deepest)) && cell.contains(row.key)) + .map(|(row, &bucket)| (bucket.min(deepest), row)) + .collect(); + ordered.sort_unstable_by_key(|&(bucket, row)| (bucket, row.key, row.priority)); + let runs = interval + .map(|bucket| ordered.iter().filter(|(held, _)| *held == bucket).count()) + .collect(); + DeliveredNodes { + rows: ordered.into_iter().map(|(_, row)| row.node).collect(), + first_bucket, + runs, + } +} + +/// The best priority claims the root even when input and stable-row orders disagree. +#[test] +fn buckets_priority_order() { + let rows = rows(&[0, 0, 0, 0]); + let schedule = ScopeSchedule::over(rows.clone()); + let cut = schedule + .cut(grid(0, 2), Zoom::MIN) + .expect("should bind the cut"); + assert_eq!(cut.bucket_of(rows[2].node), Some(Depth::MIN)); + let root = MortonCell::new(Depth::MIN, 0, 0).expect("should construct the root"); + let total = cut.total(Zoom::new(2).expect("should fit the zoom"), root); + assert_eq!( + total.rows, + [rows[2].node, rows[0].node, rows[3].node, rows[1].node] + ); + assert_eq!(total.runs, [1, 0, 3]); +} + +/// The catch-all restores key order across distinct natural buckets. +#[test] +fn delivery_catch_all_order() { + let rows = rows(&[0, 1, 1 << 62, u64::MAX, 1 << 61]); + let natural = natural(&rows); + let schedule = ScopeSchedule::over(rows.clone()); + let cut = schedule + .cut(grid(0, 1), Zoom::MIN) + .expect("should bind the cut"); + let root = MortonCell::new(Depth::MIN, 0, 0).expect("should construct the root"); + let zoom = Zoom::new(1).expect("should fit the zoom"); + assert_eq!( + cut.total(zoom, root), + delivery(&rows, &natural, 1, 0..=1, root) + ); + assert_eq!(cut.children(zoom, root), 0); + assert_eq!(cut.total(zoom, root).runs.iter().sum::(), rows.len()); +} + +/// Removing the best row promotes the next visible priority to the root. +#[test] +fn buckets_hidden_priority() { + let rows = rows(&[0, 0, 0]); + let full = ScopeSchedule::over(rows.clone()); + let narrow = ScopeSchedule::over(rows[..2].to_vec()); + assert_eq!(full.column.bucket_of(rows[0].node), Some(Depth::MAX)); + assert_eq!(narrow.column.bucket_of(rows[0].node), Some(Depth::MIN)); + assert_eq!(narrow.column.bucket_of(rows[2].node), None); +} + +/// A root-only schedule delivers every row through its catch-all, at every valid offset. +#[test] +fn delivery_terminal_root() { + let rows = rows(&[0, 1, u64::MAX, u64::MAX]); + let schedule = ScopeSchedule::over(rows.clone()); + let root = MortonCell::new(Depth::MIN, 0, 0).expect("should construct the root"); + for offset in [0, 1, 32] { + let cut = schedule + .cut( + grid(0, 0), + Zoom::new(offset).expect("should fit the offset"), + ) + .expect("should bind the cut"); + assert_eq!(cut.root_delivered(), rows.len()); + assert_eq!(cut.total(Zoom::MIN, root), cut.delta(Zoom::MIN, root)); + assert_eq!(cut.children(Zoom::MIN, root), 0); + for row in &rows { + assert_eq!(cut.first_zoom(row.node), Some(Zoom::MIN)); + } + } +} + +/// An empty scope delivers nothing at every query, and still reports one run per bucket. +#[test] +fn delivery_empty() { + let schedule = ScopeSchedule::over(Vec::new()); + let root = MortonCell::new(Depth::MIN, 0, 0).expect("should construct the root"); + let cut = schedule + .cut(grid(2, 2), Zoom::MIN) + .expect("should bind the cut"); + assert_eq!(cut.root_delivered(), 0); + assert_eq!(cut.min_resolution(), Depth::MIN); + assert_eq!(cut.children(Zoom::MIN, root), 0); + assert_eq!(cut.first_zoom(NodeRowId::MIN), None); + assert_eq!( + cut.total(Zoom::MIN, root), + DeliveredNodes { + rows: Vec::new(), + first_bucket: Depth::MIN, + runs: vec![0, 0, 0] + } + ); +} + +/// A density offset is accepted exactly when `span + zoom + offset` fits the key width. +/// +/// An accepted cut reports the deepened schedule. +#[test] +fn cut_width_boundaries() { + let schedule = ScopeSchedule::over(Vec::new()); + for (span, zoom, offset, valid) in [ + (0, 0, 32, true), + (32, 0, 0, true), + (0, 32, 0, true), + (31, 0, 1, true), + (31, 0, 2, false), + (1, 31, 1, false), + (32, 0, 32, false), + ] { + let buckets = grid(span, zoom); + let offset = Zoom::new(offset).expect("should fit the offset"); + match schedule.cut(buckets, offset) { + Ok(cut) => { + assert!(valid, "should accept only cuts within the key width"); + assert_eq!(cut.offset(), offset); + assert_eq!(cut.buckets(), grid(span + offset.get(), zoom)); + assert_eq!(cut.deepest().get(), span + zoom + offset.get()); + } + Err(error) => { + assert!(!valid, "should refuse only cuts beyond the key width"); + assert_eq!(error.schedule, buckets); + assert_eq!(error.offset, offset); + } + } + } +} + +/// Asking for a cut beyond the schedule's deepest served zoom is refused. +#[test] +#[should_panic(expected = "the schedule serves zooms 0..=max_tile_depth")] +fn cut_zoom_outside_schedule() { + let schedule = ScopeSchedule::over(Vec::new()); + let cut = schedule + .cut(grid(0, 0), Zoom::MIN) + .expect("should bind the cut"); + cut.cut_of(Zoom::MAX); +} + +/// A separate extension column agrees with a cascade over the combined key set. +#[property_test] +fn delivery_extension_column( + #[strategy = collection::vec(prop_oneof![Just(0), Just(u64::MAX), any::()], 0..30)] + keys: Vec, + #[strategy = 0_u8..4] max_zoom: u8, + #[strategy = 0_u8..4] offset: u8, +) { + let rows = rows(&keys); + let natural = natural(&rows); + let (base, extension): (Vec<_>, Vec<_>) = rows + .iter() + .copied() + .partition(|row| matches!(row.priority, NodePriority::Rank(_))); + let base = ScopeSchedule::over(base); + for row in &rows { + let expected = rows + .iter() + .filter(|row| matches!(row.priority, NodePriority::Rank(_))) + .map(|held| row.key.shared_depth(held.key)) + .max(); + prop_assert_eq!(base.column.shared_depth(row.key), expected); + } + let extension = BucketColumn::new(extension, |key| base.column.shared_depth(key)); + let cut = base + .cut( + grid(0, max_zoom), + Zoom::new(offset).expect("should fit the offset"), + ) + .expect("should bind the cut") + .with_extension(&extension); + let deepest = max_zoom + offset; + prop_assert_eq!(cut.offset().get(), offset); + prop_assert_eq!(cut.buckets(), grid(offset, max_zoom)); + prop_assert_eq!( + cut.root_delivered(), + natural + .iter() + .filter(|&&bucket| bucket.min(deepest) <= offset) + .count() + ); + prop_assert_eq!( + cut.min_resolution().get(), + natural.iter().copied().max().unwrap_or(0).min(deepest) + ); + for (row, &bucket) in rows.iter().zip(&natural) { + prop_assert_eq!( + cut.bucket_of(row.node), + Some(Depth::new(bucket.min(deepest))) + ); + prop_assert_eq!( + cut.first_zoom(row.node).map(Zoom::get), + Some(bucket.min(deepest).saturating_sub(offset)) + ); + } + for zoom in 0..=max_zoom { + let typed_zoom = Zoom::new(zoom).expect("should fit the zoom"); + let cut_depth = zoom + offset; + let cells = iter::once( + MortonCell::new(Depth::new(zoom), 0, 0).expect("should construct the origin cell"), + ) + .chain(rows.iter().map(|row| row.key.cell(Depth::new(zoom)))); + for cell in cells { + prop_assert_eq!( + cut.total(typed_zoom, cell), + delivery(&rows, &natural, deepest, 0..=cut_depth, cell) + ); + let first = if zoom == 0 { 0 } else { cut_depth }; + prop_assert_eq!( + cut.delta(typed_zoom, cell), + delivery(&rows, &natural, deepest, first..=cut_depth, cell) + ); + let mut children = 0; + if cut_depth < deepest { + for (index, child) in cell + .children() + .expect("should have children below the key width") + .into_iter() + .enumerate() + { + if rows.iter().zip(&natural).any(|(row, &bucket)| { + bucket.min(deepest) > cut_depth && child.contains(row.key) + }) { + children |= 1 << index; + } + } + } + prop_assert_eq!(cut.children(typed_zoom, cell), children); + } + } +} + +/// Each delivery query agrees with direct first-occupant assignment over mixed priorities. +#[property_test] +fn delivery_laws( + #[strategy = collection::vec(prop_oneof![Just(0), Just(u64::MAX), any::()], 0..40)] + keys: Vec, + #[strategy = 0_u8..5] span: u8, + #[strategy = 0_u8..5] max_zoom: u8, + #[strategy = 0_u8..5] offset: u8, +) { + let rows = rows(&keys); + let natural = natural(&rows); + let schedule = ScopeSchedule::over(rows.clone()); + let cut = schedule + .cut( + grid(span, max_zoom), + Zoom::new(offset).expect("should fit the offset"), + ) + .expect("should bind the cut"); + let deepest = span + max_zoom + offset; + let root_cut = span + offset; + prop_assert_eq!(cut.offset().get(), offset); + prop_assert_eq!(cut.buckets(), grid(root_cut, max_zoom)); + prop_assert_eq!( + cut.root_delivered(), + natural + .iter() + .filter(|&&bucket| bucket.min(deepest) <= root_cut) + .count() + ); + prop_assert_eq!( + cut.min_resolution().get(), + natural.iter().copied().max().unwrap_or(0).min(deepest) + ); + prop_assert_eq!(cut.bucket_of(NodeRowId::MIN), None); + for (row, &bucket) in rows.iter().zip(&natural) { + prop_assert_eq!( + cut.bucket_of(row.node).map(Depth::get), + Some(bucket.min(deepest)) + ); + prop_assert_eq!( + cut.first_zoom(row.node).map(Zoom::get), + Some(bucket.min(deepest).saturating_sub(span + offset)) + ); + } + for zoom in 0..=max_zoom { + let depth = Depth::new(zoom); + let mut cells = + vec![MortonCell::new(depth, 0, 0).expect("should construct the origin cell")]; + cells.extend(rows.iter().map(|row| row.key.cell(depth))); + let cut_depth = zoom + span + offset; + let typed_zoom = Zoom::new(zoom).expect("should fit the zoom"); + for cell in cells { + let total = cut.total(typed_zoom, cell); + prop_assert_eq!(total.runs.iter().sum::(), total.rows.len()); + prop_assert_eq!( + total, + delivery(&rows, &natural, deepest, 0..=cut_depth, cell) + ); + let first = if zoom == 0 { 0 } else { cut_depth }; + prop_assert_eq!( + cut.delta(typed_zoom, cell), + delivery(&rows, &natural, deepest, first..=cut_depth, cell) + ); + let mut expected = 0; + if cut_depth < deepest { + for (index, child) in cell + .children() + .expect("should have children below the key width") + .iter() + .enumerate() + { + if rows.iter().zip(&natural).any(|(row, &bucket)| { + bucket.min(deepest) > cut_depth && child.contains(row.key) + }) { + expected |= 1 << index; + } + } + } + prop_assert_eq!(cut.children(typed_zoom, cell), expected); + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/view/mod.rs b/libs/@local/graph/atlas/src/serve/schedule/view/mod.rs new file mode 100644 index 00000000000..690c91b1863 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/view/mod.rs @@ -0,0 +1,212 @@ +//! Captured view scheduling under corpus or scoped visibility. +//! +//! [`ViewSchedule`] preserves the visibility declaration. Saturated scopes share the generation's +//! complete base cascade and retain their extension rows separately. Narrow scopes recompute the +//! cascade over every admitted placement. + +use alloc::sync::Arc; + +use hashql_core::id::Id as _; + +use super::{ + column::{BucketColumn, ScheduleNode}, + cut::{DeliverySchedule, ScheduleWidthError}, + scope::ScopeSchedule, +}; +use crate::{ + allocator::HeapMemoryUsage, + identity::NodeRowId, + math::Bounds2, + morton::Zoom, + serve::{ + delta::epoch::Epoch, + density::ViewOccupancy, + visibility::{VisibilityKind, VisibilityMask}, + world::World, + }, +}; + +#[cfg(test)] +mod tests; + +/// The cascade a view delivers from, and what it shares with the generation. +/// +/// These cases differ in what they recompute. Corpus and saturated views reuse an assignment +/// the generation already holds and keep only their extension rows, while a narrow scope pays for +/// its own cascade over the rows it admits. +#[derive(Debug)] +enum ScheduleData { + /// The generation's recorded base assignment, plus rows added since it was recorded. + Corpus { + /// Rows assigned against the recorded base keys. + extension: BucketColumn, + }, + /// A scope that admits every base row, each of which the epoch places. + /// + /// Both conditions hold for every base row, the conjunction [`ViewSchedule::of`] tests. It + /// shares the generation's complete base cascade. + Saturated { + /// The generation's shared base cascade. + base: Arc, + /// Rows assigned against that cascade's keys. + extension: BucketColumn, + }, + /// A scope that withholds or lacks a placement for some base row. + /// + /// Its cascade is assigned over exactly the placements it admits, base and added alike. + Scoped(ScopeSchedule), +} + +/// A captured node schedule associated with its base generation. +/// +/// Corpus schedules preserve recorded base assignments through withdrawals. Scoped schedules +/// include only the mask's admitted placements at construction. Extension rows use captured keys +/// and priorities in both modes. +#[derive(Debug)] +pub(crate) struct ViewSchedule { + /// The generation the schedule captures. + world: Arc, + /// The cascade the view delivers from. + data: ScheduleData, + /// The wire-frame bounds captured according to the view kind. + /// + /// Corpus and saturated views join the recorded base bounds with the measured extent of their + /// added rows. A narrow scope measures the extent of exactly the rows it admits. + bounds: Option, +} + +impl ViewSchedule { + /// Resolves the visibility declaration against one captured publication. + /// + /// Under a corpus declaration, the view reads the generation's recorded base assignment and + /// assigns only the rows added since. A scope saturates when the mask admits every base row + /// and the epoch places each. Its admitted base rows are then the fitted rows at their fitted + /// positions and ranks, which makes its cascade over them the generation's base cascade. A + /// saturated scope therefore shares that cascade and assigns only its added rows. A scope + /// withholding or missing a base row is narrow and assigns its own cascade over exactly the + /// rows it admits. + /// + /// # Panics + /// + /// Panics if `world` does not belong to `epoch` or a gathered placement is non-finite. + pub(crate) fn of(world: Arc, epoch: &Epoch, mask: &VisibilityMask) -> Self { + let count = world.layout.node_count(epoch); + + let saturated = || { + (NodeRowId::MIN..world.layout.index.base_node_bound()).all(|node| { + mask.visible_node(node).is_some() && world.layout.position(epoch, node).is_some() + }) + }; + + let (data, bounds) = match mask.kind() { + VisibilityKind::Scope if !saturated() => { + let (schedule, bounds) = ScopeSchedule::of(&world.layout, epoch, mask); + (ScheduleData::Scoped(schedule), bounds) + } + kind @ (VisibilityKind::Corpus | VisibilityKind::Scope) => { + let (rows, bounds) = ScheduleNode::collect( + &world.layout, + epoch, + mask, + world.layout.index.base_node_bound()..NodeRowId::from_usize(count), + ); + let bounds = world + .layout + .base_bounds() + .into_iter() + .chain(bounds) + .reduce(Bounds2::union); + + let data = match kind { + VisibilityKind::Corpus => ScheduleData::Corpus { + extension: BucketColumn::new(rows, |key| { + world.layout.base_shared_depth(key) + }), + }, + VisibilityKind::Scope => { + let base = Arc::clone(world.base_scope_schedule()); + let extension = + BucketColumn::new(rows, |key| base.column().shared_depth(key)); + ScheduleData::Saturated { base, extension } + } + }; + (data, bounds) + } + }; + + Self { + world, + data, + bounds, + } + } + + /// Returns the wire-frame bounds captured for this view, or [`None`] for an empty schedule. + /// + /// For a corpus or saturated view this is the recorded base bounds + /// ([`Layout::base_bounds`](crate::serve::world::Layout::base_bounds)) joined with the measured + /// extent of the rows added since the base. The base part comes from the record rather than + /// from the rows, and withdrawals leave it in place. For a narrow scope it is the tight extent + /// of the admitted rows' placements, measured at construction. + pub(crate) const fn bounds(&self) -> Option { + self.bounds + } + + /// Builds the occupied-cell profile of the captured scope. + /// + /// Corpus schedules return [`None`]. Every scoped schedule returns [`Some`], including an empty + /// scope's zero profile. All captured keys contribute, independently of delivery cut. + /// + /// # Complexity + /// + /// O(n log n) time and O(n) temporary storage for n scheduled rows. + #[must_use] + #[tracing::instrument(skip_all, fields(generation = %self.world.generation().id()))] + pub(crate) fn occupancy(&self) -> Option { + let (base, extension) = match &self.data { + ScheduleData::Corpus { .. } => return None, + ScheduleData::Saturated { base, extension } => (base.column(), Some(extension)), + ScheduleData::Scoped(schedule) => (schedule.column(), None), + }; + let mut keys: Vec<_> = base + .keys() + .chain(extension.into_iter().flat_map(BucketColumn::keys)) + .collect(); + Some(ViewOccupancy::of(&mut keys)) + } + + /// Binds the scoped density offset or reads the corpus's recorded cuts. + /// + /// Corpus delivery keeps its recorded cuts for every `offset`. + /// + /// # Errors + /// + /// Returns [`ScheduleWidthError`] when a scoped offset exceeds the Morton key width. + pub(crate) fn cut(&self, offset: Zoom) -> Result, ScheduleWidthError> { + match &self.data { + ScheduleData::Corpus { extension } => { + Ok(DeliverySchedule::corpus(&self.world).with_extension(extension)) + } + ScheduleData::Saturated { base, extension } => base + .cut(self.world.schedule(), offset) + .map(|cut| cut.with_extension(extension)), + ScheduleData::Scoped(schedule) => schedule.cut(self.world.schedule(), offset), + } + } +} + +impl HeapMemoryUsage for ViewSchedule { + /// Returns the heap bytes this view alone holds. + /// + /// Every view over the generation shares a saturated view's base cascade, and charging it to + /// one view would count it once per request. The extension, and a narrow scope's own + /// schedule, are this view's alone and count. + fn heap_memory_usage(&self) -> u64 { + match &self.data { + ScheduleData::Corpus { extension } | ScheduleData::Saturated { base: _, extension } => { + extension.heap_memory_usage() + } + ScheduleData::Scoped(scoped) => scoped.heap_memory_usage(), + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/schedule/view/tests.rs b/libs/@local/graph/atlas/src/serve/schedule/view/tests.rs new file mode 100644 index 00000000000..8fc24f26473 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/schedule/view/tests.rs @@ -0,0 +1,202 @@ +//! Cases covering the three view schedules: corpus, saturated scope and narrow scope. + +use alloc::sync::Arc; +use core::assert_matches; + +use arc_swap::Guard; +use hashql_core::id::Id as _; +use rand::{SeedableRng as _, rngs::StdRng}; +use type_system::principal::actor::{ActorId, ActorType}; +use uuid::Uuid; + +use super::{ScheduleData, ViewSchedule}; +use crate::{ + bitset::CompressedBitSet, + identity::NodeRowId, + math::Bounds2, + morton::{Depth, MortonCell, MortonKey, Zoom}, + salt::lod::stage::WIRE_FRAME, + serve::{ + delta::{Delta, epoch::Epoch}, + density::ViewOccupancy, + tests::fixture::{NODES, TamperFixture, secret}, + visibility::{VisibilityActor, VisibilityKind, VisibilityMask}, + world::World, + }, +}; + +/// Returns a non-administrator principal, the standing every case here resolves under. +fn actor() -> VisibilityActor { + VisibilityActor { + id: ActorId::new(Uuid::nil(), ActorType::Machine), + instance_admin: false, + } +} + +/// Builds a scoped mask admitting rows `first..NODES`. +/// +/// `first` of zero admits every row and reaches the saturated schedule, and `NODES` admits none. +fn mask(first: u64) -> VisibilityMask { + let mut nodes = CompressedBitSet::default(); + for index in first..NODES { + nodes.insert(NodeRowId::new(index)); + } + VisibilityMask::partial(actor(), nodes, CompressedBitSet::default()) +} + +/// Publishes a fixture generation and opens its world with one captured publication. +/// +/// # Panics +/// +/// Panics if publishing or opening the generation fails, or constructing the delta does. +fn world(name: &str) -> (TamperFixture, Arc, Epoch) { + let fixture = TamperFixture::publish(name); + let world = Arc::new( + World::open(fixture.generation().clone(), &secret()).expect("should open the world"), + ); + let delta = Delta::new(Arc::clone(&world), StdRng::seed_from_u64(17)) + .expect("should construct the delta"); + let epoch = Epoch::from(Guard::from_inner(Arc::new(delta))); + (fixture, world, epoch) +} + +/// A corpus mask and a scope admitting every row reach different schedules. +/// +/// The saturated scope shares the generation's one base cascade. +#[test] +fn dispatch_saturated_partial() { + let (_fixture, world, epoch) = world("schedule-dispatch-saturated"); + let full = VisibilityMask::full(actor()); + let partial = mask(0); + assert_eq!(full.kind(), VisibilityKind::Corpus); + assert_eq!(partial.kind(), VisibilityKind::Scope); + let corpus = ViewSchedule::of(Arc::clone(&world), &epoch, &full); + let first = ViewSchedule::of(Arc::clone(&world), &epoch, &partial); + let second = ViewSchedule::of(Arc::clone(&world), &epoch, &partial); + assert_matches!(corpus.data, ScheduleData::Corpus { .. }); + assert_eq!(corpus.occupancy(), None); + assert!(first.occupancy().is_some()); + let (ScheduleData::Saturated { base: first, .. }, ScheduleData::Saturated { base: second, .. }) = + (&first.data, &second.data) + else { + panic!("should share the saturated scoped cascade"); + }; + assert!(Arc::ptr_eq(first, second)); + assert!(Arc::ptr_eq(first, world.base_scope_schedule())); + let cut = corpus.cut(Zoom::MAX).expect("should keep recorded cuts"); + assert_eq!(cut.offset(), Zoom::MIN); + assert_eq!(cut.buckets(), world.schedule()); + assert_eq!(cut.deepest(), world.schedule().deepest()); + let error = first + .cut(world.schedule(), Zoom::MAX) + .expect_err("should refuse a scoped offset beyond the key width"); + assert_eq!(error.offset, Zoom::MAX); + assert_eq!(error.schedule, world.schedule()); +} + +/// Withholding one row reaches the narrow schedule. +/// +/// The schedule then delivers neither that row nor a bucket for it. +#[test] +fn dispatch_missing_row() { + let (_fixture, world, epoch) = world("schedule-dispatch-missing-row"); + let view = ViewSchedule::of(Arc::clone(&world), &epoch, &mask(1)); + assert_matches!(view.data, ScheduleData::Scoped(_)); + let cut = view.cut(Zoom::MIN).expect("should bind the scope"); + let root = MortonCell::new(Depth::MIN, 0, 0).expect("should construct the root"); + assert_eq!(cut.bucket_of(NodeRowId::MIN), None); + assert_eq!( + cut.total(world.schedule().max_tile_depth(), root) + .rows + .len(), + usize::try_from(NODES).expect("should fit the fixture count") - 1 + ); +} + +/// Measures the tight extent of rows `first..NODES` from the layout rather than the schedule. +fn extent(world: &World, epoch: &Epoch, first: u64) -> Option { + Bounds2::from_points( + (first..NODES).filter_map(|index| world.layout.position(epoch, NodeRowId::new(index))), + ) +} + +/// Each view kind's bounds come from the source its construction names. +/// +/// The corpus view returns the recorded base bounds. This epoch has no added rows, and the +/// saturated view returns those same bounds. Its assertion compares them with an independent +/// measurement of every base placement. The narrow scope measures exactly the rows it admits, and +/// the empty scope has none. +#[test] +fn bounds_admission() { + let (_fixture, world, epoch) = world("schedule-bounds"); + + let corpus = ViewSchedule::of(Arc::clone(&world), &epoch, &VisibilityMask::full(actor())); + assert_eq!(corpus.bounds(), world.layout.base_bounds()); + assert!(corpus.bounds().is_some()); + + let saturated = ViewSchedule::of(Arc::clone(&world), &epoch, &mask(0)); + assert_matches!(saturated.data, ScheduleData::Saturated { .. }); + assert_eq!(saturated.bounds(), extent(&world, &epoch, 0)); + + let scoped = ViewSchedule::of(Arc::clone(&world), &epoch, &mask(NODES - 2)); + assert_matches!(scoped.data, ScheduleData::Scoped(_)); + assert_eq!(scoped.bounds(), extent(&world, &epoch, NODES - 2)); + assert_ne!(scoped.bounds(), saturated.bounds()); + + let empty = ViewSchedule::of(Arc::clone(&world), &epoch, &mask(NODES)); + assert_eq!(empty.bounds(), None); +} + +/// A scope's occupancy profile counts exactly the keys it admits, an empty scope included. +#[test] +fn occupancy_admission() { + let (_fixture, world, epoch) = world("schedule-occupancy"); + for first in [0, 1, NODES] { + let view = ViewSchedule::of(Arc::clone(&world), &epoch, &mask(first)); + let occupancy = view + .occupancy() + .expect("a scope should have an occupancy profile"); + let mut keys: Vec<_> = (first..NODES) + .filter_map(|index| world.layout.position(&epoch, NodeRowId::new(index))) + .map(|position| { + let [x, y] = WIRE_FRAME.quantize(position); + MortonKey::new(x, y) + }) + .collect(); + assert_eq!(occupancy, ViewOccupancy::of(&mut keys)); + } +} + +/// Building a schedule from one generation's world against another's publication is refused. +#[test] +#[should_panic(expected = "layout must belong to the epoch's world")] +fn dispatch_foreign_world() { + let (_first_fixture, _first, epoch) = world("schedule-first-world"); + let (_second_fixture, second, _second_epoch) = world("schedule-second-world"); + ViewSchedule::of(second, &epoch, &mask(NODES)); +} + +/// The recorded keys' shared-prefix lookup agrees with the maximum over every recorded key. +/// +/// The queries lie inside the set, outside it and one bit away from it. +#[test] +fn shared_depth_recorded_keys() { + let (_fixture, world, _epoch) = world("schedule-shared-depth"); + let root = MortonCell::new(Depth::MIN, 0, 0).expect("should construct the root"); + let keys: Vec<_> = (Depth::MIN..=world.schedule().deepest()) + .flat_map(|bucket| world.layout.base_run(bucket, root).map(|(key, _)| key)) + .collect(); + let queries = [MortonKey::from_bits(0), MortonKey::from_bits(u64::MAX)] + .into_iter() + .chain(keys.iter().copied()) + .chain( + keys.iter() + .map(|key| MortonKey::from_bits(key.to_bits() ^ 1)), + ); + for query in queries { + assert_eq!( + world.layout.base_shared_depth(query), + keys.iter().map(|&key| query.shared_depth(key)).max() + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/secret.rs b/libs/@local/graph/atlas/src/serve/secret.rs new file mode 100644 index 00000000000..0b590f307f4 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/secret.rs @@ -0,0 +1,36 @@ +use core::ops::Deref; + +use crate::integrity::SecretHexBytes; + +/// A serving deployment's 32-byte secret. +/// +/// Authorization sealing and row-id obfuscation derive keys from this value under distinct domain +/// separators. A fixed secret preserves row mappings for equal generation identities and labels. +/// Authorization also salts each authority's derivation with fresh randomness. +/// +/// Supply unpredictable key material: the type accepts every 32-byte value and checks no entropy. +/// Configuration uses the canonical lowercase hexadecimal form provided by [`SecretHexBytes`]. +/// [`core::fmt::Debug`] redacts the value, and each owned buffer zeroizes on drop. This zeroization +/// does not cover copies obtained through byte access. +#[derive(Debug, Clone, PartialEq, Eq)] +#[repr(transparent)] +pub(crate) struct ServeSecret(SecretHexBytes<32>); + +impl ServeSecret { + /// The secret's width in bytes. + const LENGTH: usize = size_of::(); +} + +impl Deref for ServeSecret { + type Target = SecretHexBytes<{ Self::LENGTH }>; + + fn deref(&self) -> &Self::Target { + &self.0 + } +} + +impl From> for ServeSecret { + fn from(bytes: SecretHexBytes<{ Self::LENGTH }>) -> Self { + Self(bytes) + } +} diff --git a/libs/@local/graph/atlas/src/serve/tests/fixture.rs b/libs/@local/graph/atlas/src/serve/tests/fixture.rs new file mode 100644 index 00000000000..723b4daae4d --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/tests/fixture.rs @@ -0,0 +1,1074 @@ +//! A synthetic generation for the serving tests. +//! +//! The default corpus contains the nodes in [`COORDINATES`], the edges in [`ENDPOINTS`] and the +//! ontology rows in [`TYPES`]. Its delivery artifacts are built with the same writers as a fit: +//! [`Lod::build`] for the base order and four permutation columns, [`QuadTree::build`], +//! [`Postings::build`], [`Adjacency::build`] and [`IdentityTable::write_into`] for the three +//! identity tables. This stages eight chosen points through the artifact writers without running +//! the fit's ingest, embedding or placement stages. Every artifact opened by serving therefore has +//! the format produced by a fit. Artifacts that serving never opens are placeholders bound to the +//! digest of their own bytes. The seal requires every manifest name, while open verifies each file +//! it reads against its digest. +//! +//! Tampered fixtures republish the generation with one file rewritten and re-bound to the digest of +//! its new bytes. This isolates structural validation from digest verification. A file edited in +//! place fails the digest check first. +//! +//! Larger edgeless corpora support checks whose samples exhaust the default fixture. + +use core::borrow::Borrow; +use std::io::Write as _; + +// serving unit tests edit individual fixture artifacts. +#[cfg(test)] +use camino::Utf8Path; +use camino::Utf8PathBuf; +use hashql_core::id::{IdSlice, IdVec}; +use smallvec::SmallVec; +use type_system::ontology::id::VersionedUrl; +use uuid::Uuid; + +use super::super::secret::ServeSecret; +// serving unit tests rebind edited artifacts to their metadata. +#[cfg(test)] +use crate::file::{digest_file, repository::FileName}; +use crate::{ + dataset::{ + DatasetOrigin, + auxiliary::{Icon, Label, OwnedLegend}, + }, + file::{ + WriteInto as _, + array::SizedColumn, + generation::{Generation, GenerationRoot, StagedGeneration}, + identity::Row, + repository::{Artifact, Binding, RepositoryVersion}, + salt::{ + SaltFiles, SaltRepository, artifact, + metadata::{ + ClassifierEvidence, Evidence, LandmarkEvidence, Placement, PolicyEvidence, + RankingOrigin, Reproducibility, SaltMetadata, Snapshot, + }, + }, + }, + identity::{EdgeRowId, NodeRowId, OntologyRowId}, + integrity::{SecretHexBytes, Sha256Digest}, + math::{ + AffinityCurve, DNonNegative, FinitePointField, Vec2, d_positive, non_negative, nz, + open_unit_fraction, positive, unit_fraction, + }, + postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, + salt::{ + adjacency::Adjacency, + embedding::{CardEmbeddingStats, EmbedderFingerprint}, + fit::{ + FitConfig, PlacementOptions, + prepare::{identity::IdentityTable, norm::NormSpotCheck}, + }, + importance::{ConstantImportance, ImportanceSignal as _, RankingConfig}, + knn::recall::RecallSpotCheck, + landmark::select::SelectionOptions, + lod::{ + quad::QuadTree, + rank::RankInputs, + stage::{Lod, MortonColumn}, + }, + postings::build::Postings, + relation::BuildMeasurements, + }, +}; + +/// The canonical coordinates, one per node row. +/// +/// Distinct points with extent on both axes. The extent fits a world frame, and the distinct points +/// give the base order a spatial order the cases can check. +const COORDINATES: [Vec2; 8] = [ + Vec2::new(-2.0, -1.0), + Vec2::new(-1.0, 2.0), + Vec2::new(0.0, 0.0), + Vec2::new(1.0, -2.0), + Vec2::new(2.0, 1.0), + Vec2::new(3.0, 3.0), + Vec2::new(-3.0, 0.5), + Vec2::new(0.5, -3.0), +]; + +/// The node rows of the corpus. +// document, schedule, delta and world unit tests construct row domains from this count. +#[cfg(test)] +pub(crate) const NODES: u64 = COORDINATES.len() as u64; + +/// The edge list, `[source row, target row]` in edge row order. +/// +/// Row 2 is a self-loop. Rows 3 and 4 are a reciprocal pair over the same two nodes. +pub(crate) const ENDPOINTS: [[NodeRowId; 2]; 6] = [ + [NodeRowId::new(0), NodeRowId::new(1)], + [NodeRowId::new(1), NodeRowId::new(2)], + [NodeRowId::new(2), NodeRowId::new(2)], + [NodeRowId::new(5), NodeRowId::new(7)], + [NodeRowId::new(7), NodeRowId::new(5)], + [NodeRowId::new(3), NodeRowId::new(6)], +]; + +/// The edge rows of the corpus. +// document, delta and world unit tests construct row domains from this count. +#[cfg(test)] +pub(crate) const EDGES: u64 = ENDPOINTS.len() as u64; + +/// Each ontology row's direct parents, in ontology row order. +/// +/// Rows 0 and 1 are the node types, row 1 a child of row 0. Row 2 is the link type. +const PARENTS: [&[OntologyRowId]; 3] = [&[], &[OntologyRowId::new(0)], &[]]; + +/// The ontology rows of the corpus. +pub(crate) const TYPES: u64 = PARENTS.len() as u64; + +/// The link type, the representative of every edge row. +const LINK_TYPE: OntologyRowId = OntologyRowId::new(2); + +/// The reproducibility seed of the base order. +const SEED: u64 = 0x5E4E; + +/// The edge-domain seed offset. +/// +/// Link entities own ids disjoint from node ids, as the store's are. +pub(crate) const EDGE_SEED: u8 = 64; + +/// The suite's serving secret, an arbitrary value of the secret's width. +const SECRET: &[u8] = b"61746c61732d746573742d73657276652d7365637265742d33322d6279746573"; + +/// Returns the serving secret every suite open uses. +pub(crate) fn secret() -> ServeSecret { + ServeSecret::from( + SecretHexBytes::from_encoded_bytes(SECRET).expect("should decode the fixture secret"), + ) +} + +/// Removes any existing fixture directory and returns its path. +/// +/// # Panics +/// +/// Panics if the temporary directory path is not UTF-8 or an existing fixture directory cannot be +/// removed. +fn scratch(name: &str) -> Utf8PathBuf { + let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) + .expect("the temp directory is UTF-8") + .join(format!( + "hash-graph-atlas-serve-{}-{name}", + std::process::id() + )); + if let Err(error) = std::fs::remove_dir_all(&dir) + && error.kind() != std::io::ErrorKind::NotFound + { + panic!("should remove the existing fixture directory: {error}"); + } + dir +} + +/// Derives one synthetic entity identity from `seed`, distinct per seed byte. +fn entity_id_of(seed: u8) -> ArchivedEntityId { + ArchivedEntityId { + web_id: Uuid::from_bytes([seed; 16]).into(), + entity_uuid: Uuid::from_bytes([seed ^ 0xFF; 16]).into(), + } +} + +/// Returns the versioned type URL behind ontology row `row`. +/// +/// Ontology identities key each row by the uuid its URL derives, as the store's do. +fn fixture_type_url(row: u64) -> String { + format!("https://example.com/types/fixture-{row}/v/1") +} + +/// Derives the ontology identity of row `row` from its fixture URL. +fn ontology_id_of(row: u64) -> ArchivedOntologyTypeUuid { + let url: VersionedUrl = fixture_type_url(row) + .parse() + .expect("the fixture URL parses"); + ArchivedOntologyTypeUuid::from_url(&url) +} + +/// Builds sequential entity identities from `seed`. +/// +/// The identities take the seed bytes `seed..seed + rows`, and `seed + rows` must not exceed 256. +/// +/// # Panics +/// +/// Panics if `rows` exceeds 256. +fn entity_table(rows: u64, seed: u8) -> IdentityTable { + let mut table = IdentityTable::new(); + for row in 0..rows { + let row = u8::try_from(row).expect("fixture row counts fit u8"); + table.push(entity_id_of(seed + row)); + } + table +} + +/// Builds a table of `rows` ontology identities. +fn ontology_table(rows: u64) -> IdentityTable { + let mut table = IdentityTable::new(); + for row in 0..rows { + table.push(ontology_id_of(row)); + } + table +} + +/// Assigns each node row one direct type, alternating between the two node types. +fn node_types(nodes: u64) -> IdVec> { + (0..nodes) + .map(|row| SmallVec::from_buf_and_len([OntologyRowId::new(row & 1), LINK_TYPE], 1)) + .collect() +} + +/// Builds the direct-parent column of [`PARENTS`]. +fn parents() -> IdVec> { + PARENTS + .iter() + .map(|parents| parents.iter().copied().collect()) + .collect() +} + +/// Returns the fit configuration the document echoes. +/// +/// The seed and the schedule are what [`Lod::build`] consumes. The remaining fields describe +/// stages a synthetic generation does not run and carry values their validators accept. +fn config() -> FitConfig { + FitConfig { + seed: SEED, + selection: SelectionOptions { + maximum_count: nz!(2), + .. + }, + curve: AffinityCurve::new(positive!(1.577), positive!(0.895)), + placement: PlacementOptions::LandmarkBaseline, + ranking: RankingConfig::ConstantColumns, + .. + } +} + +/// Stages a placeholder for an artifact the serving open never opens. +/// +/// The bytes are the artifact's own name, and the binding carries their digest. +/// +/// # Panics +/// +/// Panics if staging the placeholder fails. +fn placeholder(staging: &StagedGeneration, artifact: A) -> Binding { + staging + .stage_with(artifact, |writer| { + let name = A::NAME; + let bytes = name.as_str().as_bytes(); + writer.write_all(bytes)?; + Ok(Sha256Digest::of(bytes)) + }) + .expect("the placeholder stages") +} + +/// The delivery structure of the corpus, derived as a fit derives it. +struct Delivery<'corpus> { + /// The node row count: `types`, `lod`, `quad` and `postings` all cover this many rows. + nodes: u64, + /// Each node row's direct types. + types: IdVec>, + /// The base order and its permutation columns. + lod: Lod, + /// The quadtree cut over the base order. + quad: QuadTree, + /// The type postings over the base order. + postings: Postings, + /// The incident-edge adjacency over `endpoints`. + adjacency: Adjacency, + /// The canonical points `lod` derives from. + coordinates: &'corpus [Vec2], + /// The edge list `adjacency` and the edge artifacts derive from. + endpoints: &'corpus [[NodeRowId; 2]], +} + +/// Derives delivery artifacts from the supplied points and edges. +/// +/// # Panics +/// +/// Panics if a row cannot be encoded as a fixture identity, a point is non-finite, or an artifact +/// builder rejects the corpus. +fn derive_over<'corpus>( + coordinates: &'corpus [Vec2], + endpoints: &'corpus [[NodeRowId; 2]], + config: &FitConfig, +) -> Delivery<'corpus> { + let nodes_usize = coordinates.len(); + let nodes = u64::try_from(nodes_usize).expect("fixture node counts fit u64"); + let points = IdSlice::::from_raw(coordinates); + let points = FinitePointField::new(points).expect("the corpus points are finite"); + let node_ids: IdVec = (0..nodes) + .map(|row| entity_id_of(u8::try_from(row).expect("fixture row counts fit u8"))) + .collect(); + let types = node_types(nodes); + let parents = parents(); + + let importance = ConstantImportance.derive(nodes_usize); + let priority = IdVec::from_domain(0.0_f32, &importance); + let inputs = RankInputs::new(&importance, &priority, &node_ids) + .expect("the corpus rank columns agree with the coordinates"); + let lod = Lod::build(points, inputs, config.seed, config.lod).expect("the lod builds"); + let quad = QuadTree::build(&lod, &types, config.lod).expect("the quadtree builds"); + let postings = + Postings::build(&types, &lod.row_of_position, &parents).expect("the postings build"); + let adjacency = Adjacency::build(nodes_usize, endpoints); + + Delivery { + nodes, + types, + lod, + quad, + postings, + adjacency, + coordinates, + endpoints, + } +} + +/// Derives the delivery structure over the corpus constants. +/// +/// # Panics +/// +/// Panics when an artifact builder rejects the corpus under `config`, as +/// [`derive_over`] does. +fn derive(config: &FitConfig) -> Delivery<'static> { + derive_over(&COORDINATES, &ENDPOINTS, config) +} + +/// Generates distinct finite points with alternating vertical coordinates. +// world::tests uses larger corpora to distinguish sampled from exhaustive validation. +#[cfg(test)] +fn spread_coordinates(nodes: u8) -> Vec { + (0..nodes) + .map(|row| Vec2::new(f32::from(row), f32::from(row & 1))) + .collect() +} + +/// Stages node identities with each row's first direct type and an empty label. +/// +/// # Panics +/// +/// Panics if a row has no direct type, a row cannot be encoded as a fixture identity, or staging +/// fails. +fn stage_nodes( + staging: &StagedGeneration, + nodes: u64, + types: &IdSlice>, +) -> Binding { + let legends: Vec = types + .iter() + .map(|types| OwnedLegend::new(types[0], Label::EMPTY)) + .collect(); + + staging + .stage_with(artifact::NodeIdentities, |writer| { + entity_table::(nodes, 0) + .write_into(legends.iter().map(Borrow::borrow), writer) + }) + .expect("should stage the node identities") +} + +/// Stages every artifact of the generation and returns the manifest's typed entries. +/// +/// # Panics +/// +/// Panics if an identity cannot be encoded, a node has no direct type, or an artifact cannot be +/// staged. +fn stage_files( + staging: &StagedGeneration, + Delivery { + nodes, + types, + lod, + quad, + postings, + adjacency, + coordinates, + endpoints, + }: &Delivery<'_>, +) -> SaltFiles { + let edge_legend = OwnedLegend::new(LINK_TYPE, Label::EMPTY); + // Row 0 carries an icon and its child row 1 resolves to it through the closure. + let icons = [Icon::new("fixture-icon"), Icon::empty(), Icon::empty()]; + let edges = u64::try_from(endpoints.len()).expect("fixture edge counts fit u64"); + + SaltFiles { + representations: placeholder(staging, artifact::Representations), + card_embeddings: placeholder(staging, artifact::CardEmbeddings), + card_hashes: placeholder(staging, artifact::CardHashes), + knn: placeholder(staging, artifact::Knn), + semantic: placeholder(staging, artifact::Semantic), + landmarks: placeholder(staging, artifact::Landmarks), + classifier: placeholder(staging, artifact::Classifier), + policy: placeholder(staging, artifact::Policy), + attraction: placeholder(staging, artifact::Attraction), + protection: placeholder(staging, artifact::Protection), + coordinates: staging + .stage( + artifact::Coordinates, + SizedColumn::new(IdSlice::::from_raw(coordinates)), + ) + .expect("the coordinates stage"), + morton: staging + .stage( + artifact::Morton, + &MortonColumn { + fenceposts: &lod.fenceposts, + codes: lod.codes.as_raw(), + }, + ) + .expect("the morton column stages"), + quad: staging + .stage(artifact::Quad, quad) + .expect("the quadtree stages"), + postings: staging + .stage(artifact::Postings, postings) + .expect("the postings stage"), + wire_coordinates: staging + .stage( + artifact::WireCoordinates, + SizedColumn::new(&lod.coordinates), + ) + .expect("the wire coordinates stage"), + rank_of_position: staging + .stage( + artifact::RankOfPosition, + SizedColumn::new(&lod.rank_of_position), + ) + .expect("the rank-of-position column stages"), + position_of_rank: staging + .stage( + artifact::PositionOfRank, + SizedColumn::new(&lod.position_of_rank), + ) + .expect("the position-of-rank column stages"), + position_of_row: staging + .stage( + artifact::PositionOfRow, + SizedColumn::new(&lod.position_of_row), + ) + .expect("the position-of-row column stages"), + row_of_position: staging + .stage( + artifact::RowOfPosition, + SizedColumn::new(&lod.row_of_position), + ) + .expect("the row-of-position column stages"), + node_identities: stage_nodes(staging, *nodes, types), + edge_identities: staging + .stage_with(artifact::EdgeIdentities, |writer| { + entity_table::(edges, EDGE_SEED) + .write_into(core::iter::repeat_n(&*edge_legend, endpoints.len()), writer) + }) + .expect("the edge identities stage"), + ontology_identities: staging + .stage_with(artifact::OntologyIdentities, |writer| { + ontology_table(TYPES).write_into(icons, writer) + }) + .expect("the ontology identities stage"), + edge_endpoints: staging + .stage_with(artifact::EdgeEndpoints, |writer| { + SizedColumn::new(IdSlice::::from_raw(endpoints)) + .write_into(writer) + }) + .expect("the endpoint column stages"), + adjacency: staging + .stage(artifact::Adjacency, adjacency) + .expect("the adjacency stages"), + projector: None, + reviewed_verdicts: None, + annotation_corpus: None, + annotation_embeddings: None, + annotation_hashes: None, + } +} + +/// Builds the metadata document of the synthetic generation. +/// +/// The snapshot counts are `delivery`'s own node and edge counts, and the lod, quad and postings +/// sections are the builds' own measurements. The sections of the stages a synthetic generation +/// does not run carry empty readings. +/// +/// # Panics +/// +/// Panics if a corpus count exceeds `usize`. +fn metadata(config: FitConfig, delivery: &Delivery<'_>) -> SaltMetadata { + let nodes = usize::try_from(delivery.nodes).expect("fixture node counts fit usize"); + let edges = delivery.endpoints.len(); + let edges_u64 = u64::try_from(edges).expect("fixture edge counts fit u64"); + let types = usize::try_from(TYPES).expect("fixture type counts fit usize"); + let self_references = delivery + .endpoints + .iter() + .filter(|&&[source, target]| source == target) + .count(); + let lod = delivery.lod.measurements(config.lod); + + SaltMetadata { + snapshot: Snapshot { + axes: None, + nodes: delivery.nodes, + edges: edges_u64, + ontology_types: TYPES, + }, + reproducibility: Reproducibility { + config, + embedder: EmbedderFingerprint::new(Sha256Digest::of(b"serve fixture embedder")), + prior: None, + }, + dataset: Some(DatasetOrigin::Memory), + placement: Placement::LandmarkBaseline, + ranking: RankingOrigin::ConstantColumns, + evidence: Evidence { + cards: CardEmbeddingStats { + reused: 0, + embedded: types, + }, + norm: NormSpotCheck { + rows: nodes, + sampled_rows: 0, + tolerance: d_positive!(1.0e-4), + defect_rate: open_unit_fraction!(0.01), + confidence: open_unit_fraction!(0.999), + defects: Vec::new(), + }, + recall: RecallSpotCheck { + sampled_rows: 0, + neighbours_per_row: 0, + matched: 0, + expected: 0, + deviation: DNonNegative::ZERO, + minimum_recall: unit_fraction!(0.89), + resolution: DNonNegative::ZERO, + confidence: open_unit_fraction!(0.99), + }, + landmarks: LandmarkEvidence { + selected: 0, + retained: 0, + layout_epochs: nz!(1), + }, + policy: PolicyEvidence { + relations: 1, + overridden: 0, + }, + classifier: ClassifierEvidence::Supplied { + source: Sha256Digest::of(b"serve fixture classifier"), + }, + relations: BuildMeasurements { + pruning_threshold: non_negative!(0.0), + retained_edges: edges, + pruned_edges: 0, + retained_mass: DNonNegative::ZERO, + pruned_mass: DNonNegative::ZERO, + self_references, + multi_typed_edges: vec![edges_u64], + }, + lod, + quad: delivery.quad.measurements(), + postings: delivery.postings.measurements(), + projector: None, + }, + } +} + +/// One published synthetic generation and the root it tampers into. +/// +/// Tampering republishes the edited artifacts under their own digests. The untampered generation is +/// never written to. Dropping the fixture attempts to remove the root and every generation under +/// it, and logs a warning rather than panicking when the removal fails. +pub(crate) struct TamperFixture { + /// The temporary root this fixture publishes every generation into. + root: GenerationRoot, + /// The untampered generation. + generation: Generation, +} + +impl TamperFixture { + /// Publishes the synthetic generation under a root named `name`. + /// + /// # Panics + /// + /// Panics if constructing or publishing the fixture fails. + // serving unit tests use the default layout. + #[cfg(test)] + pub(crate) fn publish(name: &str) -> Self { + Self::with_config(name, config()) + } + + /// Builds and publishes the corpus using the supplied fixture configuration. + /// + /// # Panics + /// + /// Panics if constructing an artifact or publishing its generation fails. + fn with_config(name: &str, config: FitConfig) -> Self { + let delivery = derive(&config); + + Self::publish_delivery(name, config, &delivery) + } + + /// Publishes an edgeless corpus of distinct finite points. + /// + /// # Panics + /// + /// Panics if the corpus cannot be built or its generation cannot be published and opened. + // world::tests uses larger corpora to distinguish sampled from exhaustive validation. + #[cfg(test)] + pub(crate) fn publish_with_nodes(name: &str, nodes: u8) -> Self { + let config = config(); + let coordinates = spread_coordinates(nodes); + let delivery = derive_over(&coordinates, &[], &config); + + Self::publish_delivery(name, config, &delivery) + } + + /// Publishes the derived artifacts and their metadata under a temporary root. + /// + /// # Panics + /// + /// Panics if an artifact cannot be written or the generation cannot be published and opened. + #[expect(clippy::significant_drop_tightening, reason = "false-positive")] + fn publish_delivery(name: &str, config: FitConfig, delivery: &Delivery<'_>) -> Self { + let root = GenerationRoot::new(scratch(name)).expect("the root should open"); + let staging = root.stage().expect("the staging should create"); + let repository = SaltRepository { + version: RepositoryVersion::V2, + files: stage_files(&staging, delivery), + metadata: metadata(config, delivery), + }; + let published = staging.seal(&repository).expect("the staging should seal"); + + let generation = root + .open(published.id()) + .expect("the published generation should open"); + + Self { root, generation } + } + + /// Borrows the untampered generation. + pub(crate) const fn generation(&self) -> &Generation { + &self.generation + } + + /// Republishes the generation with `edit` applied to the artifact `name`. + /// + /// # Panics + /// + /// Panics if staging, copying, re-digesting, sealing or reopening the republished generation + /// fails. + // serving unit tests alter individual serialized artifacts. + #[cfg(test)] + #[expect(clippy::significant_drop_tightening, reason = "false-positive")] + pub(crate) fn tamper(&self, name: &FileName, edit: impl FnOnce(&Utf8Path)) -> Generation { + let staging = self.root.stage().expect("the staging should create"); + + for file in self.generation.repository().files.files() { + std::fs::copy( + self.generation.path_of(&file.name), + staging.path_of(&file.name), + ) + .expect("a published file should copy into the staging"); + } + + edit(&staging.path_of(name)); + + let mut document = serde_json::to_value(self.generation.repository()) + .expect("the manifest should serialize"); + let entries = document + .get_mut("files") + .and_then(serde_json::Value::as_object_mut) + .expect("the manifest holds its files object"); + + for entry in entries.values_mut().filter(|entry| !entry.is_null()) { + let name: FileName = entry + .get("name") + .cloned() + .map(serde_json::from_value) + .expect("an entry names its file") + .expect("an entry's name is a file name"); + + let digest = digest_file(staging.path_of(&name)).expect("a staged file should digest"); + entry["hash"] = serde_json::to_value(digest).expect("a digest should serialize"); + } + + let repository: SaltRepository = + serde_json::from_value(document).expect("the rebound manifest should deserialize"); + + let published = staging + .seal(&repository) + .expect("the edited staging should seal"); + + self.root + .open(published.id()) + .expect("the republished generation should open") + } +} + +impl Drop for TamperFixture { + fn drop(&mut self) { + if let Err(error) = std::fs::remove_dir_all(self.root.path()) + && error.kind() != std::io::ErrorKind::NotFound + { + tracing::warn!(error = %error, "failed to remove serving fixture directory"); + } + } +} + +// serving unit tests alter serialized fixture artifacts. +#[cfg(test)] +pub(crate) use self::tamper::{ + constant_u32_column, constant_u64_column, respan_adjacency, retarget_postings_points, + retarget_quad_root, set_row_position, shorten_endpoints, shorten_entities, shorten_ontology, + shorten_u32_column, +}; + +#[cfg(test)] +mod tamper { + use core::borrow::Borrow as _; + use std::{io::Write as _, path::Path}; + + use hashql_core::id::{IdSlice, IdVec}; + use zerocopy::U64; + + use super::{entity_table, ontology_table}; + use crate::{ + dataset::auxiliary::{Icon, Label, OwnedLegend}, + file::{ + ArtifactFile as _, WriteInto as _, + array::{ArrayFile, ArrayVariant, Dim, SizedArrayWriter, SizedColumn}, + identity::{Key, Row}, + postings::{read::PostingsFile, write::Regions}, + quad::{Node, TypeSets, read::QuadFile}, + }, + identity::{BasePosition, EdgeRowId, NodeRowId, OntologyRowId}, + salt::{adjacency::Adjacency, fit::prepare::identity::IdentityTable}, + }; + + /// Reopens a published artifact for rewriting. + /// + /// Sealing dropped the write permission. A tamper therefore lifts it before truncating the + /// file. + /// + /// # Panics + /// + /// Panics if reading or changing the artifact's permissions fails, or recreating the file does. + fn recreate_writable(path: impl AsRef) -> std::fs::File { + let mut permissions = std::fs::metadata(path.as_ref()) + .expect("the published artifact should stat") + .permissions(); + #[expect( + clippy::permissions_set_readonly_false, + reason = "tests rewrite their own scratch files" + )] + permissions.set_readonly(false); + + std::fs::set_permissions(path.as_ref(), permissions).expect("the permissions should set"); + std::fs::File::create(path).expect("the published artifact rewrites") + } + + /// Overwrites one identity artifact with a hand-built table and one payload per row. + /// + /// # Panics + /// + /// Panics if reopening the artifact or writing the table fails. + fn rewrite_identities<'payload, R, K>( + path: impl AsRef, + table: &IdentityTable, + payloads: impl IntoIterator, + ) where + R: Row, + K: Key, + { + let mut file = recreate_writable(path); + let _digest = table + .write_into(payloads, &mut file) + .expect("the identities should write"); + } + + /// Rewrites an entity identity artifact with `rows` sequential fixture ids from `seed`. + /// + /// Every row's legend names ontology row 0 under the empty label. + /// + /// # Panics + /// + /// Panics if `rows` exceeds 256 or rewriting the artifact fails. + pub(crate) fn shorten_entities(path: impl AsRef, rows: u64, seed: u8) { + let legend = OwnedLegend::new(OntologyRowId::new(0), Label::EMPTY); + let legends = core::iter::repeat_n( + legend.borrow(), + usize::try_from(rows).expect("fixture row counts fit usize"), + ); + rewrite_identities(path, &entity_table::(rows, seed), legends); + } + + /// Rewrites the ontology identity artifact with `rows` fixture type uuids and no icons. + /// + /// # Panics + /// + /// Panics if `rows` does not fit `usize` or rewriting the artifact fails. + pub(crate) fn shorten_ontology(path: impl AsRef, rows: u64) { + let icons = core::iter::repeat_n( + Icon::empty(), + usize::try_from(rows).expect("fixture row counts fit usize"), + ); + rewrite_identities(path, &ontology_table(rows), icons); + } + + /// Rewrites the endpoint column with `pairs`, dropping the fixture's rows beyond them. + /// + /// # Panics + /// + /// Panics if rewriting the artifact fails. + pub(crate) fn shorten_endpoints(path: impl AsRef, pairs: &[[NodeRowId; 2]]) { + let file = recreate_writable(path); + let _digest = SizedColumn::new(IdSlice::::from_raw(pairs)) + .write_into(file) + .expect("the endpoint column should write"); + } + + /// Rewrites the adjacency artifact over the same edges, spanning `rows` node rows. + /// + /// Each endpoint must lie in the `rows` node domain. Building from the same endpoints with a + /// different node count preserves the paired runs, the edge-domain column bound and exactly one + /// slot per edge per direction, while changing the node domain relative to the other artifacts. + /// + /// # Panics + /// + /// Panics if a computed adjacency slot lies outside its allocated column, or rewriting the + /// artifact fails. + pub(crate) fn respan_adjacency( + path: impl AsRef, + rows: usize, + endpoints: &[[NodeRowId; 2]], + ) { + let file = recreate_writable(path); + let _digest = Adjacency::build(rows, endpoints) + .write_into(file) + .expect("the adjacency should write"); + } + + /// Rewrites the quad artifact with the root's subtree count set to `points`. + /// + /// The topology, the runs, and the type sets are the published ones. The quad format validates + /// the header, the fenceposts, and the child indexes, and never the subtree counts, which + /// is why `open` must. + /// + /// # Panics + /// + /// Panics if opening the published quad artifact fails, it holds no root, or rewriting it + /// fails. + pub(crate) fn retarget_quad_root(path: impl AsRef, points: u32) { + // The mapping ends before the rewrite: the file backs the slices read here. + let (mut nodes, sets) = { + let quad = QuadFile::open(path.as_ref()).expect("the published quad artifact opens"); + let nodes = quad.nodes().to_vec(); + let sets: Vec> = (0..nodes.len()) + .map(|node| { + let node = u32::try_from(node).expect("fixture node tables fit u32"); + quad.type_set(node).iter().map(|id| id.get()).collect() + }) + .collect(); + + (nodes, TypeSets::from_sets(&sets)) + }; + + let root = *nodes.first().expect("the fixture quad holds a root"); + let run = root.run(); + let length = u32::try_from(run.end - run.start).expect("fixture runs fit u32"); + nodes[0] = Node::new(root.children(), run.start, length, points); + + let file = recreate_writable(path); + let mut file = std::io::BufWriter::new(file); + crate::file::quad::write::write_regions(&nodes, &sets, &mut file) + .expect("the quad regions write"); + file.flush().expect("the quad artifact flushes"); + } + + /// Rewrites the postings artifact with its point domain set to `points`. + /// + /// Setting `points` to a count different from the coordinate column's length creates a domain + /// mismatch. The format ties the dense sets and the direct fenceposts to that domain as well. + /// The dense sets rebuild over the new domain, because every frame's own domain count restates + /// the header's and open checks the agreement. The direct fenceposts resize to cover it - + /// truncating drops the stranded runs' ids, growing appends empty runs. The flags, lists and + /// parents restate the published regions. + /// + /// # Panics + /// + /// Panics if opening the published postings artifact fails, a count or fencepost does not fit + /// `usize`, a resized column breaks the fencepost law, or rewriting the artifact fails. + pub(crate) fn retarget_postings_points(path: impl AsRef, points: u64) { + // The file backs the slices read here, and the mapping ends before the rewrite. Every + // region therefore copies into build vocabulary first. + let postings = + PostingsFile::open(path.as_ref()).expect("the published postings artifact opens"); + + let published = postings.flags(); + let mut flags = crate::bitset::DenseBitSlice::new_empty( + usize::try_from(postings.types()).expect("fixture type domains fit usize"), + ); + let mut dense_sets = crate::bitset::DenseBitSliceArray::new_empty( + usize::try_from(points).expect("fixture point domains fit usize"), + usize::try_from(published.count()).expect("fixture dense counts fit usize"), + ); + for (rank, type_row) in published.iter().enumerate() { + flags.insert(type_row); + for member in postings.dense_sets()[rank].iter() { + dense_sets[rank].insert(member); + } + } + + let posts_len = usize::try_from(points).expect("fixture point domains fit usize") + 1; + let mut direct_posts = postings + .direct_posts() + .iter() + .map(|post| usize::try_from(post.get()).expect("fixture posts fit usize")) + .collect::>(); + let mut direct_ids = postings.direct_ids().to_vec(); + if direct_posts.len() > posts_len { + direct_posts.truncate(posts_len); + let close = *direct_posts + .last() + .expect("the fencepost region anchors at zero"); + direct_ids.truncate(close); + } else { + let close = *direct_posts + .last() + .expect("the fencepost region anchors at zero"); + direct_posts.resize(posts_len, close); + } + + let lists = crate::runs::Runs::from_parts( + IdVec::from_raw(postings.list_posts().to_vec()), + postings.list_entries().to_vec(), + ) + .expect("the published list columns satisfy the fencepost law"); + let parents = crate::runs::Runs::from_parts( + IdVec::from_raw(postings.parent_posts().to_vec()), + postings.parent_ids().to_vec(), + ) + .expect("the published parent columns satisfy the fencepost law"); + let direct = crate::runs::Runs::from_parts( + IdVec::from_raw( + direct_posts + .iter() + .map(|&post| U64::new(post as u64)) + .collect(), + ), + direct_ids, + ) + .expect("the resized direct columns satisfy the fencepost law"); + + drop(postings); + + let file = recreate_writable(path); + let mut file = std::io::BufWriter::new(file); + crate::file::postings::write::write_regions( + Regions { + flags: &flags, + lists: &lists, + dense_sets: &dense_sets, + parents: &parents, + direct: &direct, + }, + &mut file, + ) + .expect("the postings regions write"); + file.flush().expect("the postings artifact flushes"); + } + + /// Rewrites a little-endian `u32` column with `rows` ascending values. + /// + /// The values do not matter to the check under test - `open` compares lengths - and ascending + /// keeps the file a plausible permutation prefix rather than a shape no producer would + /// write. + /// + /// # Panics + /// + /// Panics if a written row index does not fit `u32` or writing the column fails. + #[expect( + clippy::little_endian_bytes, + reason = "the array format's `U32Le` columns are little-endian bytes" + )] + pub(crate) fn shorten_u32_column(path: impl AsRef, rows: u64) { + let file = recreate_writable(path); + let mut writer = SizedArrayWriter::new(file, ArrayVariant::U32Le, &[Dim::new(rows)]) + .expect("the header writes"); + for row in 0..rows { + let value = u32::try_from(row).expect("fixture rows fit u32"); + writer + .write_row(&value.to_le_bytes()) + .expect("the row writes"); + } + writer.finish().expect("the column seals"); + } + + /// Changes one row's stored base position without changing the column's other entries. + /// + /// # Panics + /// + /// Panics if the file cannot be read as a base-position column, `row` is outside its domain, or + /// rewriting the column fails. + pub(crate) fn set_row_position(path: impl AsRef, row: NodeRowId, position: BasePosition) { + // retain an owned column copy for rewriting after this mapping closes. + let mut rows = { + let file = ArrayFile::open(path.as_ref()).expect("should open the position column"); + file.column::() + .expect("should contain base positions indexed by node row") + .to_vec() + }; + rows[row] = position; + + let _digest = SizedColumn::new(&rows) + .write_into(recreate_writable(path)) + .expect("should rewrite the position column"); + } + + /// Rewrites a little-endian `u32` column with `rows` copies of `value`. + /// + /// Constant values preserve the column's format and requested length. Over two or more rows + /// they cannot form a permutation, which the roundtrip sample rejects. + /// + /// # Panics + /// + /// Panics if writing the column fails. + #[expect( + clippy::little_endian_bytes, + reason = "the array format's `U32Le` columns are little-endian bytes" + )] + pub(crate) fn constant_u32_column(path: impl AsRef, rows: u64, value: u32) { + let file = recreate_writable(path); + let mut writer = SizedArrayWriter::new(file, ArrayVariant::U32Le, &[Dim::new(rows)]) + .expect("the header writes"); + for _ in 0..rows { + writer + .write_row(&value.to_le_bytes()) + .expect("the row writes"); + } + writer.finish().expect("the column seals"); + } + + /// Rewrites a little-endian `u64` column with `rows` copies of `value`. + /// + /// The row column is the one `u64` column of the base order. + /// + /// # Panics + /// + /// Panics if writing the column fails. + #[expect( + clippy::little_endian_bytes, + reason = "the array format's `U64Le` columns are little-endian bytes" + )] + pub(crate) fn constant_u64_column(path: impl AsRef, rows: u64, value: u64) { + let file = recreate_writable(path); + let mut writer = SizedArrayWriter::new(file, ArrayVariant::U64Le, &[Dim::new(rows)]) + .expect("the header writes"); + for _ in 0..rows { + writer + .write_row(&value.to_le_bytes()) + .expect("the row writes"); + } + writer.finish().expect("the column seals"); + } +} diff --git a/libs/@local/graph/atlas/src/serve/tests/mod.rs b/libs/@local/graph/atlas/src/serve/tests/mod.rs new file mode 100644 index 00000000000..d011da256ae --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/tests/mod.rs @@ -0,0 +1 @@ +pub(crate) mod fixture; diff --git a/libs/@local/graph/atlas/src/serve/visibility/mod.rs b/libs/@local/graph/atlas/src/serve/visibility/mod.rs new file mode 100644 index 00000000000..4db5518f72f --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/visibility/mod.rs @@ -0,0 +1,218 @@ +//! What one actor may see of a generation. +//! +//! Serving artifacts contain every row. Serving resolves a [`VisibilityMask`] against the +//! permission store before delivery and reuses it for the life of a cache entry. The mask records +//! the node and edge rows the actor may receive, together with the principal and instance- +//! administrator status used for resolution. Filter construction receives property-protection +//! configuration as a separate input. The mask freezes the resulting permission-store decision. +//! `cache` defines when requests reuse, refresh or refuse it. +//! +//! A mask is either the whole corpus or an explicit row set, and +//! [`VisibilityKind`] names which. That distinction is the actor's admitted policy rather than an +//! observation about the row count. A scoped actor whose set happens to cover every row is +//! still scoped. + +use hash_graph_store::filter::protection::{ + PropertyProtectionFilter, PropertyProtectionFilterConfig, +}; +use hashql_core::id::Id; +use type_system::principal::actor::ActorId; + +use crate::{ + allocator::HeapMemoryUsage, + bitset::CompressedBitSet, + identity::{EdgeRowId, NodeRowId}, +}; + +/// A visible row set over one row domain. +#[derive(Debug, Clone, PartialEq, Eq)] +enum Rows { + /// Every row of the domain is visible. + Full, + /// Exactly the set rows are visible. + Mask(CompressedBitSet), +} + +impl Rows { + /// Tests admission, rejecting masked rows outside the represented domain. + fn contains(&self, row: T) -> bool + where + T: Id, + { + match self { + Self::Full => true, + Self::Mask(mask) => mask.contains(row), + } + } +} + +impl HeapMemoryUsage for Rows +where + T: Id, +{ + /// Returns this admission set's owned heap bytes. + /// + /// A full set admitting every row owns none. A masked subset returns the mask's own bytes. + fn heap_memory_usage(&self) -> u64 { + match self { + Self::Full => 0, + Self::Mask(mask) => mask.heap_bytes(), + } + } +} + +/// The principal and instance-administrator status retained by a resolved mask. +/// +/// The containing cache entry keeps the administrator status unchanged across reuse. +/// Property-protection rules remain separate. Filter construction passes them to +/// [`Self::protection`]. +#[derive(Debug, Copy, Clone)] +pub(crate) struct VisibilityActor { + /// The principal the request authenticated as. + pub id: ActorId, + /// Whether the principal administers the instance, and so reads properties unprotected. + pub instance_admin: bool, +} + +impl VisibilityActor { + /// Returns whether `config`'s property protection applies to this actor. + /// + /// An instance administrator is exempt, and an empty configuration protects nothing. Only a + /// rule that can withhold a property incurs the masking cost. + #[must_use] + pub(crate) fn masked_by(self, config: &PropertyProtectionFilterConfig<'_>) -> bool { + !config.is_empty() && !self.instance_admin + } + + /// Builds the property protection filter that applies to this actor, when one does. + /// + /// Returns [`None`] exactly when [`masked_by`](Self::masked_by) is false, which the caller + /// reads as unprotected delivery. + #[must_use] + pub(crate) fn protection<'config, 'rules>( + self, + config: &'config PropertyProtectionFilterConfig<'rules>, + ) -> Option> { + self.masked_by(config) + .then(|| config.to_property_protection_filter(Some(self.id))) + } +} + +hashql_core::id::newtype! { + /// A row of domain `T` accompanied by a visibility-admission witness. + /// + /// [`VisibilityMask::visible_node`] and [`VisibilityMask::visible_edge`] are the ordinary + /// construction paths. Macro-generated constructors and conversions remain crate-visible and + /// require their callers to establish the same admission condition. Assembly accepts `Visible` + /// where policy permits delivery, distinguishing an admitted row from an ordinary row + /// identifier. + pub(crate) struct Visible(u64) +} + +impl Visible { + /// Returns the underlying row identifier, discarding the admission evidence. + pub(crate) fn unwrap(self) -> T + where + T: Id, + { + T::from_u64(self.as_u64()) + } +} + +/// The declared delivery policy, independent of a mask's cardinality. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +pub(crate) enum VisibilityKind { + /// The actor is admitted to the whole corpus. + Corpus, + /// The actor is admitted to an explicit row set. + Scope, +} + +/// The rows one actor may receive from a generation. +/// +/// Each cache entry retains one mask from a single resolution. The mask carries no delta revision. +/// Assembly consults it for every row. Full masks admit any row queried during their lifetime, +/// including a later allocation, while partial masks admit only their captured bits. Later rows +/// remain hidden until refresh. The associated +/// [`ViewSchedule`](crate::serve::schedule::ViewSchedule) retains only the placements captured +/// during entry construction. The schedule omits a newly allocated row until refresh rebuilds that +/// entry. +/// +/// [`visible_node`](Self::visible_node) and [`visible_edge`](Self::visible_edge) are response hot +/// paths. Their [`Visible`] results let assembly require a typed admission witness instead of an +/// ordinary row identifier. Other crate-visible construction paths carry the same admission +/// obligation. +#[derive(Debug)] +pub(crate) struct VisibilityMask { + actor: VisibilityActor, + nodes: Rows, + edges: Rows, +} + +impl VisibilityMask { + /// Admits `actor` to the whole corpus. + pub(crate) const fn full(actor: VisibilityActor) -> Self { + Self { + actor, + nodes: Rows::Full, + edges: Rows::Full, + } + } + + /// Admits `actor` to exactly the given node and edge rows. + pub(crate) const fn partial( + actor: VisibilityActor, + nodes: CompressedBitSet, + edges: CompressedBitSet, + ) -> Self { + Self { + actor, + nodes: Rows::Mask(nodes), + edges: Rows::Mask(edges), + } + } + + /// Returns the principal captured during this mask's resolution. + pub(crate) const fn actor(&self) -> VisibilityActor { + self.actor + } + + /// Returns the delivery policy the mask declares. + pub(crate) const fn kind(&self) -> VisibilityKind { + match self.nodes { + Rows::Full => VisibilityKind::Corpus, + Rows::Mask(_) => VisibilityKind::Scope, + } + } + + /// Admits `node`, or returns [`None`] where the actor may not receive it. + pub(crate) fn visible_node(&self, node: NodeRowId) -> Option> { + self.nodes.contains(node).then(|| Visible::new(node.get())) + } + + /// Admits `edge`, or returns [`None`] where the actor may not receive it. + /// + /// An edge needs its own admission and the admission of both `endpoints`. Delivering an edge + /// whose endpoint is withheld would disclose that the endpoint exists, which the endpoint's + /// own exclusion is there to prevent. + pub(crate) fn visible_edge( + &self, + edge: EdgeRowId, + endpoints: [NodeRowId; 2], + ) -> Option> { + (self.edges.contains(edge) + && endpoints + .iter() + .all(|&endpoint| self.nodes.contains(endpoint))) + .then(|| Visible::new(edge.get())) + } +} + +impl HeapMemoryUsage for VisibilityMask { + /// Returns the combined heap ownership of the node and edge admission sets. + /// + /// Cache weighting uses this value, which excludes shared allocations. + fn heap_memory_usage(&self) -> u64 { + self.nodes.heap_memory_usage() + self.edges.heap_memory_usage() + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/cache.rs b/libs/@local/graph/atlas/src/serve/world/cache.rs new file mode 100644 index 00000000000..842745c0873 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/cache.rs @@ -0,0 +1,43 @@ +//! Derived values a [`World`](super::World) computes once from its fitted artifacts. +//! +//! The base cascade depends on the layout alone, and every +//! [saturated scope](crate::serve::schedule::ViewSchedule::of) shares it. Deriving it at first use +//! rather than at open keeps a world that never serves a saturated scope from paying for one. + +use alloc::sync::Arc; +use std::sync::OnceLock; + +use super::Layout; +use crate::serve::schedule::ScopeSchedule; + +/// Lazily derived values of one world, each computed at most once. +#[derive(Debug)] +pub(super) struct Cache { + /// The complete base cascade, absent until the first saturated scope asks for it. + base_scope_schedule: OnceLock>, +} + +impl Cache { + /// Creates a cache with nothing derived yet. + pub(super) const fn new() -> Self { + Self { + base_scope_schedule: OnceLock::new(), + } + } + + /// Returns the complete base cascade over `layout`, deriving it on the first call. + /// + /// Concurrent first calls block on one derivation and share its result. Every call passes the + /// world's own layout, and the first derivation answers every later call. + /// + /// # Panics + /// + /// Panics during the first derivation if a base node has no position or priority, the + /// condition [`ScopeSchedule::from_base`] refuses. Opening samples the layout's reverse + /// columns rather than checking every row, and a malformed entry at an unsampled row reaches + /// this derivation. + pub(super) fn base_scope_schedule(&self, layout: &Layout) -> &Arc { + self.base_scope_schedule + .get_or_init(|| Arc::new(ScopeSchedule::from_base(layout))) + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/encoding.rs b/libs/@local/graph/atlas/src/serve/world/encoding.rs new file mode 100644 index 00000000000..4da4181265c --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/encoding.rs @@ -0,0 +1,200 @@ +//! Wire-id encoding for one row domain of an opened world. +//! +//! A world derives one [`RowCodec`] per row domain when it opens. Every row the fitted generation +//! holds is encoded once at that point. Rows added later are encoded on demand by the codec. + +use error_stack::{Report, ResultExt as _}; +use hashql_core::id::{Id, IdVec}; + +use super::{ + super::codec::{RowCodec, RowDomain}, + OpenOptions, + error::WorldError, +}; +use crate::serve::codec::{EncodableId, EncodedRowId}; + +/// One row domain's codec with the wire ids of the opening domain precomputed. +#[derive(Debug)] +pub(crate) struct Encoding { + /// The keyed mapping between the domain's rows and their wire ids. + codec: RowCodec, + /// The wire id of every row in the opening domain, in row order. + lookup: IdVec>, +} + +impl Encoding { + /// Derives the codec and precomputes the wire id of every row in `domain`. + /// + /// # Complexity + /// + /// After codec derivation, precomputation takes O(n) time and retains n wire-id entries for + /// n rows in `domain`. Each row requires eight keyed Feistel rounds through + /// [`RowCodec::encode`]. + /// + /// # Errors + /// + /// Returns [`WorldError`] if `domain` contains a row at or beyond + /// [`WIRE_ROW_BOUND`](crate::serve::codec::WIRE_ROW_BOUND), or its row count exceeds the + /// address space. + pub(crate) fn open( + OpenOptions { generation, secret }: OpenOptions<'_>, + domain: RowDomain, + ) -> Result> + where + I: EncodableId, + { + let rows = domain.bound().as_u64(); + u32::try_from(rows.saturating_sub(1)).change_context(WorldError::TooManyRows { rows })?; + let rows = usize::try_from(rows).change_context(WorldError::TooManyRows { rows })?; + + let codec = RowCodec::derive(secret.as_ref(), generation.id()); + let lookup = IdVec::from_fn(rows, |id| codec.encode(id)); + + Ok(Self { codec, lookup }) + } + + /// Encodes `row` as its wire id. + /// + /// # Complexity + /// + /// The table lookup and codec fallback take O(1) time in the opening row count and use constant + /// additional space. A miss applies eight keyed Feistel rounds through [`RowCodec::encode`]. + /// + /// # Panics + /// + /// Panics when the table lookup misses and [`RowCodec::encode`] rejects `row` at or beyond + /// [`WIRE_ROW_BOUND`](crate::serve::codec::WIRE_ROW_BOUND). + /// + /// # Warning + /// + /// A lossy [`Id::as_usize`] conversion can select a different cached row before the codec + /// checks the original ID. For [`NodeRowId`](crate::identity::NodeRowId) on a 32-bit target, + /// row 2³² selects index zero in a nonempty lookup. + pub(super) fn encode(&self, row: I) -> EncodedRowId + where + I: EncodableId, + { + if let Some(&encoded) = self.lookup.get(row) { + return encoded; + } + + self.codec.encode(row) + } + + /// Decodes a wire value to a row accepted by `domain`. + /// + /// Returns [`None`] when the inverse permutation yields a value outside `I` or `domain`. + /// + /// # Complexity + /// + /// Every decode applies eight inverse Feistel rounds through [`RowCodec::decode`], taking + /// O(1) time in the opening row count and constant additional space. + pub(super) fn decode(&self, wire: EncodedRowId, domain: RowDomain) -> Option + where + I: Id, + { + self.codec.decode(wire, domain) + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use hashql_core::id::newtype; + + use super::Encoding; + use crate::{ + identity::NodeRowId, + serve::{ + codec::{EncodableId, RowCodec, RowDomain, WIRE_ROW_BOUND}, + tests::fixture::{NODES, TamperFixture, secret}, + world::{OpenOptions, WorldError}, + }, + }; + + newtype! { + /// Rows whose label stops codec derivation before lookup allocation. + struct ValidationRow(u64) + } + + impl EncodableId for ValidationRow { + fn label() -> &'static [u8] { + panic!("codec derivation reached"); + } + } + + /// Empty and populated domains precompute the same mapping as the derived codec. + #[test] + fn open_valid_domains() { + let fixture = TamperFixture::publish("encoding-valid-domains"); + let secret = secret(); + let options = OpenOptions { + generation: fixture.generation(), + secret: &secret, + }; + let codec = RowCodec::::derive(secret.as_ref(), options.generation.id()); + + for rows in [0, 1, NODES] { + let domain = RowDomain::new(NodeRowId::new(rows)); + let encoding = Encoding::open(options, domain).expect("a small row domain should open"); + assert_eq!( + encoding.lookup.len(), + domain.size(), + "every row should be precomputed" + ); + for row in 0..rows { + let row = NodeRowId::new(row); + assert_eq!( + encoding.encode(row), + codec.encode(row), + "the lookup should match the codec" + ); + } + let next = domain.bound(); + assert_eq!( + encoding.encode(next), + codec.encode(next), + "a later row should use the same mapping" + ); + } + } + + /// Oversized domains fail before codec derivation or lookup allocation. + #[test] + fn open_oversized_domain() { + let fixture = TamperFixture::publish("encoding-oversized-domain"); + let domain = RowDomain::new(ValidationRow::new(WIRE_ROW_BOUND + 1)); + let error = Encoding::open( + OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }, + domain, + ) + .expect_err("an oversized row domain should fail validation"); + assert_matches!( + error.current_context(), + WorldError::TooManyRows { rows } if *rows == WIRE_ROW_BOUND + 1, + ); + } + + /// The full wire domain passes validation, including its last row `u32::MAX`. + /// + /// The label stops derivation before allocating the full lookup table. + #[test] + #[cfg(target_pointer_width = "64")] + #[should_panic(expected = "codec derivation reached")] + fn open_full_wire_domain() { + let fixture = TamperFixture::publish("encoding-full-wire-domain"); + let domain = RowDomain::new(ValidationRow::new(WIRE_ROW_BOUND)); + let _encoding = Encoding::open( + OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }, + domain, + ) + .expect("the full wire domain should pass validation"); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/error.rs b/libs/@local/graph/atlas/src/serve/world/error.rs new file mode 100644 index 00000000000..a0d8b8fc41f --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/error.rs @@ -0,0 +1,210 @@ +//! The failures a [`World`](super::World) can report while opening. +//! +//! Each component's open reports its artifact and count refusals under this one type, and the +//! world adds the agreement it checks across components. + +use core::{error::Error, fmt}; + +use crate::file::repository::FileName; + +/// A failure while opening a [`World`]. +/// +/// [`World`]: super::World +#[derive(Debug)] +pub(crate) enum WorldError { + /// An artifact failed to open. + Open { + /// The artifact's repository file name. + file: FileName, + }, + /// The recorded bucket schedule failed validation. + BucketSchedule, + /// The node identity table and the node position columns disagree on the row count. + NodeIndexCountMismatch { + /// The rows the node identity table covers. + identity: u64, + /// The rows the row-of-position column covers. + lookup: usize, + /// The rows the position-of-row column covers. + reverse: usize, + }, + /// The node importance columns disagree on the row count. + NodeImportanceCountMismatch { + /// The rows the rank-of-position column covers. + lookup: usize, + /// The rows the position-of-rank column covers. + reverse: usize, + }, + /// The wire coordinate column and the Morton order disagree on the point count. + GeometryCountMismatch { + /// The points the wire coordinate column covers. + positions: usize, + /// The points the Morton order covers. + morton_order: u64, + }, + /// The spatial index root and the wire coordinate column disagree on the point count. + SpatialIndexCountMismatch { + /// The points the spatial index root covers. + root: u32, + /// The points the wire coordinate column covers. + positions: usize, + }, + /// The node index, the node importance and the geometry disagree on the node count. + LayoutCountMismatch { + /// The nodes the node index covers. + index: usize, + /// The nodes the node importance covers. + importance: usize, + /// The nodes the geometry covers. + geometry: usize, + }, + /// A sampled position failed to round-trip through a layout permutation and its inverse. + LayoutRoundtrip, + /// The node count exceeds the `u32` range of positions and wire ids. + TooManyNodes { + /// The nodes the world holds. + nodes: usize, + }, + /// The row count exceeds the wire domain or address space. + TooManyRows { + /// The opening row domain's size. + rows: u64, + }, + /// The edge identity table, the endpoint column and the adjacency disagree on the edge count. + TopologyCountMismatch { + /// The edges the edge identity table covers. + identity: u64, + /// The edges the endpoint column covers. + endpoints: usize, + /// The edges the adjacency covers. + adjacency: u64, + }, + /// The edge count exceeds the `u32` range of edge ids. + TooManyEdges { + /// The edges the world holds. + edges: u64, + }, + /// The ontology identity table and the postings disagree on the type count. + OntologyCountMismatch { + /// The types the ontology identity table covers. + identity: u64, + /// The types the postings cover. + postings: u64, + }, + /// The point count exceeds the address space. + TooManyPositions { + /// The points the postings cover. + positions: u64, + }, + /// The type count exceeds the address space. + TooManyTypes { + /// The types the ontology holds. + types: u64, + }, + /// The type closure failed to derive. + OntologyClosure, + /// The layout, the topology and the ontology disagree on the node count. + NodeCountMismatch { + /// The nodes the layout covers. + layout: usize, + /// The nodes the topology covers. + topology: usize, + /// The nodes the ontology covers. + ontology: usize, + }, +} + +impl fmt::Display for WorldError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Open { file } => write!(fmt, "the {file} artifact failed to open"), + Self::BucketSchedule => fmt.write_str("the recorded bucket schedule failed validation"), + Self::NodeIndexCountMismatch { + identity, + lookup, + reverse, + } => write!( + fmt, + "the node identity table covers {identity} rows where the row-of-position column \ + covers {lookup} and the position-of-row column covers {reverse}", + ), + Self::NodeImportanceCountMismatch { lookup, reverse } => write!( + fmt, + "the rank-of-position column covers {lookup} rows where the position-of-rank \ + column covers {reverse}", + ), + Self::GeometryCountMismatch { + positions, + morton_order, + } => write!( + fmt, + "the wire coordinate column covers {positions} points where the Morton order \ + covers {morton_order}", + ), + Self::SpatialIndexCountMismatch { root, positions } => write!( + fmt, + "the spatial index root covers {root} points where the wire coordinate column \ + covers {positions}", + ), + Self::LayoutCountMismatch { + index, + importance, + geometry, + } => write!( + fmt, + "the node index covers {index} nodes where the node importance covers \ + {importance} and the geometry covers {geometry}", + ), + Self::LayoutRoundtrip => fmt.write_str( + "a sampled position failed to round-trip through the layout permutations", + ), + Self::TooManyNodes { nodes } => write!( + fmt, + "the world holds {nodes} nodes where positions and wire ids span the u32 range", + ), + Self::TooManyRows { rows } => write!( + fmt, + "row count {rows} exceeds the wire domain or address space", + ), + Self::TopologyCountMismatch { + identity, + endpoints, + adjacency, + } => write!( + fmt, + "the edge identity table covers {identity} edges where the endpoint column covers \ + {endpoints} and the adjacency covers {adjacency}", + ), + Self::TooManyEdges { edges } => write!( + fmt, + "the world holds {edges} edges where edge ids span the u32 range", + ), + Self::OntologyCountMismatch { identity, postings } => write!( + fmt, + "the ontology identity table covers {identity} types where the postings cover \ + {postings}", + ), + Self::TooManyPositions { positions } => write!( + fmt, + "the postings cover {positions} points where the address space spans the usize \ + range", + ), + Self::TooManyTypes { types } => write!( + fmt, + "the ontology holds {types} types where the address space spans the usize range", + ), + Self::OntologyClosure => fmt.write_str("the type closure failed to derive"), + Self::NodeCountMismatch { + layout, + topology, + ontology, + } => write!( + fmt, + "the layout covers {layout} nodes where the topology covers {topology} and the \ + ontology covers {ontology}", + ), + } + } +} + +impl Error for WorldError {} diff --git a/libs/@local/graph/atlas/src/serve/world/geometry.rs b/libs/@local/graph/atlas/src/serve/world/geometry.rs new file mode 100644 index 00000000000..cd671fafb4a --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/geometry.rs @@ -0,0 +1,166 @@ +//! Wire-frame coordinates and spatial indexes for a fitted layout. +//! +//! Coordinate access uses [`BasePosition`], the shared order of the geometry artifacts. + +use error_stack::{Report, ReportSink, ResultExt as _, TryReportTupleExt as _}; + +use super::{OpenOptions, error::WorldError}; +use crate::{ + file::{morton::read::MortonFile, quad::read::QuadFile}, + identity::{BasePosition, Column}, + math::{Bounds2, Vec2}, + salt::lod::stage::WIRE_FRAME, +}; + +/// Fitted coordinates with their wire-frame bounds and spatial indexes. +#[derive(Debug)] +pub(crate) struct Geometry { + /// The recorded world frame's image in the wire frame. + /// + /// `open` derives this from the generation's frame metadata and leaves it [`None`] when the + /// Morton order holds no code. It neither measures the coordinate column nor checks the column + /// against the frame. For canonical fit output, a column produced by normalizing that world + /// frame onto the wire frame, the image is the column's tight extent up to the normalization's + /// rounding. + bounds: Option, + /// The wire-frame coordinate of every point, in base-position order. + positions: Column, + + /// The quadtree over the fitted points. + /// + /// When the quadtree has a root, `open` checks the root's point count against the coordinate + /// column's length. A quadtree without nodes passes that check unexamined. + spatial_index: QuadFile, + /// The Morton codes and bucket fenceposts in base delivery order. + morton_order: MortonFile, +} + +impl Geometry { + /// Opens the geometry artifacts and checks their point counts. + /// + /// # Errors + /// + /// Returns [`WorldError`] for artifact opening or mismatched point counts. + pub(crate) fn open( + OpenOptions { generation, .. }: OpenOptions<'_>, + ) -> Result> { + let files = &generation.repository().files; + + let positions = files + .wire_coordinates + .open(generation) + .change_context(WorldError::Open { + file: files.wire_coordinates.name(), + }); + + let spatial_index = files + .quad + .open(generation) + .change_context(WorldError::Open { + file: files.quad.name(), + }); + + let morton_order: Result = + files + .morton + .open(generation) + .change_context(WorldError::Open { + file: files.morton.name(), + }); + + let (positions, spatial_index, morton_order) = + (positions, spatial_index, morton_order).try_collect()?; + + let world = generation.repository().metadata.evidence.lod.world; + let bounds = (morton_order.count() > 0).then(|| world.image_in(WIRE_FRAME)); + + let this = Self { + bounds, + positions, + spatial_index, + morton_order, + }; + + let mut errors = ReportSink::new_armed(); + + if this.positions.len() as u64 != this.morton_order.count() { + errors.capture(WorldError::GeometryCountMismatch { + positions: this.positions.len(), + morton_order: this.morton_order.count(), + }); + } + + if let Some(root) = this.spatial_index.nodes().first() + && root.points() as usize != this.positions.len() + { + errors.capture(WorldError::SpatialIndexCountMismatch { + root: root.points(), + positions: this.positions.len(), + }); + } + + errors.finish_ok(this) + } + + /// Returns the Morton codes and bucket fenceposts in base delivery order. + pub(crate) const fn morton_order(&self) -> &MortonFile { + &self.morton_order + } + + /// Returns the recorded world frame's image in the wire frame. + /// + /// The value is [`None`] when the Morton order holds no code. It comes from the generation's + /// frame metadata, not from measuring the coordinate column. For canonical fit output it is + /// the column's tight extent up to the normalization's rounding. + pub(super) const fn bounds(&self) -> Option { + self.bounds + } + + /// Returns the wire-frame coordinate at `position`, [`None`] outside the point domain. + pub(super) fn position(&self, position: BasePosition) -> Option { + self.positions.view().get(position).copied() + } + + /// Returns the number of fitted points, the length of the coordinate column. + pub(super) fn node_count(&self) -> usize { + self.positions.len() + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use super::Geometry; + use crate::serve::{ + tests::fixture::{NODES, TamperFixture, retarget_quad_root, secret}, + world::{OpenOptions, error::WorldError}, + }; + + /// Open refuses a spatial index root with fewer points than the wire coordinate column. + /// + /// Returns [`WorldError::SpatialIndexCountMismatch`]. + #[test] + fn quad_root_subtree_short() { + let fixture = TamperFixture::publish("geometry-quad-root"); + let positions = usize::try_from(NODES).expect("fixture node counts fit usize"); + let points = u32::try_from(NODES - 1).expect("fixture point counts fit u32"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.quad.name(), |path| { + retarget_quad_root(path, points); + }); + + let report = Geometry::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a root point count below the coordinate column's"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::SpatialIndexCountMismatch { root, positions: counted }] + if *root == points && *counted == positions, + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/layout.rs b/libs/@local/graph/atlas/src/serve/world/layout.rs new file mode 100644 index 00000000000..d76f3399272 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/layout.rs @@ -0,0 +1,728 @@ +//! Node-row lookup through the fitted layout's independent permutations. +//! +//! The geometry and importance columns store coordinates and ranks in [`BasePosition`] order, the +//! bucket-major delivery order, and [`NodeIndex`] maps a stable row to its position before either +//! lookup. Storing by position keeps a bucket's delivery one contiguous read, and addressing by +//! row keeps a node's identity independent of where the fit placed it. + +use core::{error::Error, fmt}; + +use error_stack::{Report, TryReportTupleExt as _}; +use hashql_core::id::Id as _; + +use super::{ + OpenOptions, + error::WorldError, + geometry::Geometry, + node_importance::{ImportanceProvider, NodeImportance, NodePriority}, + node_index::NodeIndex, +}; +use crate::{ + identity::{BasePosition, ImportanceRank, NodeRowId}, + math::{Bounds2, Vec2}, + morton::{Depth, MortonCell, MortonKey}, + serve::delta::{ + epoch::Epoch, + layout::provider::{NaiveLayoutProvider, VersionedLayoutProvider as _}, + }, +}; + +/// Visible node coordinates addressed by stable row identity. +pub(crate) trait LayoutProvider { + /// Returns the allocated row count, including withdrawn and unplaced rows. + fn provide_node_count(&self) -> usize; + /// Returns the visible position in the [wire frame](crate::salt::lod::stage::WIRE_FRAME). + /// + /// Returns [`None`] for withdrawn or unplaced rows. + fn provide_position(&self, node: NodeRowId) -> Option; +} + +impl LayoutProvider for &T { + fn provide_node_count(&self) -> usize { + T::provide_node_count(self) + } + + fn provide_position(&self, node: NodeRowId) -> Option { + T::provide_position(self, node) + } +} + +/// A layout permutation whose recorded inverse disagrees with it at a sampled position. +#[derive(Debug)] +pub(crate) enum LayoutRoundtripError { + /// The rank columns are not inverse. + RankInverse { + /// The sampled base position. + position: BasePosition, + /// The rank the position carries. + rank: ImportanceRank, + /// The rank's reverse position, absent outside the rank domain. + roundtrip: Option, + }, + /// The row columns are not inverse. + RowInverse { + /// The sampled base position. + position: BasePosition, + /// The row the position carries. + row: NodeRowId, + /// The row's reverse position, absent outside the row domain. + roundtrip: Option, + }, +} + +impl fmt::Display for LayoutRoundtripError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::RankInverse { + position, + rank, + roundtrip: Some(roundtrip), + } => write!( + fmt, + "the rank columns are not inverse: position {position} carries rank {rank}, which \ + the reverse column sends to position {roundtrip}", + ), + Self::RankInverse { + position, + rank, + roundtrip: None, + } => write!( + fmt, + "the rank columns are not inverse: position {position} carries rank {rank}, which \ + lies outside the rank domain", + ), + Self::RowInverse { + position, + row, + roundtrip: Some(roundtrip), + } => write!( + fmt, + "the row columns are not inverse: position {position} carries row {row}, which \ + the reverse column sends to position {roundtrip}", + ), + Self::RowInverse { + position, + row, + roundtrip: None, + } => write!( + fmt, + "the row columns are not inverse: position {position} carries row {row}, which \ + lies outside the row domain", + ), + } + } +} + +impl Error for LayoutRoundtripError {} + +/// Fitted coordinates and importance ranks joined through the base-position permutation. +#[derive(Debug)] +pub(crate) struct Layout { + /// Node identities and the row/base-position permutation. + pub index: NodeIndex, + /// The rank/base-position permutation. + importance: NodeImportance, + + /// Coordinates, Morton order and the spatial index in base-position order. + geometry: Geometry, +} + +impl Layout { + /// Opens the fitted layout and checks column counts and sampled inverse mappings. + /// + /// # Errors + /// + /// Returns [`WorldError`] for artifact opening, count or sampled roundtrip failures. + pub(crate) fn open(options: OpenOptions<'_>) -> Result> { + let index = NodeIndex::open(options); + let importance = NodeImportance::open(options); + let geometry = Geometry::open(options); + + let (index, importance, geometry) = (index, importance, geometry).try_collect()?; + + let this = Self { + index, + importance, + geometry, + }; + + let nodes = this.index.len(); + if nodes != this.importance.len() || nodes != this.geometry.node_count() { + return Err(Report::new(WorldError::LayoutCountMismatch { + index: nodes, + importance: this.importance.len(), + geometry: this.geometry.node_count(), + }) + .expand()); + } + + this.try_roundtrip_sample().map_err(|error| { + Report::new(error) + .change_context(WorldError::LayoutRoundtrip) + .expand() + })?; + + Ok(this) + } + + /// Checks the rank and row permutations against their recorded inverses at sampled positions. + /// + /// The sample is at most 64 positions, evenly spaced over the base positions with both ends + /// included. At 64 nodes or fewer every position is checked. Above that, the stored position + /// of a row the sample does not reach is not checked here. Over two or more nodes, a column + /// sending every position to one value fails no later than the first sampled position past + /// zero, because a single roundtrip position can agree with at most one sampled position. + /// + /// # Errors + /// + /// Returns the first [`LayoutRoundtripError`] a sampled position produces: the rank pairing + /// is checked before the row pairing at each position. + #[expect( + clippy::integer_division, + clippy::integer_division_remainder_used, + reason = "an evenly spaced sample point is the floor of its proportional position" + )] + fn try_roundtrip_sample(&self) -> Result<(), LayoutRoundtripError> { + /// The number of positions checked, at most one per node. + const SAMPLES: u64 = 64; + let nodes = self.geometry.node_count() as u64; + + // opening the node index rejects out-of-range counts. + if nodes == 0 || u32::try_from(nodes - 1).is_err() { + return Ok(()); + } + + // The count check in `open` matched both permutation columns to the geometry, and every + // sampled position lies below that count. + let samples = SAMPLES.min(nodes); + for index in 0..samples { + let at = if samples == 1 { + 0 + } else { + index * (nodes - 1) / (samples - 1) + }; + + let position = BasePosition::from_u64(at); + + let rank = self.importance[position]; + let roundtrip = self.importance.reverse(rank); + if roundtrip != Some(position) { + return Err(LayoutRoundtripError::RankInverse { + position, + rank, + roundtrip, + }); + } + + let row = self.index[position]; + let roundtrip = self.index.base_reverse(row); + if roundtrip != Some(position) { + return Err(LayoutRoundtripError::RowInverse { + position, + row, + roundtrip, + }); + } + } + + Ok(()) + } + + /// Reads one recorded bucket's rows and keys inside a cell, in base delivery order. + pub(crate) fn base_run( + &self, + bucket: Depth, + cell: MortonCell, + ) -> impl Iterator { + let morton = self.geometry.morton_order(); + morton + .run(bucket, cell) + .map(move |position| (morton.code(position), self.index[position])) + } + + /// Returns a base row's recorded bucket, independent of visibility. + /// + /// Returns [`None`] outside the fitted row domain. + /// + /// # Panics + /// + /// Panics when the row's recorded base position lies at or beyond the Morton order's count. + /// The reverse column returns the recorded value unchecked, and opening samples that column + /// rather than checking every row. A malformed entry at an unsampled row therefore reaches the + /// Morton order's own assertion. + pub(crate) fn base_bucket_of(&self, node: NodeRowId) -> Option { + let position = self.index.base_reverse(node)?; + Some(self.geometry.morton_order().bucket_of(position)) + } + + /// Counts base rows in the recorded buckets from the root through `cut`, inclusive. + pub(crate) fn base_count_through(&self, cut: Depth) -> usize { + self.geometry + .morton_order() + .fenceposts() + .segment(cut) + .end + .as_usize() + } + + /// Returns the recorded world frame's image in the wire frame. + /// + /// The value is [`None`] when the Morton order holds no code. Open takes it from the + /// generation's frame metadata rather than measuring the fitted coordinates. For canonical fit + /// output it is their tight extent up to the normalization's rounding. + pub(crate) const fn base_bounds(&self) -> Option { + self.geometry.bounds() + } + + /// Returns the deepest recorded bucket holding a base row, [`None`] when no bucket does. + pub(crate) fn base_deepest_occupied(&self) -> Option { + self.geometry + .morton_order() + .fenceposts() + .segments() + .into_iter() + .enumerate() + .rev() + .find(|(_, segment)| !segment.is_empty()) + .map(|(bucket, _)| Depth::from_usize(bucket)) + } + + /// Returns whether a recorded bucket contains a base row inside `cell`. + pub(crate) fn base_occupied(&self, bucket: Depth, cell: MortonCell) -> bool { + !self.geometry.morton_order().run(bucket, cell).is_empty() + } + + /// Returns the deepest prefix shared with any recorded base key. + /// + /// Returns [`None`] when the generation records no key. + pub(crate) fn base_shared_depth(&self, key: MortonKey) -> Option { + let morton = self.geometry.morton_order(); + let codes = morton.codes(); + + morton + .fenceposts() + .segments() + .into_iter() + .filter_map(|segment| { + let codes = &codes[segment]; + let at = codes.partition_point(|code| code.get() < key.to_bits()); + // Codes sort within each bucket's segment. For every depth, the codes sharing that + // depth's prefix with `key` form one contiguous run of the sorted segment, and a + // non-empty run holds one of the two codes adjacent to `key`'s insertion point. + // The deepest shared prefix in the segment is therefore attained at one of those + // two codes. + [at.checked_sub(1), (at < codes.len()).then_some(at)] + .into_iter() + .flatten() + .map(|index| key.shared_depth(MortonKey::from_bits(codes[index].get()))) + .max() + }) + .max() + } + + /// Returns an allocated row's priority, independent of visibility. + /// + /// # Panics + /// + /// Panics if this layout does not belong to the epoch's world. + pub(crate) fn priority(&self, epoch: &Epoch, node: NodeRowId) -> Option { + epoch.importance(self).provide_priority(node) + } + + /// Returns the allocated node count, including withdrawn and unplaced rows. + /// + /// # Panics + /// + /// Panics if this layout does not belong to the epoch's world. + pub(crate) fn node_count(&self, epoch: &Epoch) -> usize { + let base = NaiveLayoutProvider::new(self); + epoch.layout(self).bind(&base).provide_node_count() + } + + /// Returns the visible wire-frame position at the captured epoch's revision. + /// + /// # Panics + /// + /// Panics if this layout does not belong to the epoch's world. + pub(crate) fn position(&self, epoch: &Epoch, node: NodeRowId) -> Option { + let base = NaiveLayoutProvider::new(self); + epoch + .layout(self) + .bind(&base) + .provide_position_at(node, epoch.revision()) + } +} + +impl ImportanceProvider for Layout { + /// Returns the fitted [`NodePriority::Rank`] at `node`'s base position. + /// + /// Returns [`None`] if the row has no base position or that position is outside the rank + /// column. + fn provide_priority(&self, node: NodeRowId) -> Option { + self.importance + .lookup(self.index.base_reverse(node)?) + .map(NodePriority::Rank) + } +} + +impl LayoutProvider for Layout { + fn provide_node_count(&self) -> usize { + self.geometry.node_count() + } + + /// Returns the fitted coordinate at `node`'s base position. + /// + /// Returns [`None`] if the row has no base position or that position is outside the coordinate + /// column. + fn provide_position(&self, node: NodeRowId) -> Option { + self.geometry.position(self.index.base_reverse(node)?) + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use hashql_core::id::Id as _; + + use super::{Layout, LayoutProvider, LayoutRoundtripError}; + use crate::{ + identity::{BasePosition, ImportanceRank, NodeRowId}, + serve::{ + schedule::ScopeSchedule, + tests::fixture::{ + NODES, TamperFixture, constant_u32_column, constant_u64_column, secret, + set_row_position, shorten_entities, shorten_u32_column, + }, + world::{ + OpenOptions, + error::WorldError, + node_importance::{ImportanceProvider, NodePriority}, + }, + }, + }; + + /// Priority follows the row-to-position permutation before rank lookup. + #[test] + fn priority_row_permutation() { + let fixture = TamperFixture::publish("layout-priority-row-position-rank"); + let layout = Layout::open(OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }) + .expect("should open the fitted layout"); + + let mut permuted = false; + for index in 0..LayoutProvider::provide_node_count(&layout) { + let position = BasePosition::from_usize(index); + let row = layout.index[position]; + permuted |= row.as_usize() != index; + assert_eq!( + ImportanceProvider::provide_priority(&layout, row), + Some(NodePriority::Rank(layout.importance[position])), + "should resolve the rank through the row's base position" + ); + } + assert!(permuted, "should exercise a non-identity row permutation"); + assert_eq!( + ImportanceProvider::provide_priority(&layout, NodeRowId::MAX), + None, + "should return no priority outside the row domain" + ); + } + + /// Position lookup follows the row-to-position permutation into the geometry. + /// + /// Every row answers its base position's coordinate over a non-identity permutation, and a row + /// outside the domain answers [`None`]. + #[test] + fn positions_row_permutation() { + let fixture = TamperFixture::publish("layout-position-permutation"); + let layout = Layout::open(OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }) + .expect("should open the fitted layout"); + + let mut permuted = false; + for index in 0..LayoutProvider::provide_node_count(&layout) { + let position = BasePosition::from_usize(index); + let row = layout.index[position]; + permuted |= row.as_usize() != index; + assert_eq!( + LayoutProvider::provide_position(&layout, row), + layout.geometry.position(position), + ); + } + assert!(permuted, "should exercise a non-identity row permutation"); + assert_eq!( + LayoutProvider::provide_position(&layout, NodeRowId::MAX), + None + ); + } + + /// Rejects empty rank columns beside nonempty geometry before sampling positions. + #[test] + fn open_empty_importance() { + let fixture = TamperFixture::publish("layout-empty-importance"); + let files = &fixture.generation().repository().files; + let tampered = fixture.tamper(&files.rank_of_position.name(), |path| { + shorten_u32_column(path, 0); + shorten_u32_column( + path.with_file_name(files.position_of_rank.name().as_str()), + 0, + ); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("should reject the mismatched layout counts"); + + let nodes = usize::try_from(NODES).expect("fixture node counts should fit usize"); + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutCountMismatch { index, importance: 0, geometry }] + if *index == nodes && *geometry == nodes, + ); + } + + /// Rejects an empty node index beside nonempty geometry before sampling positions. + #[test] + fn open_empty_index() { + let fixture = TamperFixture::publish("layout-empty-index"); + let files = &fixture.generation().repository().files; + let tampered = fixture.tamper(&files.row_of_position.name(), |path| { + constant_u64_column(path, 0, 0); + shorten_u32_column( + path.with_file_name(files.position_of_row.name().as_str()), + 0, + ); + shorten_entities::( + path.with_file_name(files.node_identities.name().as_str()), + 0, + 0, + ); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("should reject the mismatched layout counts"); + + let nodes = usize::try_from(NODES).expect("fixture node counts should fit usize"); + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutCountMismatch { index: 0, importance, geometry }] + if *importance == nodes && *geometry == nodes, + ); + } + + /// Rejects a position-of-rank column that is not a permutation. + /// + /// Returns [`WorldError::LayoutRoundtrip`] from [`LayoutRoundtripError::RankInverse`]. + /// + /// Every rank claiming position zero keeps the length and the format. The roundtrip sample + /// therefore refuses the pairing at the first sampled position past zero. + #[test] + fn rank_positions_constant() { + let fixture = TamperFixture::publish("layout-rank-positions-constant"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.position_of_rank.name(), |path| { + constant_u32_column(path, NODES, 0); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a position-of-rank column that is no permutation"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutRoundtrip], + ); + assert_matches!( + report.downcast_ref::(), + Some(LayoutRoundtripError::RankInverse { + position, + rank: _, + roundtrip: Some(roundtrip), + }) if *position > BasePosition::MIN && *roundtrip == BasePosition::MIN, + ); + } + + /// Rejects a rank outside the position-of-rank column's domain. + /// + /// Returns [`WorldError::LayoutRoundtrip`] from [`LayoutRoundtripError::RankInverse`]. + /// + /// The sample reports the roundtrip as absent at the first sampled position. + #[test] + fn ranks_out_of_domain() { + let fixture = TamperFixture::publish("layout-ranks-out-of-domain"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.rank_of_position.name(), |path| { + constant_u32_column(path, NODES, u32::MAX); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses an out-of-domain rank"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutRoundtrip], + ); + assert_matches!( + report.downcast_ref::(), + Some(LayoutRoundtripError::RankInverse { + position, + rank, + roundtrip: None, + }) if *position == BasePosition::MIN && *rank == ImportanceRank::MAX, + ); + } + + /// Rejects a position-of-row column that is not a permutation. + /// + /// Returns [`WorldError::LayoutRoundtrip`] from [`LayoutRoundtripError::RowInverse`]. + /// + /// Every node claiming position zero keeps the length and the format. Position zero's own node + /// roundtrips, and the first sampled position past it does not. + #[test] + fn row_positions_constant() { + let fixture = TamperFixture::publish("layout-row-positions-constant"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.position_of_row.name(), |path| { + constant_u32_column(path, NODES, 0); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a position-of-row column that is no permutation"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutRoundtrip], + ); + assert_matches!( + report.downcast_ref::(), + Some(LayoutRoundtripError::RowInverse { + position, + row: _, + roundtrip: Some(roundtrip), + }) if *position > BasePosition::MIN && *roundtrip == BasePosition::MIN, + ); + } + + /// Rejects a node row outside the position-of-row column's domain. + /// + /// Returns [`WorldError::LayoutRoundtrip`] from [`LayoutRoundtripError::RowInverse`]. + /// + /// The sample reports the roundtrip as absent at the first sampled position. + #[test] + fn rows_out_of_domain() { + let fixture = TamperFixture::publish("layout-rows-out-of-domain"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.row_of_position.name(), |path| { + constant_u64_column(path, NODES, u64::MAX); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses an out-of-domain node row"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutRoundtrip], + ); + assert_matches!( + report.downcast_ref::(), + Some(LayoutRoundtripError::RowInverse { + position, + row, + roundtrip: None, + }) if *position == BasePosition::MIN && *row == NodeRowId::MAX, + ); + } + + /// Open accepts a malformed entry outside the roundtrip sample. + /// + /// The entry's first use refuses it. 65 nodes leave position 63 outside the sample, and the + /// missing coordinate is first required during full-base schedule construction. + #[test] + #[should_panic(expected = "should resolve the base node's position")] + fn row_position_unsampled() { + let fixture = TamperFixture::publish_with_nodes("layout-unsampled-row", 65); + let files = &fixture.generation().repository().files; + let layout = Layout::open(OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }) + .expect("the untampered layout opens"); + + // the sample uses index * 64 / 63 for index in 0..64, checking positions 0..=62 and 64. + let unsampled = layout.index[BasePosition::from_u64(63)]; + + let tampered = fixture.tamper(&files.position_of_row.name(), |path| { + set_row_position(path, unsampled, BasePosition::MAX); + }); + let layout = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect("should open with the malformed entry outside the sample"); + + assert_eq!(LayoutProvider::provide_position(&layout, unsampled), None); + assert_eq!( + ImportanceProvider::provide_priority(&layout, unsampled), + None + ); + + ScopeSchedule::from_base(&layout); + } + + /// Open refuses a malformed entry inside an exhaustive roundtrip sample. + /// + /// 64 nodes make the sample exhaustive, including the malformed entry at position 63. + #[test] + fn row_position_sampled() { + let fixture = TamperFixture::publish_with_nodes("layout-unsampled-row-control", 64); + let files = &fixture.generation().repository().files; + let layout = Layout::open(OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }) + .expect("the untampered layout opens"); + + let row = layout.index[BasePosition::from_u64(63)]; + + let tampered = fixture.tamper(&files.position_of_row.name(), |path| { + set_row_position(path, row, BasePosition::MAX); + }); + let report = Layout::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("should reject the malformed entry inside the sample"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::LayoutRoundtrip], + ); + assert_matches!( + report.downcast_ref::(), + Some(LayoutRoundtripError::RowInverse { + position, + row: observed_row, + roundtrip: Some(roundtrip), + }) if *position == BasePosition::from_u64(63) && *observed_row == row && *roundtrip == BasePosition::MAX, + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/mod.rs b/libs/@local/graph/atlas/src/serve/world/mod.rs new file mode 100644 index 00000000000..896cfbac4be --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/mod.rs @@ -0,0 +1,225 @@ +//! Opened artifacts of one fitted generation. +//! +//! [`World`] checks that its components share a node domain. A component query taking an +//! [`Epoch`](super::delta::epoch::Epoch) answers at that epoch's captured revision, and a `base_` +//! query reads the fitted generation alone. + +use alloc::sync::Arc; + +use error_stack::{Report, ReportSink, ResultExt as _, TryReportTupleExt as _}; + +use self::cache::Cache; +use super::{ + schedule::{BucketSchedule, ScopeSchedule}, + secret::ServeSecret, +}; +use crate::{file::generation::Generation, math::Bounds2}; + +mod cache; +mod encoding; +mod error; +mod geometry; +pub(crate) mod layout; +pub(crate) mod node_importance; +mod node_index; +mod ontology; +pub(crate) mod topology; + +pub(crate) use self::{ + error::WorldError, layout::Layout, node_index::NodeIndex, ontology::Ontology, + topology::Topology, +}; + +/// A generation and the secret used to derive its wire-id codecs. +#[derive(Clone, Copy)] +pub(crate) struct OpenOptions<'context> { + /// The published generation whose artifacts open. + pub generation: &'context Generation, + /// The server secret the wire-id codecs derive from. + pub secret: &'context ServeSecret, +} + +/// The serving artifacts of one fitted generation with matching component node counts. +#[derive(Debug)] +pub(crate) struct World { + /// The published generation the components opened from. + generation: Generation, + + /// The generation's recorded bucket schedule, validated within the key width. + schedule: BucketSchedule, + + /// Node coordinates, ranks and identities in fitted order. + pub layout: Layout, + /// Edge endpoints, adjacency and identities. + pub topology: Topology, + /// Values derived from the components once, at first use. + cache: Cache, + + /// Ontology-type identities, postings and the type closure. + pub ontology: Ontology, +} + +impl World { + /// Opens the generation's serving components and checks their shared node domain. + /// + /// # Errors + /// + /// Returns [`WorldError`] for invalid schedules, component opening failures or mismatched node + /// counts. + pub(crate) fn open( + generation: Generation, + secret: &ServeSecret, + ) -> Result> { + let options = OpenOptions { + generation: &generation, + secret, + }; + + let schedule = + BucketSchedule::new(generation.repository().metadata.reproducibility.config.lod) + .change_context(WorldError::BucketSchedule); + let layout = Layout::open(options); + let topology = Topology::open(options); + let ontology = Ontology::open(options); + + let (schedule, layout, topology, ontology) = + (schedule, layout, topology, ontology).try_collect()?; + + let this = Self { + generation, + schedule, + layout, + topology, + cache: Cache::new(), + ontology, + }; + + let mut sink = ReportSink::new_armed(); + + let layout = layout::LayoutProvider::provide_node_count(&this.layout); + let topology = topology::TopologyProvider::provide_node_count(&this.topology); + let ontology = this.ontology.node_count(); + if layout != topology || layout != ontology { + sink.capture(WorldError::NodeCountMismatch { + layout, + topology, + ontology, + }); + } + + sink.finish_ok(this) + } + + /// Returns the complete base cascade, derived from the layout on the first call. + /// + /// # Panics + /// + /// Panics during the first derivation if a base node has no position or priority, the + /// condition [`ScopeSchedule::from_base`] refuses and opening does not rule out for unsampled + /// rows. + pub(crate) fn base_scope_schedule(&self) -> &Arc { + self.cache.base_scope_schedule(&self.layout) + } + + /// Returns the published generation the components opened from. + pub(crate) const fn generation(&self) -> &Generation { + &self.generation + } + + /// Returns the generation's recorded bucket schedule. + pub(crate) const fn schedule(&self) -> BucketSchedule { + self.schedule + } + + /// Returns the recorded world frame. + /// + /// The generation's metadata records this as the frame its normalization mapped onto the + /// [wire frame](crate::salt::lod::stage::WIRE_FRAME). For canonical fit output that is the + /// fitted coordinates' bounds before normalization. Open reads the record without re-measuring + /// it. + pub(crate) const fn bounds(&self) -> Bounds2 { + self.generation.repository().metadata.evidence.lod.world + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use super::{World, error::WorldError, layout::LayoutProvider, topology::TopologyProvider}; + use crate::serve::tests::fixture::{ + ENDPOINTS, NODES, TamperFixture, respan_adjacency, retarget_postings_points, secret, + }; + + /// The synthetic generation opens as a world. + #[test] + fn untampered_opens() { + let fixture = TamperFixture::publish("world-untampered"); + + let world = World::open(fixture.generation().clone(), &secret()) + .expect("the untampered generation opens"); + let nodes = usize::try_from(NODES).expect("fixture node counts fit usize"); + + assert_eq!(LayoutProvider::provide_node_count(&world.layout), nodes); + assert_eq!(TopologyProvider::provide_node_count(&world.topology), nodes); + assert_eq!(world.ontology.node_count(), nodes); + assert_eq!( + TopologyProvider::provide_edge_count(&world.topology), + ENDPOINTS.len() + ); + } + + /// An extra adjacency node produces [`WorldError::NodeCountMismatch`]. + /// + /// Dropping a node row from the adjacency would drop that node's edge slots with it and move + /// the edge domain in the same tamper. The tamper therefore adds a row, and widening is as + /// much a producer bug as truncation is. + #[test] + fn adjacency_extra_node_row() { + let fixture = TamperFixture::publish("world-adjacency-extra-node-row"); + let nodes = usize::try_from(NODES).expect("fixture node counts fit usize"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.adjacency.name(), |path| { + respan_adjacency(path, nodes + 1, &ENDPOINTS); + }); + let report = World::open(tampered, &secret()) + .expect_err("open refuses an adjacency spanning an extra node row"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::NodeCountMismatch { + layout, + topology, + ontology, + }] if *layout == nodes && *topology == nodes + 1 && *ontology == nodes, + ); + } + + /// Postings beyond the layout's node count produce [`WorldError::NodeCountMismatch`]. + /// + /// Narrowing the postings' point domain can strand a membership position outside it, which + /// the postings contract refuses first and under its own name. The tamper therefore adds a + /// row. + #[test] + fn postings_points_wide() { + let fixture = TamperFixture::publish("world-postings-points-wide"); + let nodes = usize::try_from(NODES).expect("fixture node counts fit usize"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.postings.name(), |path| { + retarget_postings_points(path, NODES + 1); + }); + let report = World::open(tampered, &secret()) + .expect_err("open refuses a postings point domain above the node count"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::NodeCountMismatch { + layout, + topology, + ontology, + }] if *layout == nodes && *topology == nodes && *ontology == nodes + 1, + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/node_importance.rs b/libs/@local/graph/atlas/src/serve/world/node_importance.rs new file mode 100644 index 00000000000..086ce63a134 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/node_importance.rs @@ -0,0 +1,233 @@ +//! Importance ranks in base-position order. +//! +//! [`ImportanceRank`] orders nodes by importance, with zero first. [`BasePosition`] indexes their +//! geometry. The inverse mappings support access in either order. + +use core::ops::Index; + +use error_stack::{Report, ReportSink, ResultExt as _, TryReportTupleExt as _}; + +use super::{OpenOptions, error::WorldError}; +use crate::{ + identity::{BasePosition, Column, ImportanceRank, NodeRowId}, + postgres::id::ArchivedEntityId, +}; + +/// Node-selection order, with lower values more prominent. +/// +/// Identity keys give nodes without an [`ImportanceRank`] a stable order after ranked nodes. +// The derived `Ord` compares the variant before its contents: the declaration order is the +// priority order, and every rank precedes every identity. +#[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord)] +pub(crate) enum NodePriority { + /// A fitted node's importance rank. + Rank(ImportanceRank), + /// An unranked node's entity key, ordering it after every ranked node. + Identity(ArchivedEntityId), +} + +/// Node ordering over allocated rows, independent of visibility. +pub(crate) trait ImportanceProvider { + /// Returns the node's priority, or [`None`] if no priority is available for the row. + fn provide_priority(&self, node: NodeRowId) -> Option; +} + +impl ImportanceProvider for &T { + fn provide_priority(&self, node: NodeRowId) -> Option { + T::provide_priority(self, node) + } +} + +/// Inverse mappings between importance ranks and base positions. +#[derive(Debug)] +pub(crate) struct NodeImportance { + /// The rank of every base position. + lookup: Column, + /// The base position of every rank, the inverse of `lookup`. + reverse: Column, +} + +impl NodeImportance { + /// Opens the rank permutations and checks their lengths. + /// + /// # Errors + /// + /// Returns [`WorldError`] for artifact opening or mismatched rank counts. + pub(crate) fn open( + OpenOptions { generation, .. }: OpenOptions<'_>, + ) -> Result> { + let files = &generation.repository().files; + + let lookup = files + .rank_of_position + .open(generation) + .change_context(WorldError::Open { + file: files.rank_of_position.name(), + }); + + let reverse = files + .position_of_rank + .open(generation) + .change_context(WorldError::Open { + file: files.position_of_rank.name(), + }); + + let (lookup, reverse) = (lookup, reverse).try_collect()?; + + let this = Self { lookup, reverse }; + + let mut sink = ReportSink::new_armed(); + + if this.lookup.len() != this.reverse.len() { + sink.capture(WorldError::NodeImportanceCountMismatch { + lookup: this.lookup.len(), + reverse: this.reverse.len(), + }); + } + + sink.finish_ok(this) + } + + /// Returns the rank at `index`, or [`None`] outside the position domain. + pub(crate) fn lookup(&self, index: BasePosition) -> Option { + self.lookup.view().get(index).copied() + } + + /// Returns the position at `index`, or [`None`] outside the rank domain. + pub(crate) fn reverse(&self, index: ImportanceRank) -> Option { + self.reverse.view().get(index).copied() + } + + /// Returns the number of ranked positions, the length of both columns. + pub(crate) fn len(&self) -> usize { + self.lookup.len() + } +} + +impl Index for NodeImportance { + type Output = ImportanceRank; + + fn index(&self, index: BasePosition) -> &Self::Output { + &self.lookup.view()[index] + } +} + +impl Index for NodeImportance { + type Output = BasePosition; + + fn index(&self, index: ImportanceRank) -> &Self::Output { + &self.reverse.view()[index] + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use hashql_core::id::Id as _; + use uuid::Uuid; + + use super::{NodeImportance, NodePriority}; + use crate::{ + identity::ImportanceRank, + postgres::id::ArchivedEntityId, + serve::{ + tests::fixture::{NODES, TamperFixture, secret, shorten_u32_column}, + world::{OpenOptions, error::WorldError}, + }, + }; + + /// Builds an entity key from `u128` web and entity UUID values. + fn entity(web: u128, id: u128) -> ArchivedEntityId { + ArchivedEntityId { + web_id: Uuid::from_u128(web).into(), + entity_uuid: Uuid::from_u128(id).into(), + } + } + + /// Every rank precedes every identity, including the domain extremes. + #[test] + fn priority_boundary() { + let min_rank = NodePriority::Rank(ImportanceRank::MIN); + let max_rank = NodePriority::Rank(ImportanceRank::MAX); + let min_identity = NodePriority::Identity(entity(0, 0)); + let max_identity = NodePriority::Identity(entity(u128::MAX, u128::MAX)); + + assert!(min_rank < max_rank, "should order ranks by their value"); + assert!( + min_identity < max_identity, + "should order identities by their value" + ); + assert!( + max_rank < min_identity, + "should order every rank before every identity" + ); + } + + /// Identity ordering compares the web before the entity UUID. + #[test] + fn priority_identity_order() { + let same_web_low = NodePriority::Identity(entity(1, 1)); + let same_web_high = NodePriority::Identity(entity(1, 2)); + let other_web_low = NodePriority::Identity(entity(2, 0)); + + assert!( + same_web_low < same_web_high, + "should break ties on the entity UUID" + ); + assert!( + same_web_high < other_web_low, + "should compare the web before the entity UUID" + ); + } + + /// Open refuses a rank-of-position column short of the position-of-rank column. + /// + /// Returns [`WorldError::NodeImportanceCountMismatch`]. + #[test] + fn rank_column_short() { + let fixture = TamperFixture::publish("node-importance-rank-column-short"); + let columns = usize::try_from(NODES).expect("fixture node counts fit usize"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.rank_of_position.name(), |path| { + shorten_u32_column(path, NODES - 1); + }); + let report = NodeImportance::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short rank-of-position column"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::NodeImportanceCountMismatch { lookup, reverse }] + if *lookup == columns - 1 && *reverse == columns, + ); + } + + /// Open refuses a position-of-rank column short of the rank-of-position column. + /// + /// Returns [`WorldError::NodeImportanceCountMismatch`]. + #[test] + fn rank_positions_short() { + let fixture = TamperFixture::publish("node-importance-rank-positions-short"); + let columns = usize::try_from(NODES).expect("fixture node counts fit usize"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.position_of_rank.name(), |path| { + shorten_u32_column(path, NODES - 1); + }); + let report = NodeImportance::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short position-of-rank column"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::NodeImportanceCountMismatch { lookup, reverse }] + if *lookup == columns && *reverse == columns - 1, + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/node_index.rs b/libs/@local/graph/atlas/src/serve/world/node_index.rs new file mode 100644 index 00000000000..b948b0084c0 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/node_index.rs @@ -0,0 +1,379 @@ +//! The permutation between stable node rows and fitted storage positions. +//! +//! [`NodeRowId`] identifies a node independently of layout order. [`BasePosition`] indexes the +//! geometry and importance columns. + +use core::ops::Index; + +use error_stack::{Report, ReportSink, ResultExt as _, TryReportTupleExt as _}; + +use super::{OpenOptions, encoding::Encoding, error::WorldError}; +use crate::{ + dataset::auxiliary::Legend, + identity::{BasePosition, Column, NodeRowId}, + postgres::id::ArchivedEntityId, + salt::fit::prepare::identity::IdentityTableArchive, + serve::{ + codec::{EncodedRowId, RowDomain}, + delta::{ + epoch::Epoch, + overlay::{NaiveIdentityProvider, VersionedIdentityProvider as _}, + }, + }, +}; + +/// Node identities and the inverse mappings between row and base-position order. +#[derive(Debug)] +pub(crate) struct NodeIndex { + /// The entity key and display payload of every fitted row. + pub identity: IdentityTableArchive, + /// The wire-id codec of the node row domain, every fitted row precomputed. + encoding: Encoding, + + /// The row of every base position. + lookup: Column, + /// The base position of every row, the inverse of `lookup`. + reverse: Column, +} + +impl NodeIndex { + /// Opens the identity and permutation artifacts and checks their counts. + /// + /// The fitted position domain needs an exclusive bound representable by `u32`. + /// + /// # Errors + /// + /// Returns [`WorldError`] for artifact opening, a fitted row count past `u32::MAX` or + /// mismatched row counts. + pub(crate) fn open( + options @ OpenOptions { + generation, + secret: _, + }: OpenOptions<'_>, + ) -> Result> { + let files = &generation.repository().files; + + let identity = files + .node_identities + .open(generation) + .change_context(WorldError::Open { + file: files.node_identities.name(), + }); + + let lookup = files + .row_of_position + .open(generation) + .change_context(WorldError::Open { + file: files.row_of_position.name(), + }); + + let reverse = files + .position_of_row + .open(generation) + .change_context(WorldError::Open { + file: files.position_of_row.name(), + }); + + let encoding = lookup.and_then(|column: Column| { + let nodes = column.len(); + u32::try_from(nodes).map_err(|error| { + Report::new(error).change_context(WorldError::TooManyNodes { nodes }) + })?; + + let encoding = Encoding::open(options, RowDomain::from_length(nodes))?; + + Ok((encoding, column)) + }); + + let (identity, (encoding, lookup), reverse) = + (identity, encoding, reverse).try_collect()?; + + let this = Self { + identity, + encoding, + lookup, + reverse, + }; + + let mut sink = ReportSink::new_armed(); + + if this.identity.len() != this.lookup.len() as u64 + || this.identity.len() != this.reverse.len() as u64 + { + sink.capture(WorldError::NodeIndexCountMismatch { + identity: this.identity.len(), + lookup: this.lookup.len(), + reverse: this.reverse.len(), + }); + } + + sink.finish_ok(this) + } + + /// Returns the entity's row if its identity is live at the captured revision. + /// + /// # Panics + /// + /// Panics if this index does not belong to the epoch's world. + pub(crate) fn row_of(&self, epoch: &Epoch, entity_id: ArchivedEntityId) -> Option { + let provider = epoch + .nodes(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)); + + provider.provide_row_of_at(entity_id, epoch.revision()) + } + + /// Returns the row's entity key if its identity is live at the captured revision. + /// + /// # Panics + /// + /// Panics if this index does not belong to the epoch's world. + pub(crate) fn key_of(&self, epoch: &Epoch, row: NodeRowId) -> Option { + let provider = epoch + .nodes(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)); + + provider.provide_key_of_at(row, epoch.revision()) + } + + /// Borrows the legend of a live node identity at the captured revision. + /// + /// # Panics + /// + /// Panics if this index does not belong to the epoch's world. + pub(crate) fn payload<'scene>( + &'scene self, + epoch: &'scene Epoch, + row: NodeRowId, + ) -> Option<&'scene Legend> { + epoch + .nodes(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)) + .into_payload_of_row_at(row, epoch.revision()) + } + + /// Encodes a node row as its wire id, independent of liveness or allocation. + /// + /// # Panics + /// + /// Panics if `row` lies at or beyond [`WIRE_ROW_BOUND`](crate::serve::codec::WIRE_ROW_BOUND) + /// and its `usize` conversion misses the precomputed table. + /// + /// # Warning + /// + /// On a 32-bit target, truncating a large row for the cache lookup can return another fitted + /// row's wire id, as [`Encoding::encode`] documents. + pub(crate) fn encode(&self, row: NodeRowId) -> EncodedRowId { + self.encoding.encode(row) + } + + /// Decodes rows allocated and live in the supplied [`Epoch`]. + /// + /// Returns [`None`] when the row the wire value decodes to lies outside the allocated row + /// domain, and for a row whose identity is not live at the captured revision. The check is on + /// the decoded row, not the wire value: the permutation can carry a large wire value to a + /// small allocated row. + /// + /// # Panics + /// + /// Panics if this index does not belong to `epoch`'s [`World`](super::World). + pub(crate) fn decode(&self, epoch: &Epoch, wire: EncodedRowId) -> Option { + let provider = epoch + .nodes(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)); + let row = self.encoding.decode(wire, provider.provide_domain())?; + + provider + .provide_key_of_at(row, epoch.revision()) + .map(|_key| row) + } + + /// Returns the fitted position of `index`, or [`None`] outside the fitted row domain. + pub(super) fn base_reverse(&self, index: NodeRowId) -> Option { + self.reverse.view().get(index).copied() + } + + /// Returns a live node identity's fitted position at the captured revision. + /// + /// Returns [`None`] for rows without a fitted position or a live identity. + /// + /// # Panics + /// + /// Panics if this index does not belong to the epoch's world. + pub(crate) fn reverse(&self, epoch: &Epoch, index: NodeRowId) -> Option { + let provider = epoch + .nodes(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)); + + let position = self.base_reverse(index)?; + provider + .permits_row(index, Some(epoch.revision())) + .then_some(position) + } + + /// Returns the number of fitted positions, the length of the row-of-position column. + pub(crate) fn len(&self) -> usize { + self.lookup.len() + } + + /// Returns the exclusive bound of the fitted row domain: the first row no base position holds. + pub(crate) fn base_node_bound(&self) -> NodeRowId { + self.reverse.view().bound() + } +} + +impl Index for NodeIndex { + type Output = NodeRowId; + + fn index(&self, index: BasePosition) -> &Self::Output { + &self.lookup.view()[index] + } +} + +impl Index for NodeIndex { + type Output = BasePosition; + + fn index(&self, index: NodeRowId) -> &Self::Output { + &self.reverse.view()[index] + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use super::NodeIndex; + use crate::{ + file::repository::{IntegrityVerificationError, OpenBindingError}, + identity::NodeRowId, + salt::fit::prepare::identity::OpenIdentityTableArchiveError, + serve::{ + tests::fixture::{NODES, TamperFixture, secret, shorten_entities, shorten_u32_column}, + world::{OpenOptions, error::WorldError}, + }, + }; + + /// Rejects a node identity table shorter than the position columns. + /// + /// Returns [`WorldError::NodeIndexCountMismatch`]. + #[test] + fn node_identities_short() { + let fixture = TamperFixture::publish("node-index-identities-short"); + let columns = usize::try_from(NODES).expect("fixture node counts fit usize"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.node_identities.name(), |path| { + shorten_entities::(path, NODES - 1, 0); + }); + let report = NodeIndex::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short node identity table"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::NodeIndexCountMismatch { + identity, + lookup, + reverse, + }] if *identity == NODES - 1 && *lookup == columns && *reverse == columns, + ); + } + + /// Rejects a position-of-row column shorter than the node identity table. + /// + /// Returns [`WorldError::NodeIndexCountMismatch`]. + #[test] + fn row_positions_short() { + let fixture = TamperFixture::publish("node-index-row-positions-short"); + let columns = usize::try_from(NODES).expect("fixture node counts fit usize"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.position_of_row.name(), |path| { + shorten_u32_column(path, NODES - 1); + }); + let report = NodeIndex::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short position-of-row column"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::NodeIndexCountMismatch { + identity, + lookup, + reverse, + }] if *identity == NODES && *lookup == columns && *reverse == columns - 1, + ); + } + + /// Rejects a published file rewritten in place. + /// + /// Returns [`WorldError::Open`] from [`IntegrityVerificationError::Checksum`]. + /// + /// The rewrite keeps the table's format and would fail the count check if it reached it. The + /// digest check runs first and names the file with both digests. + #[test] + fn corruption_rewritten_file() { + let fixture = TamperFixture::publish("node-index-corruption-rewritten"); + let files = &fixture.generation().repository().files; + let name = files.node_identities.name(); + + shorten_entities::(&fixture.generation().path_of(&name), NODES - 1, 0); + let report = NodeIndex::open(OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }) + .expect_err("open refuses a published file rewritten in place"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::Open { file }] if *file == name, + ); + assert_matches!( + report.downcast_ref::>(), + Some(OpenBindingError::Integrity(IntegrityVerificationError::Checksum { + file, + received, + })) if file.name == name + && file.hash == files.node_identities.hash() + && *received != file.hash, + ); + } + + /// Rejects a generation missing a published file. + /// + /// Returns [`WorldError::Open`] from [`IntegrityVerificationError::Io`]. + #[test] + fn corruption_missing_file() { + let fixture = TamperFixture::publish("node-index-corruption-missing"); + let name = fixture + .generation() + .repository() + .files + .node_identities + .name(); + + std::fs::remove_file(fixture.generation().path_of(&name)) + .expect("the published file removes"); + let report = NodeIndex::open(OpenOptions { + generation: fixture.generation(), + secret: &secret(), + }) + .expect_err("open refuses a generation missing a published file"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::Open { file }] if *file == name, + ); + assert_matches!( + report.downcast_ref::>(), + Some(OpenBindingError::Integrity(IntegrityVerificationError::Io { + name: missing, + error, + })) if *missing == name && error.kind() == std::io::ErrorKind::NotFound, + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/ontology.rs b/libs/@local/graph/atlas/src/serve/world/ontology.rs new file mode 100644 index 00000000000..ab316018a18 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/ontology.rs @@ -0,0 +1,221 @@ +//! Ontology types of one fitted generation: identities, membership postings and the type closure. +//! +//! [`OntologyRowId`] names a type row. The postings record which base positions each type covers, +//! and the closure derives every type's ancestors and its nearest icon-bearing ancestor from the +//! recorded parent graph. + +use error_stack::{Report, ReportSink, ResultExt as _, TryReportTupleExt as _}; + +use super::{OpenOptions, error::WorldError}; +use crate::{ + dataset::auxiliary::Icon, + identity::OntologyRowId, + postgres::id::ArchivedOntologyTypeUuid, + salt::{ + fit::prepare::identity::IdentityTableArchive, + postings::{artifact::PostingsArchive, closure::ClosureMap}, + }, + serve::delta::{ + epoch::Epoch, + overlay::{NaiveIdentityProvider, VersionedIdentityProvider as _}, + }, +}; + +/// The ontology-type identities, membership postings and type closure of one generation. +#[derive(Debug)] +pub(crate) struct Ontology { + /// The type key and icon of every ontology row. + pub identity: IdentityTableArchive, + + /// Which base positions each type covers. + postings: PostingsArchive, + /// Every type's ancestor set and nearest icon-bearing ancestor. + closure: ClosureMap, +} + +impl Ontology { + /// Opens the ontology artifacts, checks their counts and derives the type closure. + /// + /// # Errors + /// + /// Returns [`WorldError::Open`] for an artifact that fails to open, then together + /// [`WorldError::TooManyPositions`], [`WorldError::TooManyTypes`] and + /// [`WorldError::OntologyCountMismatch`] for counts outside the address space or a type + /// count the identity table and the postings disagree on, then [`WorldError::OntologyClosure`] + /// when the recorded parent graph holds a cycle. + pub(crate) fn open( + OpenOptions { generation, .. }: OpenOptions<'_>, + ) -> Result> { + let files = &generation.repository().files; + + let identity: Result, _> = + files + .ontology_identities + .open(generation) + .change_context(WorldError::Open { + file: files.ontology_identities.name(), + }); + + let postings: Result = + files + .postings + .open(generation) + .change_context(WorldError::Open { + file: files.postings.name(), + }); + + let (identity, postings) = (identity, postings).try_collect()?; + + let mut sink = ReportSink::new_armed(); + + if let Err(error) = usize::try_from(postings.points()) { + sink.capture( + Report::new(error).change_context(WorldError::TooManyPositions { + positions: postings.points(), + }), + ); + } + + if let Err(error) = usize::try_from(identity.len()) { + sink.capture(Report::new(error).change_context(WorldError::TooManyTypes { + types: identity.len(), + })); + } + + if identity.len() != postings.types() { + sink.capture(WorldError::OntologyCountMismatch { + identity: identity.len(), + postings: postings.types(), + }); + } + + sink.finish()?; + + let closure = ClosureMap::new(&postings, identity.displayed_rows()) + .change_context(WorldError::OntologyClosure)?; + + Ok(Self { + identity, + postings, + closure, + }) + } + + /// Returns the number of base positions the postings cover. + #[expect( + clippy::cast_possible_truncation, + reason = "open refuses a point count outside `usize`" + )] + pub(crate) fn node_count(&self) -> usize { + self.postings.points() as usize + } + + /// Returns the row's type key if its identity is live at the captured revision. + /// + /// # Panics + /// + /// Panics if this ontology does not belong to the epoch's world. + pub(crate) fn key_of( + &self, + epoch: &Epoch, + row: OntologyRowId, + ) -> Option { + epoch + .ontology(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)) + .provide_key_of_at(row, epoch.revision()) + } + + /// Borrows the icon a type row carries itself. + /// + /// Returns [`None`] unless the row's identity is live at the captured revision. The empty icon + /// is the payload of a row that displays none. [`Self::icon`] resolves such a row through the + /// type closure instead. + /// + /// # Panics + /// + /// Panics if this ontology does not belong to the epoch's world. + pub(crate) fn payload<'scene>( + &'scene self, + epoch: &'scene Epoch, + row: OntologyRowId, + ) -> Option<&'scene Icon> { + epoch + .ontology(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)) + .into_payload_of_row_at(row, epoch.revision()) + } + + /// Borrows the icon displayed for a type row at the captured revision. + /// + /// A row with no icon of its own displays its nearest icon-bearing ancestor's, as the type + /// closure resolved it at open, and the empty icon when no ancestor carries one. Returns + /// [`None`] when the resolved row's identity is not live at the revision. + /// + /// # Panics + /// + /// Panics if this ontology does not belong to the epoch's world. + pub(crate) fn icon<'scene>( + &'scene self, + epoch: &'scene Epoch, + row: OntologyRowId, + ) -> Option<&'scene Icon> { + let source = self + .closure + .icon_source(row) + .map_or(row, |icon| icon.source); + self.payload(epoch, source) + } + + /// Returns the fitted type identities, independent of any epoch. + pub(crate) const fn identity( + &self, + ) -> &IdentityTableArchive { + &self.identity + } + + /// Returns the membership postings over base positions. + pub(crate) const fn postings(&self) -> &PostingsArchive { + &self.postings + } + + /// Returns the type closure derived at open. + pub(crate) const fn closure(&self) -> &ClosureMap { + &self.closure + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use super::Ontology; + use crate::serve::{ + tests::fixture::{TYPES, TamperFixture, secret, shorten_ontology}, + world::{OpenOptions, error::WorldError}, + }; + + /// Open refuses an ontology identity table short of the postings' type domain. + /// + /// Returns [`WorldError::OntologyCountMismatch`]. + #[test] + fn ontology_identities_short() { + let fixture = TamperFixture::publish("ontology-identities-short"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.ontology_identities.name(), |path| { + shorten_ontology(path, TYPES - 1); + }); + let report = Ontology::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short ontology identity table"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::OntologyCountMismatch { identity, postings }] + if *identity == TYPES - 1 && *postings == TYPES, + ); + } +} diff --git a/libs/@local/graph/atlas/src/serve/world/topology.rs b/libs/@local/graph/atlas/src/serve/world/topology.rs new file mode 100644 index 00000000000..b2d9e1ac5cb --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/world/topology.rs @@ -0,0 +1,333 @@ +//! Directed graph queries in stable node and edge row order. +//! +//! [`Topology`] combines fitted endpoint bindings with the changes captured by an [`Epoch`]. +//! Adjacency queries exclude withdrawn edges and edges with an invisible endpoint node. + +use error_stack::{Report, ReportSink, ResultExt as _, TryReportTupleExt as _}; + +use super::{OpenOptions, error::WorldError}; +use crate::{ + dataset::auxiliary::Legend, + identity::{Column, EdgeRowId, NodeRowId}, + postgres::id::ArchivedEntityId, + salt::{ + adjacency::{AdjacencyArchive, EdgeList}, + fit::prepare::identity::IdentityTableArchive, + }, + serve::delta::{ + epoch::Epoch, + overlay::{NaiveIdentityProvider, VersionedIdentityProvider as _}, + topology::provider::{NaiveTopologyProvider, VersionedTopologyProvider as _}, + }, +}; + +/// Endpoint and adjacency lookups over allocated node and edge rows. +pub(crate) trait TopologyProvider { + /// Returns the exclusive bound of the zero-based node domain, including absent rows. + fn provide_node_count(&self) -> usize; + /// Returns the exclusive bound of the zero-based edge domain, including absent rows. + fn provide_edge_count(&self) -> usize; + /// Returns the visible `[source, target]` pair, or [`None`] for an absent edge. + fn provide_endpoints(&self, edge: EdgeRowId) -> Option<[NodeRowId; 2]>; + /// Returns existing incoming edges in strictly ascending row order. + fn provide_incoming(&self, node: NodeRowId) -> impl Iterator; + /// Returns existing outgoing edges in strictly ascending row order. + fn provide_outgoing(&self, node: NodeRowId) -> impl Iterator; +} + +impl TopologyProvider for &T { + fn provide_node_count(&self) -> usize { + T::provide_node_count(self) + } + + fn provide_edge_count(&self) -> usize { + T::provide_edge_count(self) + } + + fn provide_endpoints(&self, edge: EdgeRowId) -> Option<[NodeRowId; 2]> { + T::provide_endpoints(self, edge) + } + + fn provide_incoming(&self, node: NodeRowId) -> impl Iterator { + T::provide_incoming(self, node) + } + + fn provide_outgoing(&self, node: NodeRowId) -> impl Iterator { + T::provide_outgoing(self, node) + } +} + +/// The fitted endpoint bindings and adjacency lists of one generation. +/// +/// Queries require CSR adjacency lists whose runs agree with the endpoint column. Opening compares +/// counts without establishing that correspondence. +#[derive(Debug)] +pub(crate) struct Topology { + /// The entity key and display payload of every fitted edge row. + pub identity: IdentityTableArchive, + + /// The incoming and outgoing edge lists of every fitted node row. + adjacency: AdjacencyArchive, + /// The `[source, target]` node rows of every fitted edge row. + endpoints: Column, +} + +impl Topology { + /// Opens the fitted topology and checks its edge counts. + /// + /// # Errors + /// + /// Returns [`WorldError`] for artifact opening, mismatched counts or an oversized edge domain. + pub(crate) fn open( + OpenOptions { generation, .. }: OpenOptions<'_>, + ) -> Result> { + let files = &generation.repository().files; + + let identity = files + .edge_identities + .open(generation) + .change_context(WorldError::Open { + file: files.edge_identities.name(), + }); + + let adjacency = files + .adjacency + .open(generation) + .change_context(WorldError::Open { + file: files.adjacency.name(), + }); + + let endpoints = files + .edge_endpoints + .open(generation) + .change_context(WorldError::Open { + file: files.edge_endpoints.name(), + }); + + let (identity, adjacency, endpoints) = (identity, adjacency, endpoints).try_collect()?; + + let this = Self { + identity, + adjacency, + endpoints, + }; + + let mut sink = ReportSink::new_armed(); + + if this.identity.len() != this.endpoints.len() as u64 + || this.identity.len() != this.adjacency.edges() + { + sink.capture(WorldError::TopologyCountMismatch { + identity: this.identity.len(), + endpoints: this.endpoints.len(), + adjacency: this.adjacency.edges(), + }); + } + + if let Err(error) = u32::try_from(this.identity.len()) { + sink.capture(Report::new(error).change_context(WorldError::TooManyEdges { + edges: this.identity.len(), + })); + } + + sink.finish_ok(this) + } + + /// Returns the visible `[source, target]` pair at the captured epoch's revision. + /// + /// Returns [`None`] for a withdrawn edge and for an edge whose endpoint node has no visible + /// placement at that revision. + /// + /// # Panics + /// + /// Panics if this topology does not belong to the epoch's world. + pub(crate) fn endpoints(&self, epoch: &Epoch, edge: EdgeRowId) -> Option<[NodeRowId; 2]> { + let endpoints = epoch + .topology(self) + .bind(NaiveTopologyProvider::from_ref(self)) + .provide_endpoints_at(edge, epoch.revision())?; + + endpoints + .iter() + .all(|&node| epoch.contains_node(node)) + .then_some(endpoints) + } + + /// Returns visible incoming edges in ascending row order at the captured revision. + /// + /// # Panics + /// + /// Panics if this topology does not belong to the epoch's world. + pub(crate) fn incoming<'epoch>( + &'epoch self, + epoch: &'epoch Epoch, + node: NodeRowId, + ) -> impl Iterator + 'epoch { + epoch + .topology(self) + .bind(NaiveTopologyProvider::from_ref(self)) + .into_incoming_at(node, epoch.revision()) + .filter(move |&edge| self.endpoints(epoch, edge).is_some()) + } + + /// Returns visible outgoing edges in ascending row order at the captured revision. + /// + /// # Panics + /// + /// Panics if this topology does not belong to the epoch's world. + pub(crate) fn outgoing<'epoch>( + &'epoch self, + epoch: &'epoch Epoch, + node: NodeRowId, + ) -> impl Iterator + 'epoch { + epoch + .topology(self) + .bind(NaiveTopologyProvider::from_ref(self)) + .into_outgoing_at(node, epoch.revision()) + .filter(move |&edge| self.endpoints(epoch, edge).is_some()) + } + + /// Returns the edge's row if its identity is live at the captured revision. + /// + /// # Panics + /// + /// Panics if this topology does not belong to the epoch's world. + pub(crate) fn row_of(&self, epoch: &Epoch, edge: ArchivedEntityId) -> Option { + epoch + .edges(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)) + .provide_row_of_at(edge, epoch.revision()) + } + + /// Returns the edge's entity key if its identity is live at the captured revision. + /// + /// # Panics + /// + /// Panics if this topology does not belong to the epoch's world. + pub(crate) fn key_of(&self, epoch: &Epoch, edge: EdgeRowId) -> Option { + epoch + .edges(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)) + .provide_key_of_at(edge, epoch.revision()) + } + + /// Borrows the legend of a live edge identity at the captured revision. + /// + /// # Panics + /// + /// Panics if this topology does not belong to the epoch's world. + pub(crate) fn payload<'scene>( + &'scene self, + epoch: &'scene Epoch, + edge: EdgeRowId, + ) -> Option<&'scene Legend> { + epoch + .edges(self) + .bind(NaiveIdentityProvider::from_ref(&self.identity)) + .into_payload_of_row_at(edge, epoch.revision()) + } +} + +impl TopologyProvider for Topology { + #[expect( + clippy::cast_possible_truncation, + reason = "the archive's validation converted its row count to `usize` before deriving the \ + node count from it, and an archive that opened has a representable count" + )] + fn provide_node_count(&self) -> usize { + self.adjacency.rows() as usize + } + + fn provide_edge_count(&self) -> usize { + self.endpoints.len() + } + + fn provide_endpoints(&self, edge: EdgeRowId) -> Option<[NodeRowId; 2]> { + self.endpoints.view().get(edge).copied() + } + + fn provide_incoming(&self, node: NodeRowId) -> impl Iterator { + self.adjacency + .incoming(node) + .into_iter() + .flat_map(EdgeList::iter) + } + + fn provide_outgoing(&self, node: NodeRowId) -> impl Iterator { + self.adjacency + .outgoing(node) + .into_iter() + .flat_map(EdgeList::iter) + } +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use super::Topology; + use crate::{ + identity::EdgeRowId, + serve::{ + tests::fixture::{ + EDGE_SEED, EDGES, ENDPOINTS, TamperFixture, secret, shorten_endpoints, + shorten_entities, + }, + world::{OpenOptions, error::WorldError}, + }, + }; + + /// Rejects an edge identity table shorter than the adjacency's edge domain. + /// + /// Returns [`WorldError::TopologyCountMismatch`]. + #[test] + fn edge_identities_short() { + let fixture = TamperFixture::publish("topology-edge-identities-short"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.edge_identities.name(), |path| { + shorten_entities::(path, EDGES - 1, EDGE_SEED); + }); + let report = Topology::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short edge identity table"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::TopologyCountMismatch { + identity, + endpoints, + adjacency, + }] if *identity == EDGES - 1 && *endpoints == ENDPOINTS.len() && *adjacency == EDGES, + ); + } + + /// Rejects an endpoint column shorter than the adjacency's edge domain. + /// + /// Returns [`WorldError::TopologyCountMismatch`]. + #[test] + fn endpoint_column_short() { + let fixture = TamperFixture::publish("topology-endpoint-column-short"); + let files = &fixture.generation().repository().files; + + let tampered = fixture.tamper(&files.edge_endpoints.name(), |path| { + shorten_endpoints(path, &ENDPOINTS[..ENDPOINTS.len() - 1]); + }); + let report = Topology::open(OpenOptions { + generation: &tampered, + secret: &secret(), + }) + .expect_err("open refuses a short endpoint column"); + + assert_matches!( + report.current_contexts().collect::>().as_slice(), + [WorldError::TopologyCountMismatch { + identity, + endpoints, + adjacency, + }] if *identity == EDGES && *endpoints == ENDPOINTS.len() - 1 && *adjacency == EDGES, + ); + } +}