diff --git a/libs/@local/graph/atlas/src/serve/hydrate/client.rs b/libs/@local/graph/atlas/src/serve/hydrate/client.rs new file mode 100644 index 00000000000..eac7ece0dd8 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/client.rs @@ -0,0 +1,560 @@ +use alloc::sync::Arc; +use core::pin::pin; + +use error_stack::{Report, ResultExt as _}; +use futures::TryStreamExt as _; +use hash_graph_postgres_store::store::{ + AsClient, PostgresStorePool, postgres::query::SelectCompiler, +}; +use hash_graph_store::{ + filter::{Filter, protection::PropertyProtectionFilter}, + pool::StorePool as _, + subgraph::temporal_axes::{QueryTemporalAxes, QueryTemporalAxesUnresolved}, +}; +use hashql_core::{ + collections::FastHashMap, + id::{Id as _, IdSlice}, +}; +use tokio::{runtime::Handle, try_join}; +use tokio_postgres::GenericClient; +use type_system::{ + knowledge::entity::id::EntityId, + ontology::{ + entity_type::EntityTypeUuid, + id::{OntologyTypeUuid, VersionedUrl}, + }, +}; + +use super::{ + columns::{EdgeSlot, NodeSlot}, + locate::{ + LocateEntity, LocateLink, LocateNode, LocateProperties, LocateRequest, LocateResolver, + }, + statements::{DetailColumns, TypeColumns, TypeUrlColumns, identity_filter}, + type_urls::TypeUrlResolver, +}; +use crate::{postgres::id::ArchivedEntityId, serve::visibility::VisibilityActor}; + +/// A failure during detail hydration against the store. +#[derive(Debug, Copy, Clone)] +pub(crate) enum HydrateError { + /// No connection was available for the query. + Connect, + /// A store query failed. + Query, + /// The query returned too many rows. + TooManyRows, +} + +impl core::fmt::Display for HydrateError { + fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Connect => { + write!(fmt, "the detail hydration reached no store connection") + } + Self::Query => fmt.write_str("the detail hydration query failed"), + Self::TooManyRows => fmt.write_str("the detail hydration returned too many rows"), + } + } +} + +impl core::error::Error for HydrateError {} + +/// Resolves each requested node's direct-type URLs from the store, filling `nodes` in place. +/// +/// # Errors +/// +/// Returns [`HydrateError`] when a store read fails. +/// +/// # Panics +/// +/// Panics if the store returns a row outside the requested identities, since the slot lookup +/// indexes by that identity, and if a column fails to decode or a direct-type count is negative. +async fn read_types( + client: &impl GenericClient, + nodes: &mut IdSlice>, + temporal_axes: &QueryTemporalAxes, +) -> Result<(), Report> { + let filter = identity_filter(nodes.iter().map(|node| EntityId::from(node.identity))); + + let mut compiler = SelectCompiler::new(Some(temporal_axes), false); + compiler + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + + let columns = TypeColumns::select(&mut compiler); + let (statement, parameters) = compiler.compile(); + + let rows = client + .query_raw(&statement, parameters) + .await + .change_context(HydrateError::Query)?; + + let lookup: FastHashMap<_, _> = nodes + .iter_enumerated_mut() + .map(|(slot, node)| { + node.details = None; + (node.identity, slot) + }) + .collect(); + + let mut rows = pin!(rows); + while let Some(row) = rows.try_next().await.change_context(HydrateError::Query)? { + let slot = lookup[&columns.entity_id(&row)]; + nodes[slot].details.get_or_insert_with(|| LocateNode { + type_urls: columns.direct_type_urls(&row), + }); + } + + Ok(()) +} + +/// Resolves `source`'s actor-masked, `cap`-limited scalar properties. +/// +/// # Errors +/// +/// Returns [`HydrateError`] when a store read fails or when the store answers more than one row +/// for `source`. +/// +/// # Panics +/// +/// Panics if a column fails to decode or the scalar-property aggregate is not a JSON object. +async fn read_detail( + client: &(impl GenericClient + Sync), + source: ArchivedEntityId, + protection: Option<&PropertyProtectionFilter<'_, '_>>, + cap: usize, + temporal_axes: &QueryTemporalAxes, +) -> Result, Report> { + let filter = identity_filter([source.into()]); + + let mut compiler = SelectCompiler::new(Some(temporal_axes), false); + compiler + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + + let columns = DetailColumns::select(&mut compiler, protection); + let (statement, parameters) = compiler.compile(); + + let stream = client + .query_raw(&statement, parameters) + .await + .change_context(HydrateError::Query)?; + + let mut stream = pin!(stream); + let Some(row) = stream + .try_next() + .await + .change_context(HydrateError::Query)? + else { + return Ok(None); + }; + + if stream + .try_next() + .await + .change_context(HydrateError::Query)? + .is_some() + { + return Err(Report::new(HydrateError::TooManyRows)); + } + + let (values, complete) = columns.capped_properties(&row, cap); + Ok(Some(LocateProperties { values, complete })) +} + +/// A PostgreSQL client for live graph hydration. +/// +/// Entity lookups select current, non-archived editions. Ontology lookups include archived types. +/// +/// Property protection follows the store pool's configured rules for the request's actor. +/// +/// # Execution +/// +/// Resolution blocks the calling thread. Use a blocking worker while the supplied runtime processes +/// I/O. +/// +/// # Panics +/// +/// Resolution panics in an asynchronous execution context or if a returned column fails to decode. +/// +/// Entity lookups also panic if a query returns an unrequested identity, a direct-type count is +/// negative, or a scalar-property aggregate is not a JSON object. +#[derive(Debug)] +pub(crate) struct GraphDatabaseClient { + pool: Arc, + runtime: Handle, +} + +impl GraphDatabaseClient { + /// Builds a client hydrating detail through `pool`, with resolver calls run on `runtime`. + #[must_use] + pub(crate) const fn new(pool: Arc, runtime: Handle) -> Self { + Self { pool, runtime } + } + + /// Holds one connection for the duration of one hydration. + /// + /// # Errors + /// + /// Returns [`HydrateError::Connect`] where the pool cannot supply a connection. + async fn connection(&self) -> Result> { + self.pool + .acquire(None) + .await + .change_context(HydrateError::Connect) + } + + /// Resolves node types and the source's actor-masked properties. + /// + /// # Errors + /// + /// Returns [`HydrateError`] if no connection is available, a query fails, or the source-detail + /// query returns more than one row. + /// + /// # Panics + /// + /// Panics if the store returns a row outside the request domain, a column fails to decode, a + /// direct-type count is negative, or the scalar-property aggregate is not a JSON object. + #[tracing::instrument(skip_all, fields(points = nodes.len()))] + async fn read_locate_nodes( + &self, + nodes: &mut IdSlice>, + properties: u32, + masking: VisibilityActor, + ) -> Result, Report> { + if nodes.is_empty() { + return Ok(None); + } + let source = nodes[NodeSlot::MIN].identity; + + let connection = self.connection().await?; + let client = connection.as_client(); + + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + let protection = masking.protection(&self.pool.settings.filter_protection); + + let ((), source_properties) = try_join!( + read_types(client, nodes, &temporal_axes), + read_detail( + client, + source, + protection.as_ref(), + properties as usize, + &temporal_axes, + ), + )?; + + Ok(source_properties) + } + + /// Resolves capped link types and actor-masked properties. + /// + /// # Errors + /// + /// Returns [`HydrateError`] if no connection is available or a query fails. + /// + /// # Panics + /// + /// Panics if the store returns a row outside the request domain, a column fails to decode, a + /// direct-type count is negative, or the scalar-property aggregate is not a JSON object. + #[tracing::instrument(skip_all, fields(edges = links.len()))] + async fn read_locate_links( + &self, + links: &mut IdSlice>, + type_ids: u32, + properties: u32, + masking: VisibilityActor, + ) -> Result<(), Report> { + if links.is_empty() { + return Ok(()); + } + + let connection = self.connection().await?; + let client = connection.as_client(); + + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + + let filter = identity_filter(links.iter().map(|link| EntityId::from(link.identity))); + let protection = masking.protection(&self.pool.settings.filter_protection); + let mut compiler = SelectCompiler::new(Some(&temporal_axes), false); + compiler + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + + let columns = DetailColumns::select(&mut compiler, protection.as_ref()); + let (statement, parameters) = compiler.compile(); + + let rows = client + .query_raw(&statement, parameters) + .await + .change_context(HydrateError::Query)?; + + let lookup: FastHashMap<_, _> = links + .iter_enumerated_mut() + .map(|(slot, link)| { + link.details = None; + (link.identity, slot) + }) + .collect(); + + let mut rows = pin!(rows); + while let Some(row) = rows.try_next().await.change_context(HydrateError::Query)? { + let slot = lookup[&columns.entity_id(&row)]; + + let mut type_urls = columns.direct_type_urls(&row); + let type_urls_complete = type_urls.len() <= type_ids as usize; + type_urls.truncate(type_ids as usize); + + let (values, complete) = columns.capped_properties(&row, properties as usize); + links[slot].details = Some(LocateLink { + type_urls, + type_urls_complete, + properties: LocateProperties { values, complete }, + }); + } + + Ok(()) + } + + /// Resolves each requested ontology type uuid to its versioned URL, unordered. + /// + /// # Errors + /// + /// Returns [`HydrateError`] when a store read fails. + /// + /// # Panics + /// + /// Panics if a column fails to decode or a stored URL does not parse as its domain type. + #[tracing::instrument(skip_all, fields(types))] + async fn read_type_urls( + &self, + types: impl IntoIterator, + ) -> Result, Report> + { + let types = types.into_iter(); + tracing::Span::current().record("types", types.len()); + + if types.is_empty() { + return Ok(Vec::new()); + } + + let connection = self.connection().await?; + let client = connection.as_client(); + + let uuids: Vec<_> = types.map(EntityTypeUuid::from).collect(); + let filter = Filter::for_entity_type_uuids(&uuids); + + let mut compiler = SelectCompiler::new(None, false); + compiler + .add_filter(&filter) + .expect("the type-uuid filter compiles against the entity-type query paths"); + + let columns = TypeUrlColumns::select(&mut compiler); + let (statement, parameters) = compiler.compile(); + + let rows = client + .query_raw(&statement, parameters) + .await + .change_context(HydrateError::Query)?; + + let mut pairs = Vec::with_capacity(uuids.len()); + let mut rows = pin!(rows); + while let Some(row) = rows.try_next().await.change_context(HydrateError::Query)? { + pairs.push(columns.pair(&row)); + } + + Ok(pairs) + } +} + +impl TypeUrlResolver for GraphDatabaseClient { + fn resolve( + &self, + types: impl IntoIterator, + ) -> Result, Report> + { + self.runtime.block_on(self.read_type_urls(types)) + } +} + +impl LocateResolver for GraphDatabaseClient { + fn resolve( + &self, + LocateRequest { + actor, + nodes, + links, + properties, + link_type_ids, + link_properties, + }: LocateRequest<'_>, + ) -> Result, Report> { + self.runtime.block_on(async { + let (source_properties, ()) = try_join!( + self.read_locate_nodes(nodes, properties, actor), + self.read_locate_links(links, link_type_ids, link_properties, actor), + )?; + + Ok(source_properties) + }) + } +} + +#[cfg(test)] +mod tests { + use alloc::sync::Arc; + + use hash_graph_postgres_store::store::{ + DatabaseConnectionInfo, DatabasePoolConfig, DatabaseType, PostgresStorePool, + PostgresStoreSettings, + }; + use hash_graph_store::filter::{ + Filter, FilterExpression, Parameter, protection::PropertyProtectionFilterConfig, + }; + use hashql_core::id::IdSlice; + use tokio::runtime::Handle; + use tokio_postgres::NoTls; + use type_system::{ + knowledge::Entity, + principal::actor::{ActorId, UserId}, + }; + use uuid::Uuid; + + use super::{GraphDatabaseClient, VisibilityActor}; + use crate::{ + math::nz, + serve::hydrate::{ + TypeUrlResolver, + locate::{LocateRequest, LocateResolver}, + }, + }; + + /// Builds the masking actor over the user `actor` names. + fn masking(actor: u128, instance_admin: bool) -> VisibilityActor { + VisibilityActor { + id: ActorId::User(UserId::new(Uuid::from_u128(actor))), + instance_admin, + } + } + + /// Returns whether `filter` compares against the parameter `actor` anywhere in its tree. + fn binds_actor(filter: &Filter<'_, Entity>, actor: Uuid) -> bool { + let is_actor = |expression: &FilterExpression<'_, Entity>| { + matches!( + expression, + FilterExpression::Parameter { + parameter: Parameter::Uuid(uuid), + .. + } if *uuid == actor + ) + }; + match filter { + Filter::All(filters) | Filter::Any(filters) => { + filters.iter().any(|filter| binds_actor(filter, actor)) + } + Filter::Not(filter) => binds_actor(filter, actor), + Filter::Equal(lhs, rhs) | Filter::NotEqual(lhs, rhs) => is_actor(lhs) || is_actor(rhs), + Filter::Exists { .. } + | Filter::Greater(..) + | Filter::GreaterOrEqual(..) + | Filter::Less(..) + | Filter::LessOrEqual(..) + | Filter::In(..) + | Filter::StartsWith(..) + | Filter::EndsWith(..) + | Filter::ContainsSegment(..) => false, + } + } + + /// Both resolvers answer an empty request with no results, inside a blocking worker. + /// + /// [`TypeUrlResolver::resolve`] and [`LocateResolver::resolve`] each run from inside a + /// `spawn_blocking` worker without the runtime treating their nested blocking call as a + /// stalled task. + #[tokio::test] + async fn blocking_worker_empty_requests() { + let pool = Arc::new( + PostgresStorePool::new( + &DatabaseConnectionInfo::new( + DatabaseType::Postgres, + "hydrate-test".to_owned(), + String::new(), + "/no-hydrate-test-postgres".to_owned(), + 5432, + "hydrate-test".to_owned(), + ), + &DatabasePoolConfig { + max_connections: nz!(1), + }, + NoTls, + PostgresStoreSettings::default(), + ) + .await + .expect("should construct an unconnected pool"), + ); + let client = GraphDatabaseClient::new(pool, Handle::current()); + + // The async runtime continues polling while its blocking worker drives synchronous reads. + tokio::task::spawn_blocking(move || { + let mut urls = TypeUrlResolver::resolve(&client, []) + .expect("should resolve no URLs without a database connection") + .into_iter(); + assert!(urls.next().is_none(), "should return no unrequested URLs"); + let response = LocateResolver::resolve( + &client, + LocateRequest { + actor: masking(11, false), + nodes: IdSlice::from_raw_mut(&mut []), + links: IdSlice::from_raw_mut(&mut []), + properties: 10, + link_type_ids: 5, + link_properties: 10, + }, + ) + .expect("should resolve an empty locate request without a connection"); + assert_eq!(response, None); + }) + .await + .expect("the blocking worker should finish without a nested-runtime panic"); + } + + #[test] + fn protection_empty_config() { + let config = PropertyProtectionFilterConfig::new(); + + assert!(!masking(11, false).masked_by(&config)); + assert!(masking(11, false).protection(&config).is_none()); + } + + /// An instance admin reads unmasked under a protecting deployment. + #[test] + fn protection_instance_admin() { + let config = PropertyProtectionFilterConfig::hash_default(); + + assert!(!masking(11, true).masked_by(&config)); + assert!(masking(11, true).protection(&config).is_none()); + } + + /// Property protection binds self-access to the reading actor. + #[test] + fn protection_plain_actor() { + let config = PropertyProtectionFilterConfig::hash_default(); + + assert!(masking(11, false).masked_by(&config)); + let protection = masking(11, false) + .protection(&config) + .expect("a protecting deployment masks a plain actor"); + assert!(!protection.is_empty(), "the protection holds no rule"); + for (_property, filter) in protection.iter() { + assert!( + binds_actor(filter, Uuid::from_u128(11)), + "the rule does not compare against the reading actor: {filter:?}" + ); + assert!( + !binds_actor(filter, Uuid::from_u128(12)), + "the rule compares against another actor: {filter:?}" + ); + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/columns.rs b/libs/@local/graph/atlas/src/serve/hydrate/columns.rs new file mode 100644 index 00000000000..187694a5546 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/columns.rs @@ -0,0 +1,23 @@ +//! Slot ID domains for one hydration read. +//! +//! [`NodeSlot`] and [`EdgeSlot`] index a response's delivered nodes and edges in delivered order, +//! so hydration, assembly and encoding all address the same point or link by the same number +//! without re-deriving it from an entity id. + +hashql_core::id::newtype! { + /// A reference to a delivered node by its slot in one response's delivered order. + /// + /// Slots are dense and zero-based over one response's delivered nodes. Every node detail + /// column aligns to this domain. A slot is valid only against the response that delivered it, + /// because two responses share no slot vocabulary. + pub(crate) struct NodeSlot(u32) +} + +hashql_core::id::newtype! { + /// A reference to a delivered edge by its slot in one response's edge order. + /// + /// Slots are dense and zero-based over one response's delivered edges. Every link detail + /// column aligns to this domain. A slot is valid only against the response that delivered it, + /// because two responses share no slot vocabulary. + pub(crate) struct EdgeSlot(u32) +} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/locate.rs b/libs/@local/graph/atlas/src/serve/hydrate/locate.rs new file mode 100644 index 00000000000..7c218ccd7bc --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/locate.rs @@ -0,0 +1,123 @@ +//! Live properties and type URLs for locate documents. +//! +//! Each requested identity retains its slot when the store no longer serves it. + +use alloc::sync::Arc; + +use error_stack::Report; +use hashql_core::id::{IdSlice, IdVec}; +use type_system::ontology::VersionedUrl; + +use super::{EdgeSlot, NodeSlot, client::HydrateError, scalar::ScalarProperties}; +use crate::{postgres::id::ArchivedEntityId, serve::visibility::VisibilityActor}; + +/// One requested entity and the details the store still serves for it. +#[derive(Debug, PartialEq)] +pub(crate) struct LocateEntity { + pub identity: ArchivedEntityId, + pub details: Option, +} + +impl LocateEntity { + /// Builds an entity requested by `identity`, with no details resolved yet. + pub(crate) const fn new(identity: ArchivedEntityId) -> Self { + Self { + identity, + details: None, + } + } +} + +/// The live detail for one delivered node: its direct-type versioned URLs. +#[derive(Debug, PartialEq)] +pub(crate) struct LocateNode { + /// Direct-type versioned URLs in canonical order. + pub type_urls: Vec, +} + +/// One entity's resolved scalar properties and whether they are its whole deliverable set. +#[derive(Debug, PartialEq)] +pub(crate) struct LocateProperties { + pub values: ScalarProperties, + /// Whether the map contains the entity's whole deliverable property set. + pub complete: bool, +} + +/// The live detail for one delivered link: its capped direct types and scalar properties. +#[derive(Debug, PartialEq)] +pub(crate) struct LocateLink { + /// Capped direct-type versioned URLs in canonical order. + pub type_urls: Vec, + /// Whether the list contains the link's whole direct-type set. + pub type_urls_complete: bool, + pub properties: LocateProperties, +} + +/// One locate request's live-detail slots. +/// +/// The slots arrive with the caps and the actor a [`LocateResolver`] resolves them under. +#[derive(Debug)] +pub(crate) struct LocateRequest<'doc> { + /// The resolved actor the store masks properties for. + pub actor: VisibilityActor, + /// Delivered nodes, source first. + pub nodes: &'doc mut IdSlice>, + /// Delivered link entities, ascending identity bytes. + pub links: &'doc mut IdSlice>, + /// Most properties the source's map delivers. + pub properties: u32, + /// Most direct-type URLs each link delivers. + pub link_type_ids: u32, + /// Most properties each link's map delivers. + pub link_properties: u32, +} + +/// The resolved live detail for one locate document. +/// +/// It holds every requested node and link, plus the source's own properties. +#[derive(Debug, PartialEq)] +pub(crate) struct LocateResponse { + pub nodes: IdVec>, + pub links: IdVec>, + pub source_properties: Option, +} + +/// Live detail resolution within the supplied entity slots. +pub(crate) trait LocateResolver { + /// Fills the requested slots and returns the source's capped properties. + /// + /// Unavailable entities remain unresolved. + /// + /// # Errors + /// + /// Returns [`HydrateError`] when a store read fails. Errors may leave partially populated + /// slots. Discard those results. Reset all slot details to [`None`] before retrying. + fn resolve( + &self, + request: LocateRequest<'_>, + ) -> Result, Report>; +} + +impl LocateResolver for &T +where + T: LocateResolver, +{ + fn resolve( + &self, + request: LocateRequest<'_>, + ) -> Result, Report> { + T::resolve(self, request) + } +} + +impl LocateResolver for Arc +where + T: LocateResolver, +{ + fn resolve( + &self, + request: LocateRequest<'_>, + ) -> Result, Report> { + T::resolve(self, request) + } +} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/mod.rs b/libs/@local/graph/atlas/src/serve/hydrate/mod.rs new file mode 100644 index 00000000000..cdcb4cf6743 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/mod.rs @@ -0,0 +1,43 @@ +//! Request-time store reads for the detail a response's trailer carries. +//! +//! Hydration is graph-store enrichment performed while the request runs: an entity's direct-type +//! URLs and its scalar property values, read from the live store. The edges and locate trailers +//! hydrate. A tile trailer does not, because its labels and icons come from the publication the +//! request captured, as positions do. +//! +//! Hydration is one of the two reasons a detailed response is not reusable as an immutable +//! generation tile, and the two differ in the data lifetime a value follows. A captured label +//! follows the epoch the caller's scope resolved against, and moves when that scope re-resolves - +//! which is why even a tile trailer, reading no store, is not stable across requests. A +//! request-time read follows the read instead, and can observe an edition later than the one the +//! captured scope holds. +//! +//! [`LocateResolver`] fills one locate document's node and link slots and returns the source's +//! capped properties, reading both against the live temporal axes. [`TypeUrlResolver`] resolves +//! ontology type uuids to versioned URLs for the edges trailer, a lookup with no temporal axes +//! and no entity edition in it. A uuid derives from the versioned URL it names, which holds the +//! pair steady once resolved. [`CachedTypeUrlResolver`] answers a repeat from its retained result. +//! [`GraphDatabaseClient`] answers both resolvers against the serving store pool. [`NodeSlot`] +//! and [`EdgeSlot`] are the slot domains a resolver fills in place, and [`scalar`] is the value +//! shape a property read can take. +//! +//! The module also holds [`visibility::visibility_proof`], which resolves the rows an actor may +//! receive before any document gathers from the captured scene. + +mod client; +mod columns; +mod locate; +pub(crate) mod scalar; +mod statements; +mod type_urls; +pub(crate) mod visibility; + +// Locate document fixtures construct typed resolver answers. +#[cfg(test)] +pub(crate) use self::locate::{LocateLink, LocateNode, LocateProperties}; +pub(crate) use self::{ + client::{GraphDatabaseClient, HydrateError}, + columns::{EdgeSlot, NodeSlot}, + locate::{LocateEntity, LocateRequest, LocateResolver, LocateResponse}, + type_urls::{CachedTypeUrlResolver, TypeUrlResolver}, +}; diff --git a/libs/@local/graph/atlas/src/serve/hydrate/scalar.rs b/libs/@local/graph/atlas/src/serve/hydrate/scalar.rs new file mode 100644 index 00000000000..b45b5223183 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/scalar.rs @@ -0,0 +1,228 @@ +use alloc::vec; + +use type_system::ontology::id::BaseUrl; + +/// One scalar property value. +/// +/// A hydrated property value takes no other shape. The store filters out nested objects and +/// arrays. They never cross the connection. +#[derive(Debug, Clone, PartialEq)] +pub(crate) enum ScalarValue { + /// A text scalar. + String(String), + /// A number the store renders integral, within `i64`. + Integer(i64), + /// Any other number. + /// + /// Store scalars are doubles on the wire. + Float(f64), + /// A boolean scalar. + Bool(bool), + /// An explicit null the entity carries. + Null, +} + +/// One entity's scalar properties, ascending by base URL. +#[derive(Debug, Clone, PartialEq)] +pub(crate) struct ScalarProperties(Vec<(BaseUrl, ScalarValue)>); + +impl ScalarProperties { + /// An empty property list, usable in a `const` context. + pub(crate) const EMPTY: Self = Self::empty(); + + /// Builds an empty property list. + pub(crate) const fn empty() -> Self { + Self(vec![]) + } + + /// Returns the number of scalar entries. + pub(crate) const fn len(&self) -> usize { + self.0.len() + } + + /// Converts a property object to at most `maximum` scalar entries. + /// + /// Returns the entries and whether the cap truncated them. Conversion skips invalid property + /// URLs, nested values and unrepresentable numbers before applying the cap. A nonzero cap + /// preserves `label_property` if it has a scalar entry, selecting the remaining entries in key + /// order. + /// + /// # Panics + /// + /// Panics if `value` is not a JSON object. + pub(crate) fn new( + value: serde_json::Value, + label_property: Option<&BaseUrl>, + maximum: usize, + ) -> (Self, bool) { + let serde_json::Value::Object(mut object) = value else { + panic!("the store aggregates a JSON object") + }; + object.sort_keys(); + + let mut entries: Vec<_> = object.into_iter().filter_map(|(name, value)| { + let name = match BaseUrl::new(name) { + Ok(name) => name, + Err(error) => { + tracing::warn!( + %error, + "expected the key to be well-formed according to semtype, and be a base URL" + ); + + return None; + } + }; + + let value = match value { + serde_json::Value::String(string) => ScalarValue::String(string), + serde_json::Value::Number(number) => { + if let Some(integer) = number.as_i64() { + ScalarValue::Integer(integer) + } else if let Some(float) = number.as_f64() { + ScalarValue::Float(float) + } else { + tracing::warn!("query returned too large a number"); + + return None; + } + } + serde_json::Value::Bool(bool) => ScalarValue::Bool(bool), + serde_json::Value::Null => ScalarValue::Null, + serde_json::Value::Object(_) | serde_json::Value::Array(_) => { + tracing::warn!( + "query should have returned only scalar values, but included a JSON object or \ + array" + ); + + return None; + } + }; + + Some((name, value)) + }).collect(); + + if entries.len() <= maximum { + return (Self(entries), false); + } + + // `object.sort_keys` ensures the entries are in ascending order. + if maximum > 0 { + // swap the label property into the last retained slot, dropping whichever entry + // would otherwise have held it + if let Some(offset) = label_property.and_then(|label| { + entries[maximum..] + .iter() + .position(|(name, _)| name == label) + }) { + entries.swap(maximum - 1, maximum + offset); + } + } + + entries.truncate(maximum); + (Self(entries), true) + } +} + +impl IntoIterator for ScalarProperties { + type IntoIter = vec::IntoIter; + type Item = (BaseUrl, ScalarValue); + + fn into_iter(self) -> Self::IntoIter { + self.0.into_iter() + } +} + +#[cfg(test)] +mod tests { + use alloc::{collections::BTreeMap, sync::Arc}; + use core::fmt; + use std::sync::Mutex; + + use tracing::{ + Event, Level, Subscriber, + field::{Field, Visit}, + }; + use tracing_subscriber::{ + Layer, Registry, + layer::{Context, SubscriberExt as _}, + }; + use type_system::ontology::id::BaseUrl; + + use super::{ScalarProperties, ScalarValue}; + + /// One captured tracing event's fields, by name. + /// + /// The map holds each value's [`Debug`](core::fmt::Debug) rendering. + #[derive(Default)] + struct Fields(BTreeMap); + + impl Visit for Fields { + fn record_debug(&mut self, field: &Field, value: &dyn fmt::Debug) { + self.0.insert(field.name().to_owned(), format!("{value:?}")); + } + } + + /// A subscriber layer that records every `WARN`-level event's fields. + /// + /// A test reads the recorded fields once the traced code under test has run. + struct Warnings(Arc>>); + + impl Layer for Warnings { + /// Asserts the event's level and appends its captured [`Fields`] to the warning list. + /// + /// `WARN` is the only level this fixture expects [`ScalarProperties::new`] to emit. + /// + /// # Panics + /// + /// Panics for non-`WARN` events or a poisoned warning-list mutex. + fn on_event(&self, event: &Event<'_>, _: Context<'_, S>) { + assert_eq!(*event.metadata().level(), Level::WARN); + let mut fields = Fields::default(); + event.record(&mut fields); + self.0 + .lock() + .expect("should lock the warning list") + .push(fields); + } + } + + /// Skips nested values with warnings that omit their contents and property names. + #[test] + fn values_nested() { + let warnings = Arc::new(Mutex::new(Vec::new())); + let subscriber = Registry::default().with(Warnings(Arc::clone(&warnings))); + let input = serde_json::json!({ + "https://example.com/private-object-key/": {"private-field": "private-object-value"}, + "https://example.com/private-array-key/": ["private-array-value"], + "https://example.com/scalar/": "retained-scalar-value", + }); + let (properties, truncated) = tracing::subscriber::with_default(subscriber, || { + ScalarProperties::new(input, None, 10) + }); + + assert!(!truncated); + assert_eq!( + properties.into_iter().collect::>(), + [( + BaseUrl::new("https://example.com/scalar/".to_owned()) + .expect("should parse the scalar key"), + ScalarValue::String("retained-scalar-value".to_owned()) + )] + ); + + let warnings = warnings.lock().expect("should lock the warning list"); + assert_eq!(warnings.len(), 2, "should warn for both nested values"); + for fields in warnings.iter() { + assert_eq!( + fields.0, + BTreeMap::from([( + "message".to_owned(), + "query should have returned only scalar values, but included a JSON object or \ + array" + .to_owned() + )]), + "should record only the fixed diagnostic, without a property key or value" + ); + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap new file mode 100644 index 00000000000..2b20c12f24e --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap @@ -0,0 +1,16 @@ +--- +source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs +expression: detail.compile().0 +--- +SELECT "entity_temporal_metadata_0_0_0"."web_id", "entity_temporal_metadata_0_0_0"."entity_uuid", "entity_edition_cache_1_1_0"."versioned_urls", "entity_edition_cache_1_1_0"."direct_types", (SELECT jsonb_object_agg("scalar_property"."key", "scalar_property"."value") +FROM jsonb_each("entity_editions_1_1_0"."properties") AS "scalar_property"("key", "value") +WHERE jsonb_typeof("scalar_property"."value") = ANY(($6::text[]))), ((SELECT count(*) +FROM jsonb_each("entity_editions_1_1_0"."properties") AS "scalar_property"("key", "value"))::int4), ("entity_edition_cache_1_1_0"."label_properties")[1] +FROM "entity_temporal_metadata" AS "entity_temporal_metadata_0_0_0" +INNER JOIN "entity_editions" AS "entity_editions_0_1_0" + ON "entity_editions_0_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +INNER JOIN "entity_edition_cache" AS "entity_edition_cache_1_1_0" + ON "entity_edition_cache_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +INNER JOIN "entity_editions" AS "entity_editions_1_1_0" + ON "entity_editions_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +WHERE ("entity_temporal_metadata_0_0_0"."draft_id" IS NULL) AND ("entity_temporal_metadata_0_0_0"."transaction_time" @> $1::TIMESTAMPTZ) AND ("entity_temporal_metadata_0_0_0"."decision_time" && $2) AND (("entity_temporal_metadata_0_0_0"."web_id" = $3) AND ("entity_temporal_metadata_0_0_0"."entity_uuid" = $4) AND ("entity_editions_0_1_0"."archived" = $5)) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap new file mode 100644 index 00000000000..7cc15cec2bc --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap @@ -0,0 +1,16 @@ +--- +source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs +expression: detail.compile().0 +--- +SELECT "entity_temporal_metadata_0_0_0"."web_id", "entity_temporal_metadata_0_0_0"."entity_uuid", "entity_edition_cache_1_1_0"."versioned_urls", "entity_edition_cache_1_1_0"."direct_types", (SELECT jsonb_object_agg("scalar_property"."key", "scalar_property"."value") +FROM jsonb_each(("entity_editions_1_1_0"."properties" - (CASE WHEN ("entity_temporal_metadata_0_0_0"."entity_uuid" != $7) AND ("entity_edition_cache_1_1_0"."base_urls" @> ARRAY[$8]::text[]) THEN ARRAY[$9]::text[] ELSE ARRAY[]::text[] END))) AS "scalar_property"("key", "value") +WHERE jsonb_typeof("scalar_property"."value") = ANY(($6::text[]))), ((SELECT count(*) +FROM jsonb_each(("entity_editions_1_1_0"."properties" - (CASE WHEN ("entity_temporal_metadata_0_0_0"."entity_uuid" != $7) AND ("entity_edition_cache_1_1_0"."base_urls" @> ARRAY[$8]::text[]) THEN ARRAY[$9]::text[] ELSE ARRAY[]::text[] END))) AS "scalar_property"("key", "value"))::int4), ("entity_edition_cache_1_1_0"."label_properties")[1] +FROM "entity_temporal_metadata" AS "entity_temporal_metadata_0_0_0" +INNER JOIN "entity_editions" AS "entity_editions_0_1_0" + ON "entity_editions_0_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +INNER JOIN "entity_edition_cache" AS "entity_edition_cache_1_1_0" + ON "entity_edition_cache_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +INNER JOIN "entity_editions" AS "entity_editions_1_1_0" + ON "entity_editions_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +WHERE ("entity_temporal_metadata_0_0_0"."draft_id" IS NULL) AND ("entity_temporal_metadata_0_0_0"."transaction_time" @> $1::TIMESTAMPTZ) AND ("entity_temporal_metadata_0_0_0"."decision_time" && $2) AND (("entity_temporal_metadata_0_0_0"."web_id" = $3) AND ("entity_temporal_metadata_0_0_0"."entity_uuid" = $4) AND ("entity_editions_0_1_0"."archived" = $5)) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap new file mode 100644 index 00000000000..3e480590dd6 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap @@ -0,0 +1,11 @@ +--- +source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs +expression: type_urls.compile().0 +--- +SELECT "entity_types_1_1_0"."ontology_id", "entity_types_1_1_0"."schema"->>'$id' +FROM "ontology_temporal_metadata" AS "ontology_temporal_metadata_0_0_0" +INNER JOIN "entity_types" AS "entity_types_0_1_0" + ON "entity_types_0_1_0"."ontology_id" = "ontology_temporal_metadata_0_0_0"."ontology_id" +INNER JOIN "entity_types" AS "entity_types_1_1_0" + ON "entity_types_1_1_0"."ontology_id" = "ontology_temporal_metadata_0_0_0"."ontology_id" +WHERE "entity_types_0_1_0"."ontology_id" = ANY($1) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap new file mode 100644 index 00000000000..8b43c88cc13 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap @@ -0,0 +1,11 @@ +--- +source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs +expression: types.compile().0 +--- +SELECT "entity_temporal_metadata_0_0_0"."web_id", "entity_temporal_metadata_0_0_0"."entity_uuid", "entity_edition_cache_1_1_0"."versioned_urls", "entity_edition_cache_1_1_0"."direct_types" +FROM "entity_temporal_metadata" AS "entity_temporal_metadata_0_0_0" +INNER JOIN "entity_editions" AS "entity_editions_0_1_0" + ON "entity_editions_0_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +INNER JOIN "entity_edition_cache" AS "entity_edition_cache_1_1_0" + ON "entity_edition_cache_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" +WHERE ("entity_temporal_metadata_0_0_0"."draft_id" IS NULL) AND ("entity_temporal_metadata_0_0_0"."transaction_time" @> $1::TIMESTAMPTZ) AND ("entity_temporal_metadata_0_0_0"."decision_time" && $2) AND (("entity_temporal_metadata_0_0_0"."web_id" = $3) AND ("entity_temporal_metadata_0_0_0"."entity_uuid" = $4) AND ("entity_editions_0_1_0"."archived" = $5)) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/statements.rs b/libs/@local/graph/atlas/src/serve/hydrate/statements.rs new file mode 100644 index 00000000000..c429d4ea247 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/statements.rs @@ -0,0 +1,380 @@ +use hash_graph_postgres_store::store::postgres::query::SelectCompiler; +use hash_graph_store::{ + entity::EntityQueryPath, + entity_type::EntityTypeQueryPath, + filter::{Filter, FilterExpression, Parameter, protection::PropertyProtectionFilter}, + subgraph::edges::SharedEdgeKind, +}; +use type_system::{ + knowledge::{ + Entity, + entity::id::{EntityId, EntityUuid}, + }, + ontology::{ + entity_type::EntityTypeWithMetadata, + id::{BaseUrl, OntologyTypeUuid, VersionedUrl}, + }, + principal::actor_group::WebId, +}; + +use super::scalar::ScalarProperties; +use crate::postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId}; + +/// Filters to the non-archived entities named by `ids`. +pub(super) fn identity_filter<'params>( + ids: impl IntoIterator, +) -> Filter<'params, Entity> { + Filter::All(vec![ + Filter::Any( + ids.into_iter() + .map(Filter::for_entity_by_entity_id) + .collect(), + ), + Filter::Equal( + FilterExpression::Path { + path: EntityQueryPath::Archived, + }, + FilterExpression::Parameter { + parameter: Parameter::Boolean(false), + convert: None, + }, + ), + ]) +} + +/// The output columns of one type-URL read. +pub(super) struct TypeColumns { + /// The web half of the entity's identity. + web_id: usize, + /// The entity half of the entity's identity. + entity_uuid: usize, + /// The cached versioned-URL array, direct types first. + type_urls: usize, + /// How many leading entries of the array are direct types. + direct_types: usize, +} + +impl TypeColumns { + /// Adds the identity and type-URL selections to `compiler`. + pub(super) fn select(compiler: &mut SelectCompiler<'_, '_, Entity>) -> Self { + Self { + web_id: compiler.add_selection_path(&EntityQueryPath::WebId), + entity_uuid: compiler.add_selection_path(&EntityQueryPath::Uuid), + type_urls: compiler.add_selection_path(&EntityQueryPath::EntityTypeEdge { + edge_kind: SharedEdgeKind::IsOfType, + path: EntityTypeQueryPath::VersionedUrl, + inheritance_depth: None, + }), + direct_types: compiler.add_selection_path(&EntityQueryPath::DirectTypeCount), + } + } + + /// Reads one row's direct-type URLs: the cached array cut to its direct-type prefix. + /// + /// # Panics + /// + /// Panics if a column fails to decode or the direct-type count is negative. + pub(super) fn direct_type_urls(&self, row: &tokio_postgres::Row) -> Vec { + let direct: i32 = row.get(self.direct_types); + let direct = usize::try_from(direct).expect("the store counts direct types non-negatively"); + + let mut urls: Vec = row.get(self.type_urls); + urls.truncate(direct); + urls + } + + /// Reads one row's identity from the identity columns. + /// + /// # Panics + /// + /// This panics when a column does not decode at its assigned position. + pub(super) fn entity_id(&self, row: &tokio_postgres::Row) -> ArchivedEntityId { + let web_id: WebId = row.get(self.web_id); + let entity_uuid: EntityUuid = row.get(self.entity_uuid); + + ArchivedEntityId { + web_id: ArchivedWebId::from(web_id), + entity_uuid: ArchivedEntityUuid::from(entity_uuid), + } + } +} + +/// The output columns of one type-URL resolution read. +pub(super) struct TypeUrlColumns { + /// The type's URL-derived ontology uuid. + ontology_id: usize, + /// The type's versioned URL, the schema's own `$id`. + versioned_url: usize, +} + +impl TypeUrlColumns { + /// Adds the uuid and versioned-URL selections to `compiler`. + pub(super) fn select(compiler: &mut SelectCompiler<'_, '_, EntityTypeWithMetadata>) -> Self { + Self { + ontology_id: compiler.add_selection_path(&EntityTypeQueryPath::OntologyId), + versioned_url: compiler.add_selection_path(&EntityTypeQueryPath::VersionedUrl), + } + } + + /// Reads one row's uuid-URL pair. + /// + /// # Panics + /// + /// This panics when a column does not decode at its assigned position or when a stored URL + /// does not parse as its domain type. + pub(super) fn pair(&self, row: &tokio_postgres::Row) -> (OntologyTypeUuid, VersionedUrl) { + let uuid: OntologyTypeUuid = row.get(self.ontology_id); + let url: VersionedUrl = row.get(self.versioned_url); + + (uuid, url) + } +} + +/// The output columns of one detail read. +pub(super) struct DetailColumns { + /// The identity and type-URL positions. + types: TypeColumns, + /// The scalar property map position. + scalars: usize, + /// The whole property-count position. + total: usize, + /// The label-attribution position. + label: usize, +} + +impl DetailColumns { + /// Configures `masking` and adds the detail selections to `compiler`. + /// + /// The masking configures first. Every property selection then compiles against the masked + /// column. + pub(super) fn select<'params, 'query: 'params>( + compiler: &mut SelectCompiler<'params, 'query, Entity>, + masking: Option<&'params PropertyProtectionFilter<'params, 'query>>, + ) -> Self { + if let Some(protection) = masking { + compiler.with_property_masking(protection); + } + + let types = TypeColumns::select(compiler); + let scalars = compiler.add_selection_path(&EntityQueryPath::ScalarProperties); + let total = compiler.add_selection_path(&EntityQueryPath::PropertyCount); + let label = compiler.add_selection_path(&EntityQueryPath::FirstLabelProperty); + + Self { + types, + scalars, + total, + label, + } + } + + /// Reads one row's identity from the identity columns. + /// + /// # Panics + /// + /// This panics when a column does not decode at its assigned position. + pub(super) fn entity_id(&self, row: &tokio_postgres::Row) -> ArchivedEntityId { + self.types.entity_id(row) + } + + /// Reads one row's direct-type URLs: the cached array cut to its direct-type prefix. + /// + /// # Panics + /// + /// Panics if a column fails to decode or the direct-type count is negative. + pub(super) fn direct_type_urls(&self, row: &tokio_postgres::Row) -> Vec { + self.types.direct_type_urls(row) + } + + /// Reads one row's capped properties and their completeness flag. + /// + /// Both property columns read the same masked object, so completeness attests the + /// deliverable set: the survivors are that whole set exactly when the scalar-type filter + /// dropped nothing and the cap holds everything. A property the masking withholds is in + /// neither column and moves the flag not at all. The label property drops last under the + /// cap. + /// + /// # Panics + /// + /// Panics if a column fails to decode or the scalar-property aggregate is not a JSON object. + pub(super) fn capped_properties( + &self, + row: &tokio_postgres::Row, + maximum: usize, + ) -> (ScalarProperties, bool) { + let scalars: Option = row.get(self.scalars); + let total: i32 = row.get(self.total); + let label: Option = row.get(self.label); + + let (properties, truncated) = scalars + .map_or((ScalarProperties::EMPTY, false), |properties| { + ScalarProperties::new(properties, label.as_ref(), maximum) + }); + + let complete = !truncated && properties.len() == usize::try_from(total).unwrap_or(0); + (properties, complete) + } +} + +#[cfg(test)] +mod tests { + use hash_graph_postgres_store::store::postgres::query::SelectCompiler; + use hash_graph_store::{ + filter::protection::PropertyProtectionFilterConfig, + subgraph::temporal_axes::QueryTemporalAxesUnresolved, + }; + use type_system::{ + knowledge::entity::id::{EntityId, EntityUuid}, + ontology::entity_type::EntityTypeUuid, + principal::{ + actor::{ActorId, UserId}, + actor_group::WebId, + }, + }; + use uuid::Uuid; + + use super::{DetailColumns, Filter, TypeColumns, TypeUrlColumns, identity_filter}; + + /// Builds the reading actor the masked pins bind their self-access clause to. + fn reading_actor() -> ActorId { + ActorId::User(UserId::new(Uuid::from_u128(11))) + } + + /// Builds the identity filter over one nil identity, the fixture request. + fn nil_filter() -> super::Filter<'static, super::Entity> { + identity_filter([EntityId { + web_id: WebId::new(Uuid::nil()), + entity_uuid: EntityUuid::new(Uuid::nil()), + draft_id: None, + }]) + } + + /// The detail read masks its property columns exactly when the caller passes a masking. + /// + /// The masked spelling is the subtraction inside `jsonb_each(`, which is the compiler's + /// column hook firing inside each property subquery. The count is over the masked object + /// too, because a whole-object count against a masked map would tell an actor how many + /// properties the masking withheld, the enumeration signal the protection exists to close. + #[test] + fn detail_masked_both_subqueries() { + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + let filter = nil_filter(); + + let config = PropertyProtectionFilterConfig::hash_default(); + let protection = config.to_property_protection_filter(Some(reading_actor())); + + let mut masked = SelectCompiler::new(Some(&temporal_axes), false); + masked + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + DetailColumns::select(&mut masked, Some(&protection)); + let (masked_sql, _) = masked.compile(); + assert_eq!( + masked_sql + .matches(r#"jsonb_each(("entity_editions_1_1_0"."properties" - (CASE"#) + .count(), + 2, + "the protected detail read does not mask both property subqueries: {masked_sql}" + ); + + let mut bare = SelectCompiler::new(Some(&temporal_axes), false); + bare.add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + DetailColumns::select(&mut bare, None); + let (bare_sql, _) = bare.compile(); + assert_eq!( + bare_sql + .matches(r#"jsonb_each("entity_editions_1_1_0"."properties")"#) + .count(), + 2, + "the unprotected detail read does not read the bare object: {bare_sql}" + ); + } + + /// The rendered type-URL read, pinned as the text the store receives. + /// + /// The snapshot detects rendering changes to the query selections or compiler output. + /// Each compiled pin holds the one-identity request, which is the + /// shape the masking assertions read. A request naming more identities compiles its + /// membership as a row comparison over unnested arrays. The pinned grammar belongs to + /// the one-identity request alone. + #[test] + fn types_statement_text() { + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + let filter = nil_filter(); + + let mut types = SelectCompiler::new(Some(&temporal_axes), false); + types + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + TypeColumns::select(&mut types); + + insta::assert_snapshot!(types.compile().0); + } + + /// The rendered masked detail read, the form every hydration compiles. + /// + /// The pin uses the deployment's default protection for a resolved actor. + /// + /// The CASE conditions grow per protected property without changing the pinned grammar. Both + /// property subqueries read the masked object, and the self-access clause compares the entity + /// against the reading actor's parameter. A pin without an actor masks unconditionally and + /// cannot see that clause regress. + #[test] + fn masked_detail_statement_text() { + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + let filter = nil_filter(); + + let config = PropertyProtectionFilterConfig::hash_default(); + let protection = config.to_property_protection_filter(Some(reading_actor())); + let mut detail = SelectCompiler::new(Some(&temporal_axes), false); + detail + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + DetailColumns::select(&mut detail, Some(&protection)); + + insta::assert_snapshot!(detail.compile().0); + } + + /// The rendered bare detail read, pinned without any masking configured. + /// + /// Both property subqueries read the bare object, and the SQL contains no `CASE` subtraction. + #[test] + fn bare_detail_statement_text() { + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + let filter = nil_filter(); + + let mut detail = SelectCompiler::new(Some(&temporal_axes), false); + detail + .add_filter(&filter) + .expect("the identity filter compiles against the entity query paths"); + DetailColumns::select(&mut detail, None); + + insta::assert_snapshot!(detail.compile().0); + } + + /// The rendered type-URL resolution read, pinned as the text the store receives. + /// + /// The membership array binds as one parameter. This text is the rendering at every + /// batch width. The read carries no temporal condition on purpose. A type uuid derives + /// from the URL it names. Any row that exists answers correctly whatever its archival + /// state, and the pin makes an upstream compiler change that reintroduced a temporal + /// predicate a visible snapshot diff. + #[test] + fn type_urls_statement_text() { + let uuids = [EntityTypeUuid::from_url( + &"https://example.com/types/entity-type/fixture/v/1" + .parse() + .expect("the fixture URL parses"), + )]; + let filter = Filter::for_entity_type_uuids(&uuids); + + let mut type_urls = SelectCompiler::new(None, false); + type_urls + .add_filter(&filter) + .expect("the type-uuid filter compiles against the entity-type query paths"); + TypeUrlColumns::select(&mut type_urls); + + insta::assert_snapshot!(type_urls.compile().0); + } +} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs b/libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs new file mode 100644 index 00000000000..a924b5acd45 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs @@ -0,0 +1,257 @@ +use alloc::sync::Arc; +use std::sync::nonpoison::RwLock; + +use error_stack::Report; +use hashql_core::collections::FastHashMap; +use type_system::ontology::{VersionedUrl, id::OntologyTypeUuid}; + +use super::client::HydrateError; + +/// The capability to resolve ontology type UUIDs to their versioned URLs. +pub(crate) trait TypeUrlResolver { + /// Resolves each requested ontology type uuid to its versioned URL, unordered. + /// + /// # Errors + /// + /// Returns [`HydrateError`] when a store read fails. + fn resolve( + &self, + types: impl IntoIterator, + ) -> Result, Report>; +} + +impl TypeUrlResolver for &T +where + T: TypeUrlResolver, +{ + fn resolve( + &self, + types: impl IntoIterator, + ) -> Result, Report> + { + T::resolve(self, types) + } +} + +impl TypeUrlResolver for Arc +where + T: TypeUrlResolver, +{ + fn resolve( + &self, + types: impl IntoIterator, + ) -> Result, Report> + { + T::resolve(self, types) + } +} + +/// A resolution source with a persistent cache of successful lookups. +/// +/// Unresolved UUIDs remain uncached. A recreated type has the same UUID and can resolve on a later +/// call. +pub(crate) struct CachedTypeUrlResolver { + inner: T, + known: RwLock>, +} + +impl CachedTypeUrlResolver { + /// Wraps `inner` with an empty cache. + pub(crate) fn new(inner: T) -> Self { + Self { + inner, + known: RwLock::new(FastHashMap::default()), + } + } +} + +impl TypeUrlResolver for CachedTypeUrlResolver +where + T: TypeUrlResolver, +{ + /// Answers from the cache first and asks `inner` only for the misses. + /// + /// A read lock covers the cache lookup. The misses' answers enter the cache before the + /// combined result returns. An unresolved UUID never enters the cache, and a later call + /// retries it against `inner`. + /// + /// # Panics + /// + /// Panics where `inner` does. A UUID the cache already holds never reaches it. + fn resolve( + &self, + types: impl IntoIterator, + ) -> Result, Report> + { + let types = types.into_iter(); + let mut found = Vec::with_capacity(types.len()); + let mut misses = Vec::new(); + + { + let known = self.known.read(); + for uuid in types { + match known.get(&uuid) { + Some(url) => found.push((uuid, url.clone())), + None => misses.push(uuid), + } + } + } + + if !misses.is_empty() { + let fresh = self.inner.resolve(misses)?; + let mut known = self.known.write(); + for (uuid, url) in fresh { + known.insert(uuid, url.clone()); + found.push((uuid, url)); + } + drop(known); + } + + Ok(found) + } +} + +#[cfg(test)] +mod tests { + use alloc::sync::Arc; + use std::sync::nonpoison::Mutex; + + use error_stack::Report; + use hashql_core::collections::FastHashMap; + use type_system::ontology::{VersionedUrl, id::OntologyTypeUuid}; + + use super::{CachedTypeUrlResolver, HydrateError, TypeUrlResolver}; + + /// A [`TypeUrlResolver`] fixture answering a fixed set of UUIDs. + /// + /// It records every requested UUID batch. A test asserts which UUIDs the caller requested. + struct Ledger { + urls: FastHashMap, + asked: Mutex>>, + } + + impl Ledger { + /// Builds a deterministic fixture URL for `ordinal`. + /// + /// It is stable across calls. A test computes the same UUID [`Self::new`] derived from it. + fn type_url(ordinal: u64) -> VersionedUrl { + format!("https://example.com/types/entity-type/fixture-{ordinal}/v/1") + .parse() + .expect("should parse the fixture URL") + } + + /// Builds a [`Ledger`] resolving exactly the UUIDs derived from `ordinals`. + /// + /// The call returns those UUIDs beside the ledger, in [`FastHashMap`]'s iteration order. + fn new(ordinals: impl IntoIterator) -> (Self, Vec) { + let urls: FastHashMap<_, _> = ordinals + .into_iter() + .map(|ordinal| { + let url = Self::type_url(ordinal); + (OntologyTypeUuid::from_url(&url), url) + }) + .collect(); + let uuids = urls.keys().copied().collect(); + ( + Self { + urls, + asked: Mutex::new(Vec::new()), + }, + uuids, + ) + } + } + + impl TypeUrlResolver for Ledger { + fn resolve( + &self, + types: impl IntoIterator, + ) -> Result, Report> + { + let types: Vec<_> = types.into_iter().collect(); + let found: Vec<_> = types + .iter() + .filter_map(|uuid| self.urls.get(uuid).map(|url| (*uuid, url.clone()))) + .collect(); + self.asked.lock().push(types); + Ok(found) + } + } + + #[test] + fn cache_hit() { + let (ledger, uuids) = Ledger::new([0]); + let cached = CachedTypeUrlResolver::new(ledger); + let first: Vec<_> = cached + .resolve(uuids.iter().copied()) + .expect("should resolve the known type") + .into_iter() + .collect(); + let second: Vec<_> = cached + .resolve(uuids.iter().copied()) + .expect("should resolve the cached type") + .into_iter() + .collect(); + + assert_eq!(first.len(), 1); + assert_eq!(first, second); + assert_eq!(*cached.inner.asked.lock(), vec![uuids]); + } + + /// An unresolvable UUID asks the inner resolver again on every call. + /// + /// A miss is never cached. + #[test] + fn unresolved_retried() { + let (ledger, _) = Ledger::new([]); + let unknown = OntologyTypeUuid::from_url(&Ledger::type_url(7)); + let cached = CachedTypeUrlResolver::new(ledger); + for _ in 0..2 { + assert_eq!( + cached + .resolve([unknown]) + .expect("should accept an unresolved type") + .into_iter() + .count(), + 0 + ); + } + assert_eq!( + *cached.inner.asked.lock(), + vec![vec![unknown], vec![unknown]] + ); + } + + /// Resolving both UUIDs after warming one asks the inner resolver only for the other. + /// + /// The cache answers the warmed UUID without a second request. + #[test] + fn partial_hit() { + let (ledger, known) = Ledger::new([0, 1]); + let cached = CachedTypeUrlResolver::new(ledger); + let warmed = cached + .resolve(known.iter().copied().take(1)) + .expect("should resolve the first type"); + assert_eq!(warmed.into_iter().count(), 1); + let answer = cached + .resolve(known.iter().copied()) + .expect("should resolve both types"); + + assert_eq!(answer.into_iter().count(), 2); + let asked = cached.inner.asked.lock(); + assert_eq!(asked[1], known[1..]); + } + + /// [`CachedTypeUrlResolver`] works over an [`Arc`]-wrapped inner resolver. + /// + /// The case exercises the blanket `TypeUrlResolver for Arc` impl. + #[test] + fn arc_source() { + let (ledger, uuids) = Ledger::new([3]); + let cached = CachedTypeUrlResolver::new(Arc::new(ledger)); + let answer = cached + .resolve(uuids) + .expect("should resolve the known type"); + assert_eq!(answer.into_iter().count(), 1); + } +} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/visibility.rs b/libs/@local/graph/atlas/src/serve/hydrate/visibility.rs new file mode 100644 index 00000000000..9fab0201e64 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/hydrate/visibility.rs @@ -0,0 +1,313 @@ +use core::{error::Error, fmt, pin::pin}; + +use error_stack::{Report, ResultExt as _}; +use futures::TryStreamExt as _; +use hash_graph_authorization::policies::{ + MergePolicies, PolicyComponents, + action::ActionName, + store::{PolicyStore, PrincipalStore}, +}; +use hash_graph_postgres_store::store::{AsClient, StoreProvider, postgres::query::SelectCompiler}; +use hash_graph_store::{ + entity::EntityQueryPath, + filter::{ + Filter, + protection::{PropertyProtectionFilterConfig, transform_filter}, + }, + subgraph::temporal_axes::QueryTemporalAxesUnresolved, +}; +use hash_graph_types::ontology::DataTypeLookup; +use tokio_postgres::GenericClient as _; +use type_system::{ + knowledge::{Entity, entity::id::EntityUuid}, + principal::{actor::ActorId, actor_group::WebId}, +}; +use uuid::Uuid; + +use crate::{ + bitset::CompressedBitSet, + postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId}, + serve::{ + delta::epoch::Epoch, + visibility::{VisibilityActor, VisibilityMask}, + world::World, + }, +}; + +/// Resolving an actor's visible rows against the store failed. +/// +/// Each variant names one failing stage. A caller can separate a request it can repair from a +/// condition it cannot. +#[derive(Debug)] +pub(crate) enum VisibilityProofError { + /// No store connection was available for the resolution. + Connect, + /// Assembling the actor's policy set failed. + Policies, + /// The caller's filter does not compile against the entity query paths. + Filter, + /// The caller's filter carries a parameter that does not match its path's type. + Convert, + /// The scope's held filter document does not parse. + Document, + /// The policy filter does not compile against the entity query paths. + PolicyFilter, + /// The store rejected the visibility query. + Query, + /// The offloaded schedule-and-census computation produced no value. + ComputeView, +} + +impl fmt::Display for VisibilityProofError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Connect => fmt.write_str("the resolution reached no store connection"), + Self::Policies => fmt.write_str("the actor's policy set could not be assembled"), + Self::Filter => fmt.write_str("the request filter does not compile"), + Self::Convert => { + fmt.write_str("the request filter's parameters do not match its paths") + } + Self::Document => fmt.write_str("the scope's held filter document does not parse"), + Self::PolicyFilter => fmt.write_str("the policy filter does not compile"), + Self::Query => fmt.write_str("the store rejected the visibility query"), + Self::ComputeView => fmt.write_str("the view's schedule and census failed to compute"), + } + } +} + +impl Error for VisibilityProofError {} + +/// Reports whether the combined filter admits every row. +/// +/// That holds for no caller filter and a policy filter that is an empty `All` conjunction. +const fn admits_every_row( + filter: Option<&Filter<'_, Entity>>, + policy_filter: &Filter<'_, Entity>, +) -> bool { + filter.is_none() && matches!(policy_filter, Filter::All(conjuncts) if conjuncts.is_empty()) +} + +/// Resolves `actor`'s visible-row mask against `world`'s epoch. +/// +/// The query carries `filter` and the deployment's property protection. +/// +/// # Errors +/// +/// Returns [`VisibilityProofError`] from the first failing stage of policy assembly, filter +/// conversion or compilation, and visibility querying. +#[tracing::instrument(skip_all)] +pub(crate) async fn visibility_proof( + world: &World, + epoch: &Epoch, + + actor: ActorId, + filter: Option<&Filter<'_, Entity>>, + protection: &PropertyProtectionFilterConfig<'static>, + store: &S, +) -> Result> +where + S: PrincipalStore + PolicyStore + AsClient + Sync, + for<'store> StoreProvider<'store, S>: DataTypeLookup + Sync, +{ + let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); + let mut compiler = SelectCompiler::new(Some(&temporal_axes), false); + + let policy_components = PolicyComponents::builder(store, Some(actor)) + .with_action(ActionName::ViewEntity, MergePolicies::Yes) + .await + .change_context(VisibilityProofError::Policies)?; + + let actor = VisibilityActor { + id: actor, + instance_admin: policy_components.is_instance_admin(), + }; + + let policy_filter = Filter::::for_policies( + policy_components.extract_filter_policies(ActionName::ViewEntity), + policy_components.actor_id(), + policy_components.optimization_data(ActionName::ViewEntity), + ); + + if admits_every_row(filter, &policy_filter) { + return Ok(VisibilityMask::full(actor)); + } + + let converted; + let filter = match filter { + Some(filter) => { + let mut owned = filter.clone(); + owned + .convert_parameters(&StoreProvider::new(store, &policy_components)) + .await + .change_context(VisibilityProofError::Convert)?; + + converted = owned; + Some(&converted) + } + None => None, + }; + + // The store's read path transforms a caller's filter whenever the deployment configures + // protection and the actor is not an instance admin. The same condition governs here, since the + // same filter reaches the same compiler. + let protected; + let filter = match filter { + Some(filter) if actor.masked_by(protection) => { + protected = + transform_filter(filter.clone(), protection, 0, policy_components.actor_id()); + Some(&protected) + } + filter => filter, + }; + + if let Some(filter) = filter { + compiler + .add_filter(filter) + .change_context(VisibilityProofError::Filter)?; + } + compiler + .add_filter(&policy_filter) + .change_context(VisibilityProofError::PolicyFilter)?; + + let web_id_index = compiler.add_selection_path(&EntityQueryPath::WebId); + let uuid_index = compiler.add_selection_path(&EntityQueryPath::Uuid); + + let (statement, parameters) = compiler.compile(); + let stream = store + .as_client() + .query_raw(&statement, parameters) + .await + .change_context(VisibilityProofError::Query)?; + + let mut nodes = CompressedBitSet::default(); + let mut edges = CompressedBitSet::default(); + + let mut placed = 0_u64; + let mut unplaced = 0_u64; + + let mut stream = pin!(stream); + while let Some(row) = stream + .try_next() + .await + .change_context(VisibilityProofError::Query)? + { + let web_id: WebId = row.get(web_id_index); + let uuid: EntityUuid = row.get(uuid_index); + + let id = ArchivedEntityId { + web_id: ArchivedWebId::from(Uuid::from(web_id)), + entity_uuid: ArchivedEntityUuid::from(Uuid::from(uuid)), + }; + + if let Some(row_id) = world.layout.index.row_of(epoch, id) { + nodes.insert(row_id); + placed += 1; + } else if let Some(row_id) = world.topology.row_of(epoch, id) { + edges.insert(row_id); + placed += 1; + } else { + unplaced += 1; + } + } + + tracing::debug!( + nodes = nodes.count(), + edges = edges.count(), + placed, + unplaced, + "resolved the actor's visible rows" + ); + + Ok(VisibilityMask::partial(actor, nodes, edges)) +} + +#[cfg(test)] +mod tests { + use core::assert_matches; + + use hash_graph_authorization::policies::{ + Effect, OptimizationData, resource::ResourceConstraint, + }; + use hash_graph_store::filter::Filter; + use type_system::{ + knowledge::{Entity, entity::id::EntityId}, + principal::actor_group::WebId, + }; + use uuid::Uuid; + + use super::admits_every_row; + + /// The compiled tautology plus no caller filter is the unconstrained view. + /// + /// [`Filter::for_policies`] defines the unconstrained filter's compiled representation. + #[test] + fn unconstrained_permit() { + let optimization = OptimizationData::default(); + + let unconstrained = + Filter::::for_policies([(Effect::Permit, None)], None, &optimization); + assert!( + admits_every_row(None, &unconstrained), + "an unconstrained permit with no caller filter admits every row: {unconstrained:?}" + ); + + // A web-scoped permit is the same actor shape with one resource constraint, and it must + // keep the query. + let web = ResourceConstraint::Web { + web_id: WebId::new(Uuid::nil()), + }; + let scoped = + Filter::::for_policies([(Effect::Permit, Some(&web))], None, &optimization); + assert!(!admits_every_row(None, &scoped)); + + // A blank forbid denies everything, and no permit at all denies everything: neither is the + // tautology, and reading either as one would invert the decision. + let forbidden = + Filter::::for_policies([(Effect::Forbid, None)], None, &optimization); + assert!(!admits_every_row(None, &forbidden)); + let silent = Filter::::for_policies([], None, &optimization); + assert!(!admits_every_row(None, &silent)); + + // An unconstrained permit met by a forbid compiles to a negation, which the store must + // still evaluate. + let partly = Filter::::for_policies( + [(Effect::Permit, None), (Effect::Forbid, Some(&web))], + None, + &optimization, + ); + assert!(!admits_every_row(None, &partly)); + + // A scoped permit met by a forbid is the one shape that compiles to a non-empty + // conjunction. It is the dangerous neighbour of the tautology: reading the constructor + // rather than the conjunction's emptiness would answer this scoped actor with the whole + // generation. + let elsewhere = ResourceConstraint::Web { + web_id: WebId::new(Uuid::from_u128(1)), + }; + let scoped_with_forbid = Filter::::for_policies( + [ + (Effect::Permit, Some(&web)), + (Effect::Forbid, Some(&elsewhere)), + ], + None, + &optimization, + ); + assert_matches!(&scoped_with_forbid, Filter::All(conjuncts) if !conjuncts.is_empty(), "the fixture builds the non-empty conjunction it is here to reject: \ + {scoped_with_forbid:?}"); + assert!(!admits_every_row(None, &scoped_with_forbid)); + } + + #[test] + fn caller_filter() { + let optimization = OptimizationData::default(); + let unconstrained = + Filter::::for_policies([(Effect::Permit, None)], None, &optimization); + let requested = Filter::::for_entity_by_entity_id(EntityId { + web_id: WebId::new(Uuid::nil()), + entity_uuid: type_system::knowledge::entity::id::EntityUuid::new(Uuid::nil()), + draft_id: None, + }); + + assert!(!admits_every_row(Some(&requested), &unconstrained)); + } +} diff --git a/libs/@local/graph/atlas/src/serve/mod.rs b/libs/@local/graph/atlas/src/serve/mod.rs index 0bdfeeb846f..3c3e4667763 100644 --- a/libs/@local/graph/atlas/src/serve/mod.rs +++ b/libs/@local/graph/atlas/src/serve/mod.rs @@ -9,7 +9,7 @@ //! still use only the request's captured publication. //! //! Authority validation binds the generation and [`delta::DeltaId`], not a -//! [`delta::DeltaRevision`]. A `runtime::registry::Observation` captures the immutable revision +//! [`delta::DeltaRevision`]. A [`runtime::registry::Observation`] captures the immutable revision //! used for data reads. Authority-token expiry, visibility-cache age and //! retained-generation admission are distinct checks even when the host derives them from one //! maximum duration. Issuing or renewing a token does not force a permission-store refresh. A @@ -19,7 +19,9 @@ pub(crate) mod codec; pub(crate) mod delta; pub(crate) mod density; +pub(crate) mod hydrate; mod intern; +pub(crate) mod runtime; mod schedule; pub(crate) mod secret; #[cfg(test)] diff --git a/libs/@local/graph/atlas/src/serve/runtime/manager/error.rs b/libs/@local/graph/atlas/src/serve/runtime/manager/error.rs new file mode 100644 index 00000000000..4c380aeda4d --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/manager/error.rs @@ -0,0 +1,37 @@ +//! The failures generation maintenance reports. + +use core::{error::Error, fmt}; + +/// A generation maintenance configuration or operation failure. +#[derive(Debug)] +pub(crate) enum ManagerError { + /// The current-pointer polling interval is zero. + InvalidInterval, + /// Reading the current-generation pointer failed. + Current, + /// Opening the generation's metadata failed. + Open, + /// Initializing the generation's runtime failed. + Runtime, + /// Removing an expired generation failed. + Remove, + /// An offloaded operation failed to return a value. + Offload, +} + +impl fmt::Display for ManagerError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::InvalidInterval => { + fmt.write_str("the current-pointer polling interval must be non-zero") + } + Self::Current => fmt.write_str("could not read the current-generation pointer"), + Self::Open => fmt.write_str("could not open the generation metadata"), + Self::Runtime => fmt.write_str("could not initialize the generation runtime"), + Self::Remove => fmt.write_str("could not remove the expired generation"), + Self::Offload => fmt.write_str("the generation maintenance worker failed"), + } + } +} + +impl Error for ManagerError {} diff --git a/libs/@local/graph/atlas/src/serve/runtime/manager/mod.rs b/libs/@local/graph/atlas/src/serve/runtime/manager/mod.rs new file mode 100644 index 00000000000..1d27a632db7 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/manager/mod.rs @@ -0,0 +1,365 @@ +//! Process-level generation promotion and feed retirement. +//! +//! [`GenerationManager`] keeps execution ownership separate from the request handles in +//! [`UniverseRegistry`]. Opening failures preserve the last published world and delta. An ended +//! present feed retries with a new [delta lifetime](crate::serve::delta::DeltaId), while an ended +//! retained feed preserves its final publication until expiry or reactivation. Each restart +//! samples a new tag without obtaining a uniqueness guarantee. + +use alloc::sync::Arc; +use core::{ + future::{self, Future}, + pin::pin, + task::{Context, Poll, ready}, + time::Duration, +}; +use std::time::Instant; + +use error_stack::Report; +use futures::FutureExt as _; +use hashql_core::collections::FastHashMap; +use tokio::time::MissedTickBehavior; + +use self::{ + error::ManagerError, + slot::{Execution, Registration, RuntimeSlot}, + source::{RuntimeSource, RuntimeSourceHandle}, +}; +use super::registry::{Observer, UniverseRegistry}; +use crate::{file::generation::GenerationId, offload::OffloadState}; + +pub(crate) mod error; +mod slot; +pub(crate) mod source; +#[cfg(test)] +mod tests; + +/// Current-pointer polling and optional removal of expired generations. +#[derive(Default, Copy, Clone)] +pub(crate) struct ManagerOptions { + /// Time between maintenance passes, one second by default. + pub poll_interval: Duration = Duration::from_secs(1), + /// Controls removal of expired directories after feed joining. + /// + /// The default retains them. + pub unlink: bool = false, +} + +/// Owned generation execution with time-bounded admission to retained publications. +/// +/// Opening, joining and removal remain owned across cancellation of a borrowing [`Self::run`] or +/// [`Self::shutdown`] wait. Shutdown stops every initialized feed before waiting for any result. +/// Removing an expired directory does not wait for request-held worlds or epochs. +#[must_use = "the generation manager must run and drain its owned operations"] +pub(crate) struct GenerationManager { + options: ManagerOptions, + + source: Arc, + registry: Arc, + + slots: FastHashMap, + + desired: Option, + present: Option, + current: Option>>, + + expired: Vec, + stopping: bool, +} + +impl GenerationManager { + /// Creates an empty registry and an unstarted maintenance owner. + /// + /// `retention` is the interval from replacement promotion to the old generation's admission + /// expiry. Directory removal is independent of this interval's enforcement. + /// + /// # Errors + /// + /// Returns [`ManagerError::InvalidInterval`] for a zero polling interval. + pub(crate) fn new( + source: RuntimeSource, + options: ManagerOptions, + retention: Duration, + ) -> Result> { + if options.poll_interval.is_zero() { + return Err(Report::new(ManagerError::InvalidInterval)); + } + + Ok(Self { + source: Arc::new(source), + registry: Arc::new(UniverseRegistry::new(retention)), + options, + slots: FastHashMap::default(), + desired: None, + present: None, + current: None, + expired: Vec::new(), + stopping: false, + }) + } + + /// Borrows the registry requests take publication handles from. + pub(crate) const fn registry(&self) -> &Arc { + &self.registry + } + + /// Reads the current-pointer lookup's result when it is already available. + /// + /// This never blocks on the pointer read and leaves a running lookup alone. + fn try_join_current(&mut self) { + let Some(current) = &mut self.current else { + return; + }; + + let result = match current.try_join() { + Ok(OffloadState::Running) => return, + Ok(OffloadState::Finished(desired)) => Ok(desired), + Err(error) => Err(error), + }; + + self.finish_current(result); + } + + /// Polls the current-pointer lookup to completion, ready at once when none is outstanding. + fn poll_current(&mut self, context: &mut Context<'_>) -> Poll<()> { + let Some(current) = &mut self.current else { + return Poll::Ready(()); + }; + + let result = ready!(current.poll_unpin(context)); + self.finish_current(result); + Poll::Ready(()) + } + + /// Records a finished pointer read, keeping the previous selection on failure. + fn finish_current(&mut self, result: Result, Report>) { + self.current = None; + match result { + Ok(desired) => self.desired = desired, + Err(error) => tracing::warn!(?error, "failed to read current-generation pointer"), + } + } + + /// Admits `generation`'s runtime to the registry and makes it the present one. + /// + /// Promotion is refused where the slot holds no runtime, and a closed registry stops the + /// manager instead, because a manager that cannot publish has nothing left to maintain. + fn promote(&mut self, generation: GenerationId, now: Instant) { + let Some(slot) = self.slots.get_mut(&generation) else { + tracing::warn!(%generation, "tried to promote a non-existent generation"); + return; + }; + + let Some(runtime) = slot.runtime() else { + tracing::warn!( + %generation, + "tried to promote a generation without a runtime" + ); + + return; + }; + + if self.registry.promote(Observer::from(runtime), now).is_err() { + tracing::info!( + "universe registry is closed, and is no longer accepting new observers, shutting \ + down" + ); + + self.stopping = true; + return; + } + + slot.promote(); + self.present = Some(generation); + tracing::info!(%generation, "promoted the generation runtime"); + } + + /// Runs one maintenance pass over the pointer, the slots and the registry. + /// + /// The pass reads the current pointer and advances every slot. It expires admissions that + /// have outlived their retention, promotes the selected generation where one is ready, and + /// then reconciles the slot set against what is wanted. Slots advance both before and after + /// reconciliation so a slot opened during the pass makes progress within it. + fn tick(&mut self, now: Instant) { + self.try_join_current(); + + if self.current.is_none() { + self.current = Some(Arc::clone(&self.source).current()); + self.try_join_current(); + } + + for (&generation, slot) in &mut self.slots { + slot.tick(generation); + } + + self.registry.expire(now, &mut self.expired); + for generation in self.expired.drain(..) { + if let Some(slot) = self.slots.get_mut(&generation) { + slot.expire(); + } + } + + self.slots + .retain(|_generation, slot| !matches!(slot.execution, Execution::Removed)); + + if let Some(desired) = self.desired + && self.has_runtime(desired) + && (self.present != Some(desired) || self.is_ready(desired)) + { + self.promote(desired, now); + } + + if let Some(present) = self.present + && self.is_ready(present) + { + self.promote(present, now); + } + + if self.stopping { + self.stop(); + return; + } + + if let Some(desired) = self.desired { + self.slots + .entry(desired) + .or_insert_with(RuntimeSlot::candidate); + } + self.reconcile(); + + // Advance all slots, even if they're not active yet. + for (&generation, slot) in &mut self.slots { + slot.tick(generation); + } + } + + /// Returns whether `generation` holds an initialized runtime not yet promoted. + fn is_ready(&self, generation: GenerationId) -> bool { + self.slots + .get(&generation) + .is_some_and(|slot| matches!(slot.execution, Execution::Ready(_))) + } + + /// Returns whether `generation` holds an initialized runtime, ready or already running. + /// + /// Answers `false` for every other execution state, an absent slot among them. + fn has_runtime(&self, generation: GenerationId) -> bool { + self.slots + .get(&generation) + .is_some_and(|slot| slot.runtime().is_some()) + } + + /// Brings the slot set in line with the desired and present generations. + /// + /// A wanted generation is opened or reopened, an unwanted stopped one leaves unless it is + /// still published, and an expired one is removed where the operator enabled unlinking. + fn reconcile(&mut self) { + self.slots.retain(|&generation, slot| { + let wanted = self.desired == Some(generation) || self.present == Some(generation); + + if matches!(slot.execution, Execution::Ready(_)) { + // Only promotion accepts an initialized candidate. Superseded openings still stop + // and join their newly started feed. + slot.stop(); + } + + if !slot.is_stopped() { + return true; + } + + if wanted { + slot.open(Arc::clone(&self.source), generation); + return true; + } + + match slot.registration { + Registration::Published => true, + Registration::Expired if self.options.unlink => { + slot.remove(Arc::clone(&self.source), generation); + true + } + Registration::Candidate | Registration::Expired => false, + } + }); + } + + /// Closes admission and asks every slot to stop, without waiting for any of them. + fn stop(&mut self) { + self.stopping = true; + self.registry.close(); + + for slot in self.slots.values_mut() { + slot.stop(); + } + } + + /// Maintains generations until shutdown, then drains all owned operations. + /// + /// A missing or unreadable current pointer preserves the present publication. Initialization + /// failures retry at the maintenance cadence while the generation remains selected or present. + /// Resolving `shutdown` closes admission and drains owned work. Dropping this borrowing future + /// before completion instead leaves pending handles and results in the manager for a later + /// call. + /// + /// # Panics + /// + /// Panics outside a Tokio runtime with time enabled. A sufficiently late tick can also panic + /// when adding [`ManagerOptions::poll_interval`] to the current instant would exceed Tokio's + /// representable deadline. + #[expect( + clippy::integer_division_remainder_used, + reason = "Tokio select traverses its branch set with a remainder" + )] + pub(crate) async fn run(&mut self, shutdown: impl Future) { + let mut shutdown = pin!(shutdown); + let mut interval = tokio::time::interval(self.options.poll_interval); + interval.set_missed_tick_behavior(MissedTickBehavior::Delay); + + while !self.stopping { + tokio::select! { + biased; + () = &mut shutdown => break, + _tick = interval.tick() => self.tick(Instant::now()), + } + } + + self.shutdown().await; + } + + /// Closes admission and joins every feed and outstanding maintenance operation. + /// + /// Completed openings also stop and join their feeds. Already-started removals finish, and + /// shutdown starts no new removal or recovery. Dropping this borrowing future before completion + /// retains every pending handle and result in the manager. + pub(crate) async fn shutdown(&mut self) { + self.stop(); + + future::poll_fn(|context| { + let mut complete = self.poll_current(context).is_ready(); + for (&generation, slot) in &mut self.slots { + complete &= slot.poll_shutdown(generation, context).is_ready(); + } + + if complete { + Poll::Ready(()) + } else { + Poll::Pending + } + }) + .await; + + self.slots.clear(); + } +} + +impl Drop for GenerationManager { + /// Signals every slot to stop when the manager is dropped. + /// + /// This signals and does not wait: a destructor cannot await. Therefore, a caller that needs + /// the feeds joined calls [`Self::shutdown`] before dropping. Dropping without it leaves + /// runtime tasks stopping in the background. Current-pointer, opening and removal work already + /// submitted to Rayon also continues, but the manager abandons its result handles. + fn drop(&mut self) { + self.stop(); + } +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/manager/slot.rs b/libs/@local/graph/atlas/src/serve/runtime/manager/slot.rs new file mode 100644 index 00000000000..27a6a900d79 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/manager/slot.rs @@ -0,0 +1,264 @@ +//! One generation's execution state machine. +//! +//! [`GenerationManager`](super::GenerationManager) stores one slot per generation to track its +//! execution and publication history. It drives slot transitions during maintenance and shutdown, +//! including drop-triggered stop requests. + +use alloc::sync::Arc; +use core::{ + mem, + task::{Context, Poll, ready}, +}; + +use futures::FutureExt as _; + +use super::source::{RuntimeSource, RuntimeSourceHandle}; +use crate::{ + file::generation::GenerationId, + offload::OffloadState, + serve::{ + runtime::{FeedState, Runtime}, + world::World, + }, +}; + +/// Admission history used to distinguish failed candidates from expired generations. +pub(super) enum Registration { + /// Selected but never promoted, with no publications to retain after stopping. + Candidate, + /// Promoted at least once, so requests may still hold its publications. + Published, + /// Published, and past the retention interval that followed its replacement. + Expired, +} + +/// Where one generation stands in the open, run, join, remove sequence. +/// +/// The states cycle. A stopped generation that becomes wanted reopens, and a failed opening +/// returns to stopped for retry during the next maintenance pass. [`Execution::Opening`] and +/// [`Execution::Stopped`] may retain an opened world, allowing a reopened generation to skip +/// repeated artifact mapping. +pub(super) enum Execution { + /// An opening is running on the offload pool. + Opening { + /// The world a previous opening left mapped, reused by this one. + world: Option>, + /// The opening's handle. + task: RuntimeSourceHandle, + }, + /// Initialized and not yet promoted. + Ready(Runtime), + /// Promoted, with requests reading its publications. + Running(Runtime), + /// Shutdown requested, waiting for the feed's result. + Joining(Runtime), + /// Not running, retaining the opened world where one survived. + Stopped(Option>), + /// The generation's directory is being unlinked. + Removing(RuntimeSourceHandle<()>), + /// The directory is gone and the slot may leave the manager. + Removed, +} + +impl Execution { + /// Requests feed shutdown, moving an initialized runtime into joining. + /// + /// Every other state already has its shutdown underway or behind it. + fn stop(&mut self) { + if matches!(self, Self::Ready(_) | Self::Running(_)) { + let (Self::Ready(runtime) | Self::Running(runtime)) = + mem::replace(self, Self::Stopped(None)) + else { + unreachable!("only initialized runtimes enter joining"); + }; + runtime.stop(); + *self = Self::Joining(runtime); + } + } + + /// Advances as far as the already-available results allow. + /// + /// The loop runs because one finished operation can enable the next: an opening that finished + /// leaves a ready runtime, and a joined feed leaves a stopped slot the same pass can reopen. + /// Each arm returns rather than looping where its operation is still running. + fn tick(&mut self, generation: GenerationId) { + loop { + let next = match self { + Self::Opening { world, task } => match task.try_join() { + Ok(OffloadState::Finished(runtime)) => Self::Ready(runtime), + Ok(OffloadState::Running) => return, + Err(error) => { + tracing::warn!(%generation, ?error, "retrying generation initialization"); + + Self::Stopped(world.take()) + } + }, + Self::Running(runtime) => { + match runtime.try_join() { + Ok(FeedState::Finished) => { + tracing::warn!(%generation, "generation feed ended prematurely"); + } + Err(error) => { + tracing::warn!(%generation, ?error, "unable to join runtime, generation feed failed"); + } + Ok(FeedState::Absent | FeedState::Running) => return, + } + + Self::Stopped(Some(Arc::clone(runtime.world()))) + } + Self::Joining(runtime) => { + match runtime.try_join() { + Ok(FeedState::Absent | FeedState::Finished) => {} + Err(error) => { + tracing::warn!(%generation, ?error, "unable to join runtime, generation feed failed"); + } + Ok(FeedState::Running) => return, + } + + Self::Stopped(Some(Arc::clone(runtime.world()))) + } + Self::Removing(task) => match task.try_join() { + Ok(OffloadState::Finished(())) => { + tracing::info!(%generation, "removed expired generation"); + + Self::Removed + } + Ok(OffloadState::Running) => return, + Err(error) => { + tracing::warn!(%generation, ?error, "failed to remove expired generation"); + // Removal may fail after deleting only part of the directory. Reopening + // must validate its remaining artifacts. + Self::Stopped(None) + } + }, + Self::Ready(_) | Self::Stopped(_) | Self::Removed => return, + }; + + *self = next; + } + } +} + +/// Execution ownership independent of the registry's retained request publications. +pub(super) struct RuntimeSlot { + pub registration: Registration, + pub execution: Execution, +} + +impl RuntimeSlot { + /// Returns an unopened slot for a generation the manager selected but never published. + pub(super) const fn candidate() -> Self { + Self { + registration: Registration::Candidate, + execution: Execution::Stopped(None), + } + } + + /// Borrows the slot's initialized runtime, absent in every other execution state. + pub(super) const fn runtime(&self) -> Option<&Runtime> { + match &self.execution { + Execution::Ready(runtime) | Execution::Running(runtime) => Some(runtime), + Execution::Opening { .. } + | Execution::Joining(_) + | Execution::Stopped(_) + | Execution::Removing(_) + | Execution::Removed => None, + } + } + + /// Records publication and moves a ready runtime into running. + pub(super) fn promote(&mut self) { + self.registration = Registration::Published; + if matches!(self.execution, Execution::Ready(_)) { + let Execution::Ready(runtime) = + mem::replace(&mut self.execution, Execution::Stopped(None)) + else { + unreachable!("only a ready runtime changes to running"); + }; + self.execution = Execution::Running(runtime); + } + } + + /// Records that retention has elapsed and requests shutdown. + pub(super) fn expire(&mut self) { + self.registration = Registration::Expired; + self.stop(); + } + + /// Requests feed shutdown without waiting. + pub(super) fn stop(&mut self) { + self.execution.stop(); + } + + /// Advances the slot's execution as far as available results allow. + pub(super) fn tick(&mut self, generation: GenerationId) { + self.execution.tick(generation); + } + + /// Stops initialized feeds and polls every operation this slot already owns. + pub(super) fn poll_shutdown( + &mut self, + generation: GenerationId, + context: &mut Context<'_>, + ) -> Poll<()> { + loop { + let next = match &mut self.execution { + Execution::Opening { world, task } => match ready!(task.poll_unpin(context)) { + Ok(runtime) => { + runtime.stop(); + Execution::Joining(runtime) + } + Err(error) => { + tracing::warn!(%generation, ?error, "generation initialization failed during shutdown"); + Execution::Stopped(world.take()) + } + }, + Execution::Ready(_) | Execution::Running(_) => { + self.stop(); + continue; + } + Execution::Joining(runtime) => { + if let Some(Err(error)) = ready!(runtime.poll_join(context)) { + tracing::warn!(%generation, ?error, "unable to join runtime, generation feed failed"); + } + Execution::Stopped(Some(Arc::clone(runtime.world()))) + } + Execution::Removing(task) => match ready!(task.poll_unpin(context)) { + Ok(()) => { + tracing::info!(%generation, "removed expired generation"); + Execution::Removed + } + Err(error) => { + tracing::warn!(%generation, ?error, "failed to remove expired generation"); + Execution::Stopped(None) + } + }, + Execution::Stopped(_) | Execution::Removed => return Poll::Ready(()), + }; + + self.execution = next; + } + } + + /// Starts an opening for a stopped slot, reusing a retained world where one survived. + /// + /// A repeated call cannot start a second opening. Slots in any other state remain unchanged. + pub(super) fn open(&mut self, source: Arc, generation: GenerationId) { + if let Execution::Stopped(world) = &mut self.execution { + let world = world.take(); + + let task = source.open(generation, world.as_ref().map(Arc::clone)); + self.execution = Execution::Opening { world, task }; + } + } + + /// Starts unlinking the generation's directory. + pub(super) fn remove(&mut self, source: Arc, generation: GenerationId) { + self.execution = Execution::Removing(source.remove(generation)); + } + + /// Returns whether the slot has no operation left to poll. + pub(super) const fn is_stopped(&self) -> bool { + matches!(self.execution, Execution::Stopped(_) | Execution::Removed) + } +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/manager/source.rs b/libs/@local/graph/atlas/src/serve/runtime/manager/source.rs new file mode 100644 index 00000000000..f5ec2903f79 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/manager/source.rs @@ -0,0 +1,161 @@ +//! The blocking generation operations, run off the maintenance loop. +//! +//! Opening maps and verifies a generation's artifacts, while removal unlinks its directory. The +//! offload pool performs both blocking filesystem operations, and the manager retains handles it +//! can probe or await. + +use alloc::sync::Arc; +use core::{panic::AssertUnwindSafe, pin, task}; + +use error_stack::{Report, ResultExt as _}; +use futures::FutureExt as _; +use hash_graph_postgres_store::store::PostgresStorePool; +use rand::rngs::SysRng; +use tokio::runtime::Handle; + +use super::error::ManagerError; +use crate::{ + file::generation::{GenerationId, GenerationRoot}, + offload::{self, OffloadHandle, OffloadState}, + serve::{ + runtime::{FeedOptions, Runtime}, + secret::ServeSecret, + world::World, + }, +}; + +/// A generation operation running on the offload pool. +/// +/// Opening, current-pointer reads and directory removal are all blocking filesystem operations. +/// Maintenance must not block on them, so [`try_join`](Self::try_join) probes for a completed +/// result. The [`Future`] implementation lets shutdown await completion. Dropping the handle +/// abandons the result but does not cancel work already submitted to Rayon. +/// +/// An [`OffloadState::Running`] probe leaves the handle available for another probe or await. Every +/// other probe result consumes the one-shot. A subsequent probe returns [`ManagerError::Offload`], +/// including after a successful operation. +/// +/// # Panics +/// +/// Polling through [`Future`] after a terminal probe or a completed poll panics. +pub(crate) struct RuntimeSourceHandle(OffloadHandle>>); + +impl RuntimeSourceHandle { + /// Takes the operation's result when it is already available. + /// + /// # Errors + /// + /// Returns the operation's own [`ManagerError`], or [`ManagerError::Offload`] when the worker + /// panics, exits without delivering a result, or this handle's result has already been + /// consumed. + pub(crate) fn try_join(&mut self) -> Result, Report> { + match self.0.try_join() { + Ok(OffloadState::Running) => Ok(OffloadState::Running), + Ok(OffloadState::Finished(Ok(result))) => Ok(OffloadState::Finished(result)), + Ok(OffloadState::Finished(Err(error))) => Err(error), + Err(offload) => Err(Report::new(offload).change_context(ManagerError::Offload)), + } + } +} + +impl Future for RuntimeSourceHandle { + type Output = Result>; + + /// Polls the offloaded step and flattens its two failure shapes into one. + /// + /// The operation's own failure passes through unchanged. A worker panic or disappearance + /// becomes [`ManagerError::Offload`], giving both failure shapes the same result type. + fn poll(mut self: pin::Pin<&mut Self>, cx: &mut task::Context<'_>) -> task::Poll { + self.0.poll_unpin(cx).map(|result| match result { + Ok(Ok(current)) => Ok(current), + Err(offload) => Err(Report::new(offload).change_context(ManagerError::Offload)), + Ok(Err(error)) => Err(error), + }) + } +} + +/// Shared resources for opening generations and restarting their feeds. +pub(crate) struct RuntimeSource { + /// The directory published generations live under. + pub root: GenerationRoot, + /// The server secret used to derive an opened world's wire-ID codecs. + pub secret: ServeSecret, + /// The store connection pool a started feed reads through. + pub pool: Arc, + /// The feed configuration, absent where the deployment serves without one. + pub feed: Option, +} + +impl RuntimeSource { + /// Reads the root's current-generation pointer. + /// + /// Returns [`None`] when no current pointer selects a generation, including when unselected + /// generation directories exist. + pub(super) fn current(self: Arc) -> RuntimeSourceHandle> { + RuntimeSourceHandle(offload::run(AssertUnwindSafe(move || { + self.root.current().change_context(ManagerError::Current) + }))) + } + + /// Opens `generation` or reuses `world` as a runtime with optional feed execution. + /// + /// `Some(world)` ignores `generation` and starts a new delta lifetime over the already-open + /// artifacts. `None` opens and verifies `generation` first. Reuse avoids repeating artifact + /// mapping and verification. Either path returns a static reader when feed options or temporal + /// axes are absent. + /// + /// # Panics + /// + /// Panics outside a Tokio runtime because the operation captures its handle before either open + /// path. + pub(super) fn open( + self: Arc, + generation: GenerationId, + world: Option>, + ) -> RuntimeSourceHandle { + let handle = Handle::current(); + + RuntimeSourceHandle(offload::run(AssertUnwindSafe(move || { + // Feed initialization starts Tokio tasks from the Rayon worker. + let _entered = handle.enter(); + + let feed = self.feed.clone(); + let pool = Arc::clone(&self.pool); + + if let Some(world) = world { + Runtime::start(world, pool, SysRng, feed).change_context(ManagerError::Runtime) + } else { + let generation = self + .root + .open(generation) + .change_context(ManagerError::Open)?; + + Runtime::open(generation, &self.secret, pool, SysRng, feed) + .change_context(ManagerError::Runtime) + } + }))) + } + + /// Unlinks `generation`'s directory. + pub(super) fn remove(self: Arc, generation: GenerationId) -> RuntimeSourceHandle<()> { + RuntimeSourceHandle(offload::run(AssertUnwindSafe(move || { + self.root + .remove(generation) + .change_context(ManagerError::Remove) + }))) + } +} + +#[cfg(test)] +pub(super) mod tests { + use error_stack::Report; + + use super::{ManagerError, OffloadHandle, RuntimeSourceHandle}; + + /// Wraps an offload handle as a source handle, so manager tests can drive one directly. + pub(in super::super) fn from_offload( + handle: OffloadHandle>>, + ) -> RuntimeSourceHandle { + RuntimeSourceHandle(handle) + } +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/manager/tests.rs b/libs/@local/graph/atlas/src/serve/runtime/manager/tests.rs new file mode 100644 index 00000000000..2fef6ad06fd --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/manager/tests.rs @@ -0,0 +1,1984 @@ +//! Promotion, retention, recovery and retirement of one process's generations. +//! +//! Controlled feeds separate cancellation from completion. Supplied maintenance instants exercise +//! retirement without sleeping. + +use alloc::{sync::Arc, task::Wake}; +use core::{ + any::Any, + assert_matches, fmt, + future::{self, Future}, + mem, + pin::pin, + sync::atomic::{AtomicBool, Ordering}, + task::{Context, Poll, Waker}, + time::Duration, +}; +use std::{fs, sync::Mutex, time::Instant}; + +use error_stack::Report; +use futures::FutureExt as _; +use hash_graph_postgres_store::store::{ + DatabaseConnectionInfo, DatabasePoolConfig, DatabaseType, PostgresStorePool, + PostgresStoreSettings, +}; +use hashql_core::id::Id as _; +use rand::{SeedableRng as _, rngs::StdRng}; +use tokio::sync::oneshot; +use tokio_postgres::NoTls; +use tokio_util::sync::CancellationToken; + +use super::{ + GenerationManager, ManagerOptions, + error::ManagerError, + slot::{Execution, Registration, RuntimeSlot}, + source::{self, RuntimeSource, RuntimeSourceHandle}, +}; +use crate::{ + file::{ + generation::{GenerationId, GenerationRoot}, + repository::Artifact as _, + salt::artifact, + }, + identity::NodeRowId, + math::nz, + offload, + serve::{ + delta::{Delta, DeltaReader, DeltaReference}, + runtime::{ + Feed, Runtime, + registry::{ObserveError, UniverseRegistry}, + }, + tests::fixture::{TamperFixture, secret}, + world::World, + }, +}; + +impl fmt::Debug for Execution { + /// Names the execution's variant and nothing inside it. + /// + /// The contents are runtimes, join handles and offload tasks, none of them printable. + /// The variant is what the pattern assertions here report on a mismatch. Each variant with + /// a payload prints as non-exhaustive. + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + let variant = match self { + Self::Opening { .. } => "Opening", + Self::Ready(_) => "Ready", + Self::Running(_) => "Running", + Self::Joining(_) => "Joining", + Self::Stopped(_) => "Stopped", + Self::Removing(_) => "Removing", + Self::Removed => return fmt.write_str("Removed"), + }; + fmt.debug_tuple(variant).finish_non_exhaustive() + } +} + +/// The retention interval every case measures its deadlines against. +const HARD: Duration = Duration::from_secs(60); + +/// One published generation and its opened world. +struct Fixture { + files: TamperFixture, + world: Arc, +} + +impl Fixture { + /// Publishes one synthetic generation under `name` and opens its world. + /// + /// # Panics + /// + /// Panics if publishing the synthetic generation fails or its world does not open. + fn new(name: &str) -> Self { + let files = TamperFixture::publish(name); + let world = Arc::new( + World::open(files.generation().clone(), &secret()) + .expect("the synthetic generation should open"), + ); + + Self { files, world } + } + + /// Publishes a second generation in the same root, marked by `marker`. + /// + /// # Panics + /// + /// Panics on generation tampering, republication, or world opening failure, or if `marker` + /// leaves the generation identity unchanged. + #[track_caller] + fn variant(&self, marker: &str) -> Arc { + let generation = self.files.tamper(&artifact::Representations::NAME, |path| { + fs::remove_file(path).expect("the staged placeholder should be removable"); + fs::write(path, marker).expect("the variant placeholder should write"); + }); + let world = + Arc::new(World::open(generation, &secret()).expect("the variant world should open")); + assert_ne!( + world.generation().id(), + self.world.generation().id(), + "the variant should carry its own generation identity" + ); + + world + } + + /// Returns the generation root used by the manager. + /// + /// # Panics + /// + /// Panics if the fixture generation has no parent directory or root opening fails. + fn root(&self) -> GenerationRoot { + GenerationRoot::new( + self.world + .generation() + .path() + .parent() + .expect("the fixture generation has a root"), + ) + .expect("the fixture root should open") + } +} + +/// The handshakes of one controlled feed. +struct Handshake { + /// Completes once the task observes cancellation. + observed: oneshot::Receiver<()>, + /// Releases the task from its wait. + release: oneshot::Sender<()>, + /// Completes once the task's body ends. + ended: oneshot::Receiver<()>, +} + +/// Runs `test` on a paused current-thread runtime with a one-second virtual timeout. +/// +/// # Panics +/// +/// Panics if Tokio runtime construction fails, if `test` panics, or if `test` exceeds the virtual +/// timeout. +#[track_caller] +fn controlled(test: impl Future) { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .start_paused(true) + .build() + .expect("should build the runtime"); + + let result = + runtime.block_on(async { tokio::time::timeout(Duration::from_secs(1), test).await }); + result.expect("the controlled test should finish without a stalled task"); +} + +/// A waker target that records notifications without rescheduling the future. +struct WakeFlag(AtomicBool); + +impl Wake for WakeFlag { + fn wake(self: Arc) { + self.0.store(true, Ordering::Relaxed); + } +} + +/// Checks that `complete` wakes a suspended `task` and makes it ready. +/// +/// # Panics +/// +/// Panics when the task is ready before `complete` runs, when the completion does not wake it, or +/// when it is not ready after being woken. Propagates panics from polling `task` or invoking +/// `complete`. +#[track_caller] +fn assert_wakeup(task: impl Future, complete: impl FnOnce()) { + let wake = Arc::new(WakeFlag(AtomicBool::new(false))); + let waker = Waker::from(Arc::clone(&wake)); + let mut context = Context::from_waker(&waker); + let mut task = pin!(task); + assert_matches!(task.as_mut().poll(&mut context), Poll::Pending); + + complete(); + assert!( + wake.0.load(Ordering::Relaxed), + "completion should wake the suspended shutdown" + ); + assert_matches!(task.as_mut().poll(&mut context), Poll::Ready(())); +} + +/// Captures WARN and ERROR events from `pass` without ANSI escapes. +/// +/// # Panics +/// +/// Panics if `pass` panics, the capture lock has become poisoned, or formatted output is not UTF-8. +#[track_caller] +fn captured_warnings(pass: impl FnOnce()) -> String { + /// A shared byte buffer that captures one pass's formatted log events. + struct Capture(Arc>>); + + impl std::io::Write for Capture { + /// Appends every offered byte, returns `buf.len()` and never writes short. + /// + /// # Panics + /// + /// Panics when the buffer's lock is poisoned, which needs a panic inside another + /// writer. + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.0 + .lock() + .expect("the capture lock is never poisoned") + .extend_from_slice(buf); + Ok(buf.len()) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } + } + + impl<'writer> tracing_subscriber::fmt::MakeWriter<'writer> for Capture { + type Writer = Self; + + fn make_writer(&'writer self) -> Self::Writer { + Self(Arc::clone(&self.0)) + } + } + + let capture = Capture(Arc::new(Mutex::new(Vec::new()))); + let subscriber = tracing_subscriber::fmt() + .with_max_level(tracing::Level::WARN) + .with_writer(Capture(Arc::clone(&capture.0))) + .with_ansi(false) + .finish(); + tracing::subscriber::with_default(subscriber, pass); + + let bytes = capture + .0 + .lock() + .expect("the capture lock is never poisoned") + .clone(); + String::from_utf8(bytes).expect("formatted log output is UTF-8") +} + +/// Builds an empty store pool for the configured test socket path. +/// +/// # Panics +/// +/// Panics if store-pool construction fails. +async fn pool() -> Arc { + Arc::new( + PostgresStorePool::new( + &DatabaseConnectionInfo::new( + DatabaseType::Postgres, + "manager-test".to_owned(), + String::new(), + "/no-manager-test-postgres".to_owned(), + 5432, + "manager-test".to_owned(), + ), + &DatabasePoolConfig { + max_connections: nz!(1), + }, + NoTls, + PostgresStoreSettings::default(), + ) + .await + .expect("should construct an unconnected pool"), + ) +} + +/// A controlled source task's completion channel, carrying its slot result or panic payload. +type SourceReply = oneshot::Sender>, Box>>; + +/// Builds a source handle whose sender controls completion. +fn pending() -> (RuntimeSourceHandle, SourceReply) { + let (reply, receiver) = oneshot::channel(); + let task = source::tests::from_offload(offload::tests::from_receiver(receiver)); + (task, reply) +} + +/// Wraps `result` in an already-settled source handle. +/// +/// # Panics +/// +/// Panics when the task dropped its receiving end before the result was sent. +#[track_caller] +fn settled(result: Result>) -> RuntimeSourceHandle { + let (task, reply) = pending(); + assert!(reply.send(Ok(result)).is_ok(), "should retain the receiver"); + task +} + +/// Keeps the current-pointer read pending while each case supplies its selection. +/// +/// # Panics +/// +/// Panics on fixture root opening or store pool construction failure, or if the manager rejects +/// `options`. +async fn maintainer( + fixture: &Fixture, + options: ManagerOptions, +) -> (GenerationManager, SourceReply>) { + let source = RuntimeSource { + root: fixture.root(), + secret: secret(), + pool: pool().await, + feed: None, + }; + let mut manager = GenerationManager::new(source, options, HARD) + .expect("a non-zero interval should construct a manager"); + let (current, reply) = pending(); + manager.current = Some(current); + + (manager, reply) +} + +/// Returns the instant `offset` after `base`. +/// +/// # Panics +/// +/// Panics if adding `offset` exceeds the range of [`Instant`]. +#[track_caller] +fn after(base: Instant, offset: Duration) -> Instant { + base.checked_add(offset) + .expect("the offset should fit the instant's range") +} + +/// Builds a feed that ends after observing cancellation. +/// +/// # Panics +/// +/// Panics outside a Tokio runtime. The spawned task also panics if the caller drops its observation +/// receiver before cancellation. +fn prompt_feed() -> (Feed, oneshot::Receiver<()>) { + let (notify, observed) = oneshot::channel::<()>(); + let shutdown = CancellationToken::new(); + let cancelled = shutdown.clone().cancelled_owned(); + let task = tokio::spawn(async move { + cancelled.await; + notify.send(()).expect("the observation should remain open"); + Ok(()) + }); + + (Feed { shutdown, task }, observed) +} + +/// Builds a feed that waits for release after observing cancellation. +/// +/// # Panics +/// +/// Panics outside a Tokio runtime. The spawned task also panics if the caller drops an observation +/// or completion receiver, or drops the release sender before release. +fn stalled_feed() -> (Feed, Handshake) { + let (notify, observed) = oneshot::channel::<()>(); + let (release, released) = oneshot::channel::<()>(); + let (completion, ended) = oneshot::channel::<()>(); + let shutdown = CancellationToken::new(); + let cancelled = shutdown.clone().cancelled_owned(); + let task = tokio::spawn(async move { + cancelled.await; + notify.send(()).expect("the observation should remain open"); + released.await.expect("the test should release the feed"); + completion + .send(()) + .expect("the completion should remain open"); + Ok(()) + }); + + ( + Feed { shutdown, task }, + Handshake { + observed, + release, + ended, + }, + ) +} + +/// Builds a feed that ends on release without awaiting cancellation. +/// +/// # Panics +/// +/// Panics outside a Tokio runtime. The spawned task also panics if the caller drops its release +/// sender or completion receiver before the corresponding event. +fn ending_feed() -> (Feed, oneshot::Sender<()>, oneshot::Receiver<()>) { + let (release, released) = oneshot::channel::<()>(); + let (completion, ended) = oneshot::channel::<()>(); + let task = tokio::spawn(async move { + released.await.expect("the test should release the feed"); + completion + .send(()) + .expect("the completion should remain open"); + Ok(()) + }); + + ( + Feed { + shutdown: CancellationToken::new(), + task, + }, + release, + ended, + ) +} + +/// Builds a runtime over `world` with its delta identity seeded by `seed`. +fn runtime(world: Arc, seed: u64, feed: Option) -> Runtime { + let delta = Delta::new(Arc::clone(&world), StdRng::seed_from_u64(seed)) + .expect("the seeded generator should draw a delta identity"); + + Runtime { + world, + reader: DeltaReader::from(delta), + feed, + } +} + +/// Builds a candidate slot containing an initialized runtime. +fn ready(runtime: Runtime) -> RuntimeSlot { + RuntimeSlot { + registration: Registration::Candidate, + execution: Execution::Ready(runtime), + } +} + +/// Builds a candidate slot containing an active opening. +fn opening(task: RuntimeSourceHandle) -> RuntimeSlot { + RuntimeSlot { + registration: Registration::Candidate, + execution: Execution::Opening { world: None, task }, + } +} + +/// Returns the world and delta lifetime visible to a fresh request for `generation`. +/// +/// The active world and captured epoch must both name `generation`, whose first node must have a +/// visible position. +/// +/// # Panics +/// +/// Panics if the registry does not admit its active generation, the captured world or epoch names +/// another generation, or the first node has no visible position. +#[track_caller] +fn published( + registry: &UniverseRegistry, + generation: GenerationId, +) -> (Arc, DeltaReference) { + let observation = registry + .observe(None) + .expect("the active generation should admit"); + let present = observation.present(); + + assert_eq!( + present.world().generation().id(), + generation, + "the active selection should hold the promoted generation's world" + ); + assert_eq!( + present.epoch().generation(), + generation, + "the epoch should name the promoted generation" + ); + assert!( + present + .world() + .layout + .position(present.epoch(), NodeRowId::MIN) + .is_some(), + "the first node row should have a position at the published epoch" + ); + + (Arc::clone(present.world()), present.epoch().reference()) +} + +/// Checks the registry refuses `generation` as unavailable. +/// +/// # Panics +/// +/// Panics if the registry admits `generation` or refuses it for another reason. +#[track_caller] +fn assert_unavailable(registry: &UniverseRegistry, generation: GenerationId) { + let Err(error) = registry.observe(Some(generation)) else { + panic!("the unavailable generation should not admit") + }; + assert_matches!( + error, ObserveError::Unavailable(refused) if refused == generation, + "the refusal should name the unavailable generation" + ); +} + +/// Checks the registry has closed admission. +/// +/// # Panics +/// +/// Panics if the registry admits the request or refuses it for another reason. +#[track_caller] +fn assert_closed(registry: &UniverseRegistry) { + let Err(error) = registry.observe(None) else { + panic!("a closed registry should not admit") + }; + assert_matches!( + error, + ObserveError::Closed, + "the refusal should name closed admission" + ); +} + +/// Completes an opening and returns its result to the manager for the next maintenance pass. +/// +/// # Panics +/// +/// Panics if `generation` has no slot or its unfinished execution is not an active opening. +async fn settle_opening(manager: &mut GenerationManager, generation: GenerationId) { + let slot = manager + .slots + .get_mut(&generation) + .expect("the generation should hold a slot"); + if let Execution::Ready(_) = slot.execution { + return; + } + let Execution::Opening { world, task } = mem::replace(&mut slot.execution, Execution::Removed) + else { + panic!("the generation should hold a started opening") + }; + + let result = task.await; + slot.execution = Execution::Opening { + world, + task: settled(result), + }; +} + +/// Completes a removal and returns its result to the manager for the next maintenance pass. +/// +/// # Panics +/// +/// Panics if `generation` has no slot or its unfinished execution is not an active removal. +async fn settle_removal(manager: &mut GenerationManager, generation: GenerationId) { + let slot = manager + .slots + .get_mut(&generation) + .expect("the generation should hold a slot"); + if matches!(slot.execution, Execution::Removed) { + return; + } + let Execution::Removing(task) = mem::replace(&mut slot.execution, Execution::Removed) else { + panic!("the generation should hold a started removal") + }; + + let result = task.await; + slot.execution = Execution::Removing(settled(result)); +} + +/// Removes the pointer read, drains the manager and checks every feed observed cancellation. +/// +/// # Panics +/// +/// Panics if a stopped feed drops its observation sender without reporting cancellation. +async fn drain( + manager: &mut GenerationManager, + observations: impl IntoIterator>, +) { + manager.current = None; + manager.shutdown().await; + for observation in observations { + observation + .await + .expect("every stopped feed should observe cancellation"); + } +} + +/// Fails a controlled case whose task stalls. +#[test] +#[should_panic(expected = "the controlled test should finish without a stalled task")] +fn controlled_stall() { + controlled(async { + tokio::time::advance(Duration::ZERO).await; + future::pending::<()>().await; + }); +} + +/// Advances completed openings and drains pointer reads at the polling cadence. +#[test] +fn run_poll_interval() { + controlled(async { + let fixture = Fixture::new("manager-run-poll-interval"); + let interval = Duration::from_millis(10); + let (mut manager, _pointer_reply) = maintainer( + &fixture, + ManagerOptions { + poll_interval: interval, + .. + }, + ) + .await; + let registry = Arc::clone(manager.registry()); + let generation = fixture.world.generation().id(); + let (current, pointer) = pending(); + manager.current = Some(current); + let (task, initialize) = pending(); + manager.slots.insert(generation, opening(task)); + manager.desired = Some(generation); + + let shutdown = CancellationToken::new(); + let mut running = pin!(manager.run(shutdown.clone().cancelled_owned())); + assert!( + running.as_mut().now_or_never().is_none(), + "the run loop should await initialization" + ); + assert_matches!( + registry.observe(None).err(), + Some(ObserveError::Empty), + "a pending opening should admit no request" + ); + + let (feed, observed) = prompt_feed(); + assert!( + initialize + .send(Ok(Ok(runtime(Arc::clone(&fixture.world), 1, Some(feed))))) + .is_ok(), + "the manager should retain its opening future" + ); + assert!( + running.as_mut().now_or_never().is_none(), + "the run loop should await its next maintenance pass" + ); + assert_matches!( + registry.observe(None).err(), + Some(ObserveError::Empty), + "a completed opening should await the next maintenance pass" + ); + + tokio::time::advance(interval).await; + assert!( + running.as_mut().now_or_never().is_none(), + "the run loop should continue after promotion" + ); + let (world, _lifetime) = published(®istry, generation); + assert!( + Arc::ptr_eq(&world, &fixture.world), + "the next maintenance pass should promote the completed opening" + ); + + shutdown.cancel(); + assert!( + running.as_mut().now_or_never().is_none(), + "shutdown should wait for owned operations" + ); + assert_closed(®istry); + observed.await.expect("shutdown should stop the feed"); + assert!( + running.as_mut().now_or_never().is_none(), + "shutdown should retain the pending pointer read" + ); + assert!( + pointer.send(Ok(Ok(None))).is_ok(), + "shutdown should retain the pointer future's receiver" + ); + running.await; + }); +} + +/// Runs maintenance once a second and keeps expired directories. +#[test] +fn options_default() { + let options = ManagerOptions::default(); + + assert_eq!( + options.poll_interval, + Duration::from_secs(1), + "the default maintenance pass should run once a second" + ); + assert!( + !options.unlink, + "expired directories should stay on disk by default" + ); +} + +/// Rejects a zero manager polling interval with [`ManagerError::InvalidInterval`]. +#[tokio::test] +async fn new_zero_interval() { + let fixture = Fixture::new("manager-new-zero-interval"); + let source = RuntimeSource { + root: fixture.root(), + secret: secret(), + pool: pool().await, + feed: None, + }; + + let Err(error) = GenerationManager::new( + source, + ManagerOptions { + poll_interval: Duration::ZERO, + .. + }, + HARD, + ) else { + panic!("a zero polling interval should not construct a manager") + }; + + assert_matches!( + error.current_context(), + ManagerError::InvalidInterval, + "the refusal should name the invalid interval" + ); +} + +/// Publishes no selected generation until its opening succeeds. +#[tokio::test] +async fn open_promotes_selection() { + let fixture = Fixture::new("manager-open-promotes-selection"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let generation = fixture.world.generation().id(); + + manager.desired = Some(generation); + manager.tick(base); + + let Err(error) = registry.observe(None) else { + panic!("an unfinished opening should not publish") + }; + assert_matches!( + error, + ObserveError::Empty, + "the refusal should name the missing active generation" + ); + + settle_opening(&mut manager, generation).await; + manager.tick(after(base, Duration::from_secs(1))); + + let (world, _lifetime) = published(®istry, generation); + assert!( + !Arc::ptr_eq(&world, &fixture.world), + "the manager should publish the world its own opening produced" + ); + + drain(&mut manager, []).await; +} + +/// Publishes a successfully opened selection without a warning. +#[tokio::test] +async fn open_promotes_without_warning() { + let control = captured_warnings(|| { + tracing::warn!("warning capture control"); + tracing::error!("error capture control"); + }); + assert!(control.contains("warning capture control")); + assert!(control.contains("error capture control")); + + let fixture = Fixture::new("manager-open-promotes-without-warning"); + let replacement = fixture.variant("pending opening"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let generation = fixture.world.generation().id(); + let candidate = replacement.generation().id(); + + manager.desired = Some(generation); + let absent = captured_warnings(|| manager.tick(base)); + + assert_eq!( + absent, "", + "the manager should not promote a generation whose slot this pass creates" + ); + + settle_opening(&mut manager, generation).await; + let promotion = captured_warnings(|| manager.tick(after(base, Duration::from_secs(1)))); + + assert_eq!( + promotion, "", + "a completed opening should promote without a warning" + ); + let (_world, _lifetime) = published(®istry, generation); + + // An opening that cannot finish, which is what makes the next pass an `Opening` pass rather + // than whatever a live opening has reached by now. + let (task, _opening_reply) = pending(); + manager.slots.insert(candidate, opening(task)); + manager.desired = Some(candidate); + let unfinished = captured_warnings(|| manager.tick(after(base, Duration::from_secs(2)))); + + assert_eq!( + unfinished, "", + "the manager should not promote a candidate without an initialized runtime" + ); + let (_retained, _unchanged) = published(®istry, generation); + + drop(manager.slots.remove(&candidate)); + drain(&mut manager, []).await; +} + +/// Keeps the previous generation published while its candidate opening remains pending. +#[test] +fn promote_pending_opening() { + controlled(async { + let fixture = Fixture::new("manager-promote-pending-opening"); + let replacement = fixture.variant("pending opening"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let active = fixture.world.generation().id(); + let candidate = replacement.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + active, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(active); + manager.tick(base); + let (world, lifetime) = published(®istry, active); + + let (task, _opening_reply) = pending(); + manager.slots.insert(candidate, opening(task)); + manager.desired = Some(candidate); + manager.tick(after(base, Duration::from_secs(1))); + + let (retained, unchanged) = published(®istry, active); + assert!( + Arc::ptr_eq(&retained, &world), + "a pending candidate should keep the published world" + ); + assert_eq!( + unchanged, lifetime, + "a pending candidate should keep the published lifetime" + ); + assert_unavailable(®istry, candidate); + + drop(manager.slots.remove(&candidate)); + drain(&mut manager, [observed]).await; + }); +} + +/// Keeps the previous publication while reopening a failed selected candidate. +#[tokio::test] +async fn promote_failed_opening() { + let fixture = Fixture::new("manager-promote-failed-opening"); + let replacement = fixture.variant("failed opening"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let active = fixture.world.generation().id(); + let candidate = replacement.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + active, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(active); + manager.tick(base); + let (world, lifetime) = published(®istry, active); + + manager.slots.insert( + candidate, + opening(settled(Err(Report::new(ManagerError::Runtime)))), + ); + manager.desired = Some(candidate); + manager.tick(after(base, Duration::from_secs(1))); + + let (retained, unchanged) = published(®istry, active); + assert!( + Arc::ptr_eq(&retained, &world), + "a failed candidate should keep the published world" + ); + assert_eq!( + unchanged, lifetime, + "a failed candidate should keep the published lifetime" + ); + let slot = manager + .slots + .get(&candidate) + .expect("the selected candidate should keep its slot"); + assert_matches!( + slot.execution, + Execution::Opening { .. } | Execution::Ready(_), + "the selected candidate should reopen after its failure" + ); + + settle_opening(&mut manager, candidate).await; + drain(&mut manager, [observed]).await; +} + +/// Reuses a retained world and lifetime when reselected before expiry. +#[test] +fn promote_reselected_before_expiry() { + controlled(async { + let fixture = Fixture::new("manager-promote-reselected-before-expiry"); + let replacement = fixture.variant("reselected before expiry"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(first); + manager.tick(base); + let (world, lifetime) = published(®istry, first); + + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(after(base, Duration::from_secs(1))); + let (displacing, _published) = published(®istry, second); + assert!( + Arc::ptr_eq(&displacing, &replacement), + "the replacement should take the active selection" + ); + + manager.desired = Some(first); + manager.tick(after(base, Duration::from_secs(2))); + + let (reselected, resumed) = published(®istry, first); + assert!( + Arc::ptr_eq(&reselected, &world), + "reselection before expiry should keep the running world" + ); + assert_eq!( + resumed, lifetime, + "reselection before expiry should keep the running delta lifetime" + ); + + drain(&mut manager, [observed]).await; + }); +} + +/// Reopens an expired generation with a fresh delta lifetime. +/// +/// Its previous feed joins before reopening begins. +#[tokio::test] +async fn promote_reactivated_after_expiry() { + let fixture = Fixture::new("manager-promote-reactivated-after-expiry"); + let replacement = fixture.variant("reactivated after expiry"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + let (feed, handshake) = stalled_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(first); + manager.tick(base); + let (world, lifetime) = published(®istry, first); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + manager.tick(after(retired, HARD)); + handshake + .observed + .await + .expect("expiry should stop the retained feed"); + + manager.desired = Some(first); + manager.tick(after(retired, HARD)); + let (still_active, _lifetime) = published(®istry, second); + assert!( + Arc::ptr_eq(&still_active, &replacement), + "an unjoined feed should keep the replacement published" + ); + let slot = manager + .slots + .get(&first) + .expect("the reselected generation should keep its slot"); + assert_matches!( + slot.execution, + Execution::Joining(_), + "reactivation should wait for the feed to join" + ); + + handshake + .release + .send(()) + .expect("the feed should await release"); + handshake + .ended + .await + .expect("the feed should run to its end"); + + manager.tick(after(retired, HARD)); + settle_opening(&mut manager, first).await; + manager.tick(after(retired, HARD)); + + let (reactivated, restarted) = published(®istry, first); + assert!( + Arc::ptr_eq(&reactivated, &world), + "reactivation should reuse the retained world" + ); + assert_ne!( + restarted, lifetime, + "reactivation should publish a fresh delta lifetime" + ); + + drain(&mut manager, []).await; +} + +/// Stops an expired retained feed without removing its directory. +/// +/// Admission closes with the feed. +#[test] +fn expire_stops_retained_feed() { + controlled(async { + let fixture = Fixture::new("manager-expire-stops-retained-feed"); + let replacement = fixture.variant("expire stops retained feed"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(first); + manager.tick(base); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + let held = registry + .observe(Some(first)) + .expect("the retained generation should admit before its expiry"); + + manager.tick(after(retired, HARD)); + observed + .await + .expect("expiry should stop the retained feed without unlinking"); + + assert_unavailable(®istry, first); + assert!( + Arc::ptr_eq(held.requested().world(), &fixture.world), + "the held request should keep the retained world" + ); + assert!( + held.requested() + .world() + .layout + .position(held.requested().epoch(), NodeRowId::MIN) + .is_some(), + "the held request should still answer after cleanup" + ); + assert!( + fixture.world.generation().path().is_dir(), + "expiry without unlink should keep the generation's directory" + ); + + drain(&mut manager, []).await; + }); +} + +/// Lets another expired feed observe shutdown while one join stalls. +#[test] +fn expire_stalled_join() { + controlled(async { + let fixture = Fixture::new("manager-expire-stalled-join"); + let second_world = fixture.variant("stalled join second"); + let third_world = fixture.variant("stalled join third"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = second_world.generation().id(); + let third = third_world.generation().id(); + + let (stalled, handshake) = stalled_feed(); + let (prompt, observed) = prompt_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(stalled))), + ); + manager.desired = Some(first); + manager.tick(base); + + let first_retired = after(base, Duration::from_secs(1)); + manager.slots.insert( + second, + ready(runtime(Arc::clone(&second_world), 2, Some(prompt))), + ); + manager.desired = Some(second); + manager.tick(first_retired); + + let second_retired = after(first_retired, Duration::from_secs(1)); + manager + .slots + .insert(third, ready(runtime(Arc::clone(&third_world), 3, None))); + manager.desired = Some(third); + manager.tick(second_retired); + + manager.tick(after(second_retired, HARD)); + handshake + .observed + .await + .expect("the stalled feed should observe cancellation"); + observed + .await + .expect("the second expired feed should observe cancellation behind a stalled join"); + + manager.tick(after(second_retired, HARD)); + assert!( + !manager.slots.contains_key(&second), + "the joined expired generation should leave no slot" + ); + assert!( + manager.slots.contains_key(&first), + "the stalled expired generation should keep its slot" + ); + + handshake + .release + .send(()) + .expect("the stalled feed should await release"); + drain(&mut manager, []).await; + handshake + .ended + .await + .expect("the stalled feed should run to its end"); + }); +} + +/// Recovers an ended present feed with a fresh lifetime over its retained world. +#[tokio::test] +async fn recovery_after_present_feed_end() { + let fixture = Fixture::new("manager-recovery-after-present-feed-end"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let generation = fixture.world.generation().id(); + + let (feed, release, ended) = ending_feed(); + manager.slots.insert( + generation, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(generation); + manager.tick(base); + let (world, lifetime) = published(®istry, generation); + + release.send(()).expect("the feed should await release"); + ended.await.expect("the feed should run to its end"); + manager.tick(after(base, Duration::from_secs(1))); + + let (retained, unchanged) = published(®istry, generation); + assert!( + Arc::ptr_eq(&retained, &world), + "recovery should keep the published world" + ); + assert_eq!( + unchanged, lifetime, + "recovery should keep the published lifetime until it succeeds" + ); + + settle_opening(&mut manager, generation).await; + manager.tick(after(base, Duration::from_secs(2))); + + let (recovered, restarted) = published(®istry, generation); + assert!( + Arc::ptr_eq(&recovered, &world), + "recovery should reuse the retained world" + ); + assert_ne!( + restarted, lifetime, + "recovery should publish a fresh delta lifetime" + ); + + drain(&mut manager, []).await; +} + +/// Recovers an ended retained feed only at its next selection. +#[tokio::test] +async fn recovery_deferred_while_retained() { + let fixture = Fixture::new("manager-recovery-deferred-while-retained"); + let replacement = fixture.variant("deferred recovery"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + let (feed, release, ended) = ending_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(first); + manager.tick(base); + let (world, lifetime) = published(®istry, first); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + release.send(()).expect("the feed should await release"); + ended.await.expect("the feed should run to its end"); + manager.tick(after(retired, Duration::from_secs(1))); + + let observation = registry + .observe(Some(first)) + .expect("the retained generation should admit within its retention"); + assert!( + Arc::ptr_eq(observation.requested().world(), &world), + "a retained generation should keep its final world" + ); + assert_eq!( + observation.requested().epoch().reference(), + lifetime, + "a retained generation should keep its final publication" + ); + let slot = manager + .slots + .get(&first) + .expect("the retained generation should keep its slot"); + assert_matches!( + slot.execution, + Execution::Stopped(_), + "a retained generation should not start recovery" + ); + + manager.desired = Some(first); + manager.tick(after(retired, Duration::from_secs(2))); + settle_opening(&mut manager, first).await; + manager.tick(after(retired, Duration::from_secs(3))); + + let (recovered, restarted) = published(®istry, first); + assert!( + Arc::ptr_eq(&recovered, &world), + "reselection should recover over the retained world" + ); + assert_ne!( + restarted, lifetime, + "reselection should recover into a fresh delta lifetime" + ); + + drain(&mut manager, []).await; +} + +/// Stops a feedless runtime while retaining its world. +/// +/// The next maintenance pass reaches the stable stopped state regardless of prior promotion. +#[test] +fn joining_feedless() { + let fixture = Fixture::new("manager-joining-feedless"); + let generation = fixture.world.generation().id(); + + for promoted in [false, true] { + let mut slot = ready(runtime(Arc::clone(&fixture.world), 1, None)); + if promoted { + slot.promote(); + } + slot.stop(); + assert_matches!(slot.execution, Execution::Joining(_)); + + slot.tick(generation); + assert_matches!( + slot.execution, + Execution::Stopped(Some(ref world)) if Arc::ptr_eq(world, &fixture.world), + "a feedless runtime should stop while retaining its world" + ); + slot.tick(generation); + assert_matches!(slot.execution, Execution::Stopped(Some(_))); + } +} + +/// Keeps a runtime joining until its feed worker ends. +/// +/// The next maintenance pass then retains the world and reaches the stopped state. +#[test] +fn joining_pending_feed() { + controlled(async { + let fixture = Fixture::new("manager-joining-pending-feed"); + let generation = fixture.world.generation().id(); + let (feed, handshake) = stalled_feed(); + let mut slot = ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))); + slot.promote(); + slot.stop(); + handshake.observed.await.expect("should observe shutdown"); + + slot.tick(generation); + assert_matches!(slot.execution, Execution::Joining(_)); + + handshake + .release + .send(()) + .expect("should retain the worker"); + handshake.ended.await.expect("should finish the worker"); + slot.tick(generation); + assert_matches!( + slot.execution, + Execution::Stopped(Some(ref world)) if Arc::ptr_eq(world, &fixture.world), + "a completed feed should stop while retaining its world" + ); + }); +} + +/// Preserves a feedless generation's lifetime across maintenance passes. +#[test] +fn recovery_absent_without_feed() { + controlled(async { + let fixture = Fixture::new("manager-recovery-absent-without-feed"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let generation = fixture.world.generation().id(); + + manager.slots.insert( + generation, + ready(runtime(Arc::clone(&fixture.world), 1, None)), + ); + manager.desired = Some(generation); + manager.tick(base); + let (world, lifetime) = published(®istry, generation); + + manager.tick(after(base, Duration::from_secs(1))); + manager.tick(after(base, Duration::from_secs(2))); + + let (unchanged, held) = published(®istry, generation); + assert!( + Arc::ptr_eq(&unchanged, &world), + "a static generation should keep its published world" + ); + assert_eq!( + held, lifetime, + "a static generation should keep its published lifetime" + ); + let slot = manager + .slots + .get(&generation) + .expect("the static generation should keep its slot"); + assert_matches!( + slot.execution, + Execution::Running(_), + "a static generation should stay healthy" + ); + + drain(&mut manager, []).await; + }); +} + +/// Wakes shutdown only when the stopping feed completes, not when it observes the stop request. +#[test] +fn shutdown_join_wakeup() { + controlled(async { + let fixture = Fixture::new("manager-shutdown-join-wakeup"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + manager.current = None; + let (feed, handshake) = stalled_feed(); + manager.slots.insert( + fixture.world.generation().id(), + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + + let wake = Arc::new(WakeFlag(AtomicBool::new(false))); + let waker = Waker::from(Arc::clone(&wake)); + let mut context = Context::from_waker(&waker); + let mut shutdown = pin!(manager.shutdown()); + assert_matches!(shutdown.as_mut().poll(&mut context), Poll::Pending); + handshake.observed.await.expect("should observe shutdown"); + assert!(!wake.0.load(Ordering::Relaxed)); + + handshake + .release + .send(()) + .expect("should retain the worker"); + handshake.ended.await.expect("should finish the worker"); + assert!( + wake.0.load(Ordering::Relaxed), + "the joined worker should wake the suspended shutdown" + ); + assert_matches!(shutdown.as_mut().poll(&mut context), Poll::Ready(())); + }); +} + +/// Suspends shutdown on a current-pointer read and wakes it when the read completes. +#[tokio::test] +async fn shutdown_pointer_wakeup() { + let fixture = Fixture::new("manager-shutdown-pointer-wakeup"); + let (mut manager, reply) = maintainer(&fixture, ManagerOptions::default()).await; + assert_wakeup(manager.shutdown(), || { + assert!( + reply.send(Ok(Ok(None))).is_ok(), + "should retain the pointer read" + ); + }); +} + +/// Suspends shutdown during generation opening and wakes it on completion. +#[tokio::test] +async fn shutdown_opening_wakeup() { + let fixture = Fixture::new("manager-shutdown-opening-wakeup"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + manager.current = None; + let (task, reply) = pending(); + manager + .slots + .insert(fixture.world.generation().id(), opening(task)); + let runtime = runtime(Arc::clone(&fixture.world), 1, None); + + assert_wakeup(manager.shutdown(), || { + assert!( + reply.send(Ok(Ok(runtime))).is_ok(), + "should retain the opening" + ); + }); +} + +/// Suspends shutdown on generation removal and wakes it when removal completes. +#[tokio::test] +async fn shutdown_removal_wakeup() { + let fixture = Fixture::new("manager-shutdown-removal-wakeup"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + manager.current = None; + let (task, reply) = pending(); + manager.slots.insert( + fixture.world.generation().id(), + RuntimeSlot { + registration: Registration::Expired, + execution: Execution::Removing(task), + }, + ); + + assert_wakeup(manager.shutdown(), || { + assert!(reply.send(Ok(Ok(()))).is_ok(), "should retain the removal"); + }); +} + +/// Stops every initialized feed and closes admission before any join returns. +#[test] +fn shutdown_signals_before_join() { + controlled(async { + let fixture = Fixture::new("manager-shutdown-signals-before-join"); + let replacement = fixture.variant("shutdown signals"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + let (retained_feed, retained) = stalled_feed(); + let (active_feed, active) = stalled_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(retained_feed))), + ); + manager.desired = Some(first); + manager.tick(base); + + manager.slots.insert( + second, + ready(runtime(Arc::clone(&replacement), 2, Some(active_feed))), + ); + manager.desired = Some(second); + manager.tick(after(base, Duration::from_secs(1))); + + manager.current = None; + assert!( + manager.shutdown().now_or_never().is_none(), + "the wait should not finish while both feeds are joining" + ); + assert_closed(®istry); + retained + .observed + .await + .expect("the retained feed should observe shutdown"); + active + .observed + .await + .expect("the active feed should observe shutdown"); + + retained + .release + .send(()) + .expect("the retained feed should await release"); + active + .release + .send(()) + .expect("the active feed should await release"); + manager.shutdown().await; + + retained + .ended + .await + .expect("the retained feed should run to its end"); + active + .ended + .await + .expect("the active feed should run to its end"); + }); +} + +/// Retains an unjoined feed result after cancellation interrupts shutdown waiting. +#[test] +fn shutdown_cancelled_wait() { + controlled(async { + let fixture = Fixture::new("manager-shutdown-cancelled-wait"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let base = Instant::now(); + let generation = fixture.world.generation().id(); + + let (feed, handshake) = stalled_feed(); + manager.slots.insert( + generation, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(generation); + manager.tick(base); + + manager.current = None; + assert!( + manager.shutdown().now_or_never().is_none(), + "the first wait should not finish while the feed is joining" + ); + handshake + .observed + .await + .expect("the feed should observe shutdown"); + assert!( + manager.shutdown().now_or_never().is_none(), + "the retained result should keep the resumed wait pending" + ); + + handshake + .release + .send(()) + .expect("the feed should await release"); + manager.shutdown().await; + handshake + .ended + .await + .expect("the retained result should reach the resumed wait"); + }); +} + +/// Stops and joins a superseded initialized candidate instead of publishing it. +#[test] +fn reconcile_superseded_candidate() { + controlled(async { + let fixture = Fixture::new("manager-reconcile-superseded-candidate"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let generation = fixture.world.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + generation, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.tick(base); + + observed + .await + .expect("the superseded candidate should observe cancellation"); + let Err(error) = registry.observe(None) else { + panic!("an unselected candidate should not publish") + }; + assert_matches!( + error, + ObserveError::Empty, + "the refusal should name the missing active generation" + ); + + manager.tick(after(base, Duration::from_secs(1))); + assert!( + !manager.slots.contains_key(&generation), + "the joined candidate should leave no slot" + ); + + drain(&mut manager, []).await; + }); +} + +/// Stops and joins an opening that completes unselected during shutdown. +#[test] +fn shutdown_superseded_opening() { + controlled(async { + let fixture = Fixture::new("manager-shutdown-superseded-opening"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let generation = fixture.world.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + generation, + opening(settled(Ok(runtime( + Arc::clone(&fixture.world), + 1, + Some(feed), + )))), + ); + + drain(&mut manager, [observed]).await; + assert_closed(®istry); + }); +} + +/// Reopens a selected expired generation instead of unlinking its directory. +#[tokio::test] +async fn expire_precedes_unlink() { + let fixture = Fixture::new("manager-expire-precedes-unlink"); + let replacement = fixture.variant("expiry precedes unlink"); + let (mut manager, _pointer_reply) = + maintainer(&fixture, ManagerOptions { unlink: true, .. }).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(first); + manager.tick(base); + let (world, lifetime) = published(®istry, first); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + manager.tick(after(retired, HARD)); + observed.await.expect("expiry should stop the feed"); + + manager.desired = Some(first); + manager.tick(after(retired, HARD)); + let slot = manager + .slots + .get(&first) + .expect("the reselected generation should keep its slot"); + assert_matches!( + slot.execution, + Execution::Opening { .. } | Execution::Ready(_), + "selection should reopen the expired generation" + ); + assert!( + world.generation().path().is_dir(), + "selection should keep the directory an unstarted unlink would remove" + ); + + settle_opening(&mut manager, first).await; + manager.tick(after(retired, HARD)); + + let (reopened, restarted) = published(®istry, first); + assert!( + Arc::ptr_eq(&reopened, &world), + "the reselected generation should serve its retained world" + ); + assert_ne!( + restarted, lifetime, + "the reselected generation should publish a fresh delta lifetime" + ); + + drain(&mut manager, []).await; +} + +/// Finishes a started removal before reopening the generation. +#[tokio::test] +async fn open_after_started_removal() { + let fixture = Fixture::new("manager-open-after-started-removal"); + let replacement = fixture.variant("open after started removal"); + let (mut manager, _pointer_reply) = + maintainer(&fixture, ManagerOptions { unlink: true, .. }).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + manager + .slots + .insert(first, ready(runtime(Arc::clone(&fixture.world), 1, None))); + manager.desired = Some(first); + manager.tick(base); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + manager.tick(after(retired, HARD)); + let (removal, release) = pending(); + let slot = manager + .slots + .get_mut(&first) + .expect("the expired generation should keep its slot"); + assert_matches!( + slot.execution, + Execution::Stopped(_), + "expiry should join the generation before its removal" + ); + slot.execution = Execution::Removing(removal); + + manager.desired = Some(first); + manager.tick(after(retired, HARD)); + let slot = manager + .slots + .get(&first) + .expect("the removing generation should keep its slot"); + assert_matches!( + slot.execution, + Execution::Removing(_), + "a started removal should finish before a fresh opening" + ); + + assert!( + release.send(Ok(Ok(()))).is_ok(), + "should retain the removal" + ); + manager.tick(after(retired, HARD)); + settle_opening(&mut manager, first).await; + manager.tick(after(retired, HARD)); + + let (reopened, _lifetime) = published(®istry, first); + assert!( + !Arc::ptr_eq(&reopened, &fixture.world), + "the opening after a removal should read the generation from disk" + ); + + drain(&mut manager, []).await; +} + +/// Removes an expired directory only after joining, while held requests remain readable. +#[tokio::test] +async fn remove_expired_directory() { + let fixture = Fixture::new("manager-remove-expired-directory"); + let replacement = fixture.variant("remove expired directory"); + let (mut manager, _pointer_reply) = + maintainer(&fixture, ManagerOptions { unlink: true, .. }).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + let directory = fixture.world.generation().path().to_owned(); + + let (feed, observed) = prompt_feed(); + manager.slots.insert( + first, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(first); + manager.tick(base); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + let held = registry + .observe(Some(first)) + .expect("the retained generation should admit before its expiry"); + + manager.tick(after(retired, HARD)); + observed.await.expect("expiry should stop the feed"); + assert!( + directory.is_dir(), + "removal should follow the feed's joining" + ); + + manager.tick(after(retired, HARD)); + settle_removal(&mut manager, first).await; + manager.tick(after(retired, HARD)); + + assert!( + !directory.is_dir(), + "removal should not wait for the request-held world" + ); + assert!( + !manager.slots.contains_key(&first), + "the removed generation should leave no slot" + ); + assert!( + held.requested() + .world() + .layout + .position(held.requested().epoch(), NodeRowId::MIN) + .is_some(), + "the held request should answer after the removal deletes its directory" + ); + + drain(&mut manager, []).await; +} + +/// Reopens from disk instead of reusing the world after generation removal fails. +#[tokio::test] +async fn remove_failure_reopens_from_disk() { + let fixture = Fixture::new("manager-remove-failure-reopens-from-disk"); + let replacement = fixture.variant("remove failure reopens"); + let (mut manager, _pointer_reply) = + maintainer(&fixture, ManagerOptions { unlink: true, .. }).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let first = fixture.world.generation().id(); + let second = replacement.generation().id(); + + manager + .slots + .insert(first, ready(runtime(Arc::clone(&fixture.world), 1, None))); + manager.desired = Some(first); + manager.tick(base); + + let retired = after(base, Duration::from_secs(1)); + manager + .slots + .insert(second, ready(runtime(Arc::clone(&replacement), 2, None))); + manager.desired = Some(second); + manager.tick(retired); + + manager.tick(after(retired, HARD)); + let slot = manager + .slots + .get_mut(&first) + .expect("the expired generation should keep its slot"); + slot.execution = Execution::Removing(settled(Err(Report::new(ManagerError::Remove)))); + + manager.desired = Some(first); + manager.tick(after(retired, HARD)); + settle_opening(&mut manager, first).await; + manager.tick(after(retired, HARD)); + + let (reopened, _lifetime) = published(®istry, first); + assert!( + !Arc::ptr_eq(&reopened, &fixture.world), + "a failed removal should force the generation's validation from disk" + ); + + drain(&mut manager, []).await; +} + +/// Completes the real opening before substituting a failed recovery result. +/// +/// # Panics +/// +/// Panics if `generation` has no slot, its execution is neither opening nor ready, or the active +/// opening fails. +async fn fail_started_opening(manager: &mut GenerationManager, generation: GenerationId) { + let slot = manager + .slots + .get_mut(&generation) + .expect("the generation should hold a slot"); + let world = match mem::replace(&mut slot.execution, Execution::Removed) { + Execution::Opening { world, task } => { + drop(task.await.expect("the manager's recovery should open")); + world + } + Execution::Ready(runtime) => Some(Arc::clone(runtime.world())), + execution @ (Execution::Running(_) + | Execution::Joining(_) + | Execution::Stopped(_) + | Execution::Removing(_) + | Execution::Removed) => panic!("should hold a started recovery, found {execution:?}"), + }; + + slot.execution = Execution::Opening { + world, + task: settled(Err(Report::new(ManagerError::Runtime))), + }; +} + +/// Retains an ended feed's publication until a later recovery attempt succeeds. +#[tokio::test] +async fn recovery_failed_attempt() { + let fixture = Fixture::new("manager-recovery-failed-attempt"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let base = Instant::now(); + let generation = fixture.world.generation().id(); + + let (feed, release, ended) = ending_feed(); + manager.slots.insert( + generation, + ready(runtime(Arc::clone(&fixture.world), 1, Some(feed))), + ); + manager.desired = Some(generation); + manager.tick(base); + let (world, lifetime) = published(®istry, generation); + + release.send(()).expect("the feed should await release"); + ended.await.expect("the feed should run to its end"); + manager.tick(after(base, Duration::from_secs(1))); + fail_started_opening(&mut manager, generation).await; + manager.tick(after(base, Duration::from_secs(2))); + + let (retained, unchanged) = published(®istry, generation); + assert!( + Arc::ptr_eq(&retained, &world), + "a failed recovery should keep the published world" + ); + assert_eq!( + unchanged, lifetime, + "a failed recovery should keep the published lifetime" + ); + + settle_opening(&mut manager, generation).await; + manager.tick(after(base, Duration::from_secs(3))); + + let (recovered, restarted) = published(®istry, generation); + assert!( + Arc::ptr_eq(&recovered, &world), + "a later recovery should reuse the retained world" + ); + assert_ne!( + restarted, lifetime, + "a later recovery should publish a fresh delta lifetime" + ); + + drain(&mut manager, []).await; +} + +/// Resumes shutdown after cancellation interrupts its wait and the opening finishes. +#[test] +fn shutdown_pending_opening() { + controlled(async { + let fixture = Fixture::new("manager-shutdown-pending-opening"); + let (mut manager, _pointer_reply) = maintainer(&fixture, ManagerOptions::default()).await; + let registry = Arc::clone(manager.registry()); + let generation = fixture.world.generation().id(); + + let (task, delivery) = pending(); + manager.slots.insert(generation, opening(task)); + manager.desired = Some(generation); + manager.current = None; + + assert!( + manager.shutdown().now_or_never().is_none(), + "the wait should not finish while the opening is pending" + ); + assert_closed(®istry); + + let (feed, observed) = prompt_feed(); + assert!( + delivery + .send(Ok(Ok(runtime(Arc::clone(&fixture.world), 1, Some(feed))))) + .is_ok(), + "the pending opening should still await its runtime" + ); + + manager.shutdown().await; + observed + .await + .expect("the delivered runtime should stop and join its feed"); + assert_closed(®istry); + }); +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/mod.rs b/libs/@local/graph/atlas/src/serve/runtime/mod.rs new file mode 100644 index 00000000000..548ed3aa90a --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/mod.rs @@ -0,0 +1,275 @@ +//! Per-generation execution ownership. +//! +//! [`Runtime`] keeps feed shutdown independent of request lifetimes. Retained worlds and +//! publications remain readable after its background work finishes. Feed recovery belongs to the +//! [`manager::GenerationManager`]. A runtime itself reports one runner's terminal result. + +use alloc::sync::Arc; +use core::{ + error::Error, + fmt, + task::{Context, Poll, Waker, ready}, +}; + +use error_stack::{Report, ResultExt as _}; +use futures::FutureExt as _; +use hash_graph_postgres_store::store::PostgresStorePool; +use rand::TryCryptoRng; +use tokio::task::JoinHandle; +use tokio_util::sync::CancellationToken; + +use super::{ + delta::{Delta, DeltaReader, DeltaTask, DeltaTaskError, DeltaTaskOptions, EmbeddingWorkflow}, + secret::ServeSecret, + world::World, +}; +use crate::{device::PhysicalDevice, file::generation::Generation}; + +pub(crate) mod manager; +pub(crate) mod registry; +#[cfg(test)] +mod tests; + +/// Execution settings for an enabled generation feed. +#[derive(Clone)] +pub(crate) struct FeedOptions { + /// How the feed polls the store and applies what it reads. + pub task: DeltaTaskOptions, + /// The device the feed's embedding work runs on. + pub device: PhysicalDevice, + /// The workflow that produces embeddings for new entities, absent where none is configured. + pub workflow: Option>, +} + +/// A failure opening or joining a generation's runtime. +#[derive(Debug)] +pub(crate) enum RuntimeError { + /// The generation's serving artifacts could not be opened. + World, + /// Drawing the [delta lifetime tag](super::delta::DeltaId) failed. + Entropy, + /// Opening or running the generation feed failed. + Feed, + /// Joining the feed runner failed. + Join, +} + +impl fmt::Display for RuntimeError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::World => fmt.write_str("could not open the generation's serving artifacts"), + Self::Entropy => fmt.write_str("could not initialize the delta identifier"), + Self::Feed => fmt.write_str("the generation feed failed"), + Self::Join => fmt.write_str("could not join the generation feed"), + } + } +} + +impl Error for RuntimeError {} + +/// A running feed and the token that asks it to stop. +/// +/// The handle is retained until a poll reads its result, letting a runtime report why a feed +/// ended rather than only that it did. +struct Feed { + shutdown: CancellationToken, + task: JoinHandle>>, +} + +/// The outcome of checking for an unjoined feed runner. +#[derive(Debug)] +enum FeedState { + /// No runner exists, or an earlier join consumed its result. + Absent, + /// A runner exists and its result is not available to this probe yet. + Running, + /// The runner ended and this probe consumed its result. + Finished, +} + +/// The opened world, publication reader and shutdown authority for one generation. +/// +/// Dropping the runtime requests graceful shutdown. [`Self::stop`] requests it without waiting, +/// and [`Self::poll_join`] then reads the runner's result. Request-held worlds and readers remain +/// valid in either case. Without a feed, the initial publication remains static and can serve for +/// as long as the manager keeps the generation admitted, including indefinitely while it remains +/// current. Shutdown preserves the generation's directory for subsequent opening or explicit +/// retirement. +#[must_use = "dropping the runtime requests shutdown"] +pub(crate) struct Runtime { + world: Arc, + + reader: DeltaReader, + feed: Option, +} + +impl Runtime { + /// Opens a world and starts its configured feed after initialization succeeds. + /// + /// `None` disables the feed. A generation without temporal axes also has a static reader. + /// + /// # Errors + /// + /// Returns [`RuntimeError`] for world opening, entropy or feed initialization failures. + /// + /// # Panics + /// + /// Panics outside a Tokio runtime when a feed needs to start. Feed initialization also panics + /// if it reaches placement-channel construction with a capacity above + /// [`tokio::sync::Semaphore::MAX_PERMITS`], as [`DeltaTask::open`] documents. + #[tracing::instrument(skip_all, err(Debug), fields(generation = %generation.id()))] + pub(crate) fn open( + generation: Generation, + secret: &ServeSecret, + pool: Arc, + rng: impl TryCryptoRng, + feed: Option, + ) -> Result> { + let world = World::open(generation, secret).change_context(RuntimeError::World)?; + Self::start(Arc::new(world), pool, rng, feed) + } + + /// Starts a new [delta lifetime](super::delta::DeltaId) over an already-open world. + /// + /// The new lifetime samples its tag from `rng` without checking whether it matches an earlier + /// [`DeltaId`](super::delta::DeltaId). + /// + /// `None` disables the feed. A generation without temporal axes also has a static reader. + /// + /// # Errors + /// + /// Returns [`RuntimeError`] for entropy or feed initialization failures. + /// + /// # Panics + /// + /// Panics outside a Tokio runtime when a feed needs to start. Feed initialization also panics + /// if it reaches placement-channel construction with a capacity above + /// [`tokio::sync::Semaphore::MAX_PERMITS`], as [`DeltaTask::open`] documents. + #[tracing::instrument(skip_all, err(Debug), fields(generation = %world.generation().id()))] + pub(crate) fn start( + world: Arc, + pool: Arc, + rng: impl TryCryptoRng, + feed: Option, + ) -> Result> { + let delta = Delta::new(Arc::clone(&world), rng).change_context(RuntimeError::Entropy)?; + let (reader, task) = match feed { + Some(FeedOptions { + task, + device, + workflow, + }) => DeltaTask::open(delta, pool, task, device, workflow) + .change_context(RuntimeError::Feed)?, + None => (DeltaReader::from(delta), None), + }; + + let feed = task.map(|task| { + let cancel = CancellationToken::new(); + let task = tokio::spawn(task.run(cancel.clone().cancelled_owned())); + + Feed { + shutdown: cancel, + task, + } + }); + + Ok(Self { + world, + reader, + feed, + }) + } + + /// Borrows the opened world, which outlives the runtime through its own reference count. + pub(crate) const fn world(&self) -> &Arc { + &self.world + } + + /// Borrows the publication reader requests observe this generation through. + const fn reader(&self) -> &DeltaReader { + &self.reader + } + + /// Requests graceful shutdown without waiting for in-flight work. + fn stop(&self) { + if let Some(feed) = &self.feed { + feed.shutdown.cancel(); + } + } + + /// Probes the runner without waiting, retaining the handle until a poll returns its result. + /// + /// A finished runner can still report [`FeedState::Running`], retaining the handle for a later + /// probe. Repeated probes can defer the result until a poll has cooperative budget to read it. + /// + /// # Errors + /// + /// Returns [`RuntimeError`] for a feed failure or a failed join. + fn try_join(&mut self) -> Result> { + let Some(feed) = self.feed.as_mut() else { + return Ok(FeedState::Absent); + }; + + if !feed.task.is_finished() { + return Ok(FeedState::Running); + } + + // JoinHandle polls consume cooperative budget before reading the output, even after + // is_finished returns true. This probe uses a no-op waker, which discards notifications. + // A later pass must therefore poll again for any deferred result. + let mut probe = Context::from_waker(Waker::noop()); + let Poll::Ready(result) = feed.task.poll_unpin(&mut probe) else { + return Ok(FeedState::Running); + }; + + self.feed = None; + result + .change_context(RuntimeError::Join) + .and_then(|result| result.change_context(RuntimeError::Feed)) + .map(|()| FeedState::Finished) + } + + /// Polls the runner to completion, or returns `None` without an unjoined runner. + /// + /// A pending poll retains the join handle for later polling. Polling a ready result consumes + /// the handle, and a later poll reports `None`. + /// + /// # Errors + /// + /// Returns [`RuntimeError`] for a feed failure or a failed join. + fn poll_join( + &mut self, + context: &mut Context<'_>, + ) -> Poll>>> { + let Some(feed) = self.feed.as_mut() else { + return Poll::Ready(None); + }; + + let result = ready!(feed.task.poll_unpin(context)); + self.feed = None; + Poll::Ready(Some( + result + .change_context(RuntimeError::Join) + .and_then(|result| result.change_context(RuntimeError::Feed)), + )) + } +} + +impl Drop for Runtime { + /// Requests shutdown for an unjoined feed and records the unjoined runtime drop. + /// + /// Reaching here with a feed still owned means nobody joined it. No owner receives its result. + /// The warning names the generation it belonged to. This path asks the task to stop but + /// never aborts it. Dropping Tokio's [`JoinHandle`] detaches the runner, which keeps the + /// store pool and generation files alive until it finishes in the background. + fn drop(&mut self) { + if self.feed.is_some() { + tracing::warn!( + generation = %self.world.generation().id(), + "Drop generation runtime without joining the feed" + ); + + self.stop(); + } + } +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/registry/mod.rs b/libs/@local/graph/atlas/src/serve/runtime/registry/mod.rs new file mode 100644 index 00000000000..8ae6c3f38e3 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/registry/mod.rs @@ -0,0 +1,302 @@ +//! Coherent generation selection at request admission. +//! +//! [`UniverseRegistry`] selects the active and requested generations under one read lock. +//! Observed [`World`] and [`Epoch`] values retain their publications independently of subsequent +//! promotion or retirement. + +use alloc::sync::Arc; +use core::{error::Error, fmt, time::Duration}; +use std::{sync::nonpoison::RwLock, time::Instant}; + +use hashql_core::collections::FastHashMap; + +use super::Runtime; +use crate::{ + file::generation::GenerationId, + serve::{ + delta::{DeltaReader, epoch::Epoch}, + world::World, + }, +}; + +#[cfg(test)] +mod tests; + +/// A world paired with its captured delta publication. +pub(crate) struct Universe { + world: Arc, + epoch: Epoch, +} + +impl Universe { + /// Borrows the opened world this observation was admitted against. + pub(crate) const fn world(&self) -> &Arc { + &self.world + } + + /// Borrows the captured publication every read of this observation uses. + pub(crate) const fn epoch(&self) -> &Epoch { + &self.epoch + } + + /// Duplicates the handles onto the same world and the same captured publication. + fn fork(&self) -> Self { + Self { + world: Arc::clone(&self.world), + epoch: self.epoch.fork(), + } + } +} + +/// The present and requested [`Universe`]s for one request. +/// +/// Matching generations share one observed [`Epoch`]. An observation remains valid after +/// promotion, expiry or registry closure. +pub(crate) struct Observation { + admitted_at: Instant, + + present: Universe, + requested: Universe, +} + +impl Observation { + /// Returns the request admission time used to measure cache ages. + pub(crate) const fn admitted_at(&self) -> Instant { + self.admitted_at + } + + /// Returns the active generation selected under the registry read lock. + /// + /// The observation records its admission timestamp before selecting this universe, which later + /// promotion cannot change. + pub(crate) const fn present(&self) -> &Universe { + &self.present + } + + /// Returns the generation this request reads, the selected active one unless it named another. + pub(crate) const fn requested(&self) -> &Universe { + &self.requested + } +} + +/// A generation selection that cannot admit a new request. +#[derive(Debug)] +pub(crate) enum ObserveError { + /// The process has closed request admission. + Closed, + /// The process has not opened an active generation. + Empty, + /// The requested generation is absent or its retention interval has expired. + Unavailable(GenerationId), +} + +impl fmt::Display for ObserveError { + fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Closed => fmt.write_str("generation admission has closed"), + Self::Empty => fmt.write_str("no active generation is available"), + Self::Unavailable(generation) => write!(fmt, "generation {generation} is unavailable"), + } + } +} + +impl Error for ObserveError {} + +/// The read handles of one opened [`Runtime`]. +pub(super) struct Observer { + world: Arc, + delta: DeltaReader, +} + +impl Observer { + /// Returns the generation these handles read. + fn id(&self) -> GenerationId { + self.world.generation().id() + } + + /// Captures the publication current at this moment, as an admitted universe. + fn observe(&self) -> Universe { + Universe { + world: Arc::clone(&self.world), + epoch: self.delta.load(), + } + } +} + +impl From<&Runtime> for Observer { + /// Clones the two handles from which later requests capture universes. + /// + /// The world remains fixed while the delta reader may receive newer immutable publications. + /// Each call to [`Observer::observe`] captures one of those publications. An already returned + /// [`Universe`] never changes. + fn from(runtime: &Runtime) -> Self { + Self { + world: Arc::clone(runtime.world()), + delta: runtime.reader().clone(), + } + } +} + +/// An observer kept readable for a while after its generation stopped being the active one. +/// +/// Existing observations retain their generation independently of registry retirement. The +/// retention interval controls how long new requests may select a replaced generation after its +/// replacement is promoted. +struct RetiredObserver { + observer: Observer, + + retired_at: Instant, +} + +impl RetiredObserver { + /// Returns whether retention has elapsed at `now`. + /// + /// The generation no longer admits requests after retention elapses. + fn is_expired(&self, now: Instant, hard: Duration) -> bool { + now.saturating_duration_since(self.retired_at) >= hard + } +} + +/// The generations a request may be admitted to. +#[derive(Default)] +struct Observatory { + /// The generation the process serves by default. + active: Option, + /// Replaced generations still inside their retention interval. + retained: FastHashMap, +} + +/// Read-side generation selection with time-bounded access to retired generations. +/// +/// The retention interval starts at promotion of a replacement. Request admission checks that +/// interval independently of background cleanup. +pub(crate) struct UniverseRegistry { + state: RwLock>, + hard: Duration, +} + +impl UniverseRegistry { + /// Builds an open registry retaining replaced generations for `hard`. + pub(super) fn new(hard: Duration) -> Self { + Self { + state: RwLock::new(Some(Observatory::default())), + hard, + } + } + + /// Observes the active generation and the requested generation together. + /// + /// `None` selects the active generation. This method records admission time before the registry + /// read. A promotion between those operations can determine which active generation becomes + /// [`Observation::present`], while the earlier instant remains the observation's timestamp. + /// + /// # Errors + /// + /// Returns [`ObserveError`] for closed admission or an unavailable generation. + pub(crate) fn observe( + &self, + requested: Option, + ) -> Result { + let admitted_at = Instant::now(); + self.observe_at(requested, admitted_at) + } + + /// Observes at a caller-supplied admission time. + /// + /// This method reads the present universe first and forks it for a request naming the active + /// generation. Both halves of that observation share one captured publication. Selection and + /// capture occur under one registry read lock. The method releases that lock before returning. + /// + /// # Errors + /// + /// Returns [`ObserveError`]. Closed admission is refused first, then an empty registry, and + /// a named generation is refused last, when it is absent or past its retention. + fn observe_at( + &self, + requested: Option, + admitted_at: Instant, + ) -> Result { + let state = self.state.read(); + let Some(registry) = &*state else { + return Err(ObserveError::Closed); + }; + + let active = registry.active.as_ref().ok_or(ObserveError::Empty)?; + let requested = requested.unwrap_or_else(|| active.id()); + + let active_universe = active.observe(); + let selected = if requested == active.id() { + active_universe.fork() + } else { + let retained = registry + .retained + .get(&requested) + .filter(|retained| !retained.is_expired(admitted_at, self.hard)) + .ok_or(ObserveError::Unavailable(requested))?; + + retained.observer.observe() + }; + drop(state); + + Ok(Observation { + admitted_at, + present: active_universe, + requested: selected, + }) + } + + /// Publishes an opened runtime's observer and starts the previous generation's retention. + /// + /// Replacing the active generation's delta lifetime removes its previous observer from new + /// admission without retaining that observer under the same generation ID. Universes already + /// captured from the replaced observer remain valid. + /// + /// # Errors + /// + /// Returns the supplied observer unchanged when admission has closed. + pub(super) fn promote(&self, observer: Observer, now: Instant) -> Result<(), Observer> { + let mut state = self.state.write(); + let Some(registry) = &mut *state else { + return Err(observer); + }; + + let id = observer.id(); + registry.retained.remove(&id); + + if let Some(previous) = registry.active.replace(observer) + && previous.id() != id + { + registry.retained.insert( + previous.id(), + RetiredObserver { + observer: previous, + retired_at: now, + }, + ); + } + + drop(state); + Ok(()) + } + + /// Removes expired observers and appends their generation IDs to `expired`. + pub(super) fn expire(&self, now: Instant, expired: &mut Vec) { + let mut state = self.state.write(); + let Some(registry) = &mut *state else { + return; + }; + + expired.extend( + registry + .retained + .extract_if(|_id, retained| retained.is_expired(now, self.hard)) + .map(|(id, _retained)| id), + ); + drop(state); + } + + /// Closes admission while preserving already-observed publications. + pub(super) fn close(&self) { + *self.state.write() = None; + } +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/registry/tests.rs b/libs/@local/graph/atlas/src/serve/runtime/registry/tests.rs new file mode 100644 index 00000000000..e20b72bc9a3 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/registry/tests.rs @@ -0,0 +1,607 @@ +//! Coherent selection, retention boundaries and the lifetime of a held observation. +//! +//! Supplying each case its own admission instant keeps every deadline exact and every case free +//! of sleeping. One case calls [`UniverseRegistry::observe`] instead, and bounds the timestamp it +//! samples. + +use alloc::sync::Arc; +use core::{assert_matches, time::Duration}; +use std::{fs, time::Instant}; + +use hashql_core::id::Id as _; +use rand::{SeedableRng as _, rngs::StdRng}; + +use super::{Observation, ObserveError, Observer, Runtime, Universe, UniverseRegistry}; +use crate::{ + file::{generation::GenerationId, repository::Artifact as _, salt::artifact}, + identity::NodeRowId, + serve::{ + delta::{Delta, DeltaReader, DeltaReference}, + tests::fixture::{TamperFixture, secret}, + world::World, + }, +}; + +/// The retention interval every case measures its deadlines against. +const HARD: Duration = Duration::from_secs(60); + +/// One published generation and its opened world. +struct Fixture { + files: TamperFixture, + world: Arc, +} + +impl Fixture { + /// Publishes a synthetic generation under `name` and opens its world. + /// + /// # Panics + /// + /// Panics if publishing the synthetic generation fails or its world does not open. + fn new(name: &str) -> Self { + let files = TamperFixture::publish(name); + let world = Arc::new( + World::open(files.generation().clone(), &secret()) + .expect("the synthetic generation should open"), + ); + + Self { files, world } + } + + /// Publishes a generation whose unused representations placeholder holds `marker`. + /// + /// # Panics + /// + /// Panics on generation tampering, republication, or world opening failure, or if `marker` + /// leaves the generation identity unchanged. + #[track_caller] + fn variant(&self, marker: &str) -> Arc { + let generation = self.files.tamper(&artifact::Representations::NAME, |path| { + fs::remove_file(path).expect("the staged placeholder should be removable"); + fs::write(path, marker).expect("the variant placeholder should write"); + }); + let world = + Arc::new(World::open(generation, &secret()).expect("the variant world should open")); + assert_ne!( + world.generation().id(), + self.world.generation().id(), + "the variant should carry its own generation identity" + ); + + world + } +} + +/// Returns a generation identity no fixture publishes. +/// +/// # Panics +/// +/// Panics if the fixed hexadecimal string does not parse as a generation identity. +fn absent() -> GenerationId { + "ab".repeat(32) + .parse() + .expect("64 hexadecimal digits should name a generation") +} + +/// Returns the instant `offset` after `base`. +/// +/// # Panics +/// +/// Panics if adding `offset` exceeds the range of [`Instant`]. +#[track_caller] +fn after(base: Instant, offset: Duration) -> Instant { + base.checked_add(offset) + .expect("the offset should fit the instant's range") +} + +/// Builds an observer over `world` with a delta identity drawn from `seed`. +fn observer(world: Arc, seed: u64) -> Observer { + let delta = Delta::new(Arc::clone(&world), StdRng::seed_from_u64(seed)) + .expect("the seeded RNG should allocate a delta identity"); + let runtime = Runtime { + world, + reader: DeltaReader::from(delta), + feed: None, + }; + + Observer::from(&runtime) +} + +/// Returns the delta lifetime published by `observer`. +fn lifetime(observer: &Observer) -> DeltaReference { + observer.delta.load().reference() +} + +/// Promotes `observer` at `now`, asserting that the registry is still open. +/// +/// # Panics +/// +/// Panics if the registry has closed admission. +#[track_caller] +fn promote(registry: &UniverseRegistry, observer: Observer, now: Instant) { + assert!( + registry.promote(observer, now).is_ok(), + "an open registry should accept the promotion" + ); +} + +/// Builds a registry after a second generation displaces the fixture's generation. +/// +/// The returned instant is the displacement time. +/// +/// # Panics +/// +/// Panics if the displacement instant exceeds the range of [`Instant`] or the registry closes +/// during promotion. +fn displaced_registry( + fixture: &Fixture, + replacement: Arc, + hard: Duration, +) -> (UniverseRegistry, Instant) { + let registry = UniverseRegistry::new(hard); + let opened_at = Instant::now(); + promote( + ®istry, + observer(Arc::clone(&fixture.world), 1), + opened_at, + ); + + let retired_at = after(opened_at, Duration::from_secs(1)); + promote(®istry, observer(replacement, 2), retired_at); + + (registry, retired_at) +} + +/// Checks a selection's world identity, its epoch's generation and a query through the pair. +/// +/// # Panics +/// +/// Panics if the selection does not share `world` or its epoch names another generation. It also +/// panics if the first node has no visible position. +#[track_caller] +fn assert_universe(universe: &Universe, world: &Arc) { + assert!( + Arc::ptr_eq(universe.world(), world), + "the selection should hold the promoted world" + ); + assert_eq!( + universe.epoch().generation(), + world.generation().id(), + "the epoch should name its world's generation" + ); + assert!( + universe + .world() + .layout + .position(universe.epoch(), NodeRowId::MIN) + .is_some(), + "the first node row should have a position at the captured epoch" + ); +} + +/// Checks that both sides of `observation` are `world` at `published`. +/// +/// # Panics +/// +/// Panics if either side fails [`assert_universe`] or carries a delta lifetime other than +/// `published`. +#[track_caller] +fn assert_matching(observation: &Observation, world: &Arc, published: DeltaReference) { + assert_universe(observation.present(), world); + assert_universe(observation.requested(), world); + assert_eq!( + observation.present().epoch().reference(), + published, + "the active side should carry the promoted lifetime" + ); + assert_eq!( + observation.requested().epoch().reference(), + published, + "a matching selection should carry the active side's publication" + ); +} + +/// Rejects requests from an empty registry with [`ObserveError::Empty`]. +/// +/// This error precedes availability checking. +#[test] +fn observe_empty() { + let registry = UniverseRegistry::new(HARD); + + let Err(implicit) = registry.observe(None) else { + panic!("a registry without an active generation should not admit") + }; + assert_matches!( + implicit, + ObserveError::Empty, + "the refusal should name the missing active generation" + ); + + let Err(explicit) = registry.observe(Some(absent())) else { + panic!("a registry without an active generation should not admit a request") + }; + assert_matches!( + explicit, + ObserveError::Empty, + "the refusal should name the missing active generation rather than the request" + ); +} + +/// Rejects every request from a closed registry with [`ObserveError::Closed`]. +#[test] +fn observe_closed() { + let fixture = Fixture::new("registry-observe-closed"); + let registry = UniverseRegistry::new(HARD); + promote( + ®istry, + observer(Arc::clone(&fixture.world), 1), + Instant::now(), + ); + registry.close(); + + for requested in [None, Some(fixture.world.generation().id())] { + let Err(error) = registry.observe(requested) else { + panic!("a closed registry should not admit {requested:?}") + }; + assert_matches!( + error, + ObserveError::Closed, + "the refusal should name closed admission for {requested:?}" + ); + } +} + +/// Rejects an unpromoted generation with [`ObserveError::Unavailable`] and names it. +#[test] +fn observe_unknown_generation() { + let fixture = Fixture::new("registry-observe-unknown-generation"); + let unpromoted = fixture.variant("unknown generation").generation().id(); + let registry = UniverseRegistry::new(HARD); + promote( + ®istry, + observer(Arc::clone(&fixture.world), 1), + Instant::now(), + ); + + let Err(error) = registry.observe(Some(unpromoted)) else { + panic!("an unpromoted generation should not admit") + }; + assert_matches!(error, ObserveError::Unavailable(generation) if generation == unpromoted, "the refusal should name the requested generation"); +} + +/// Selects the active generation when a request names no generation. +#[test] +fn observe_implicit_active() { + let fixture = Fixture::new("registry-observe-implicit-active"); + let registry = UniverseRegistry::new(HARD); + let promoted = observer(Arc::clone(&fixture.world), 1); + let published = lifetime(&promoted); + promote(®istry, promoted, Instant::now()); + + let observation = registry + .observe(None) + .expect("an active generation should admit"); + + assert_matching(&observation, &fixture.world, published); +} + +/// Selects the active publication when a request names that generation. +#[test] +fn observe_matching_request() { + let fixture = Fixture::new("registry-observe-matching-request"); + let registry = UniverseRegistry::new(HARD); + let promoted = observer(Arc::clone(&fixture.world), 1); + let published = lifetime(&promoted); + promote(®istry, promoted, Instant::now()); + + let observation = registry + .observe(Some(fixture.world.generation().id())) + .expect("the active generation should admit its own request"); + + assert_matching(&observation, &fixture.world, published); +} + +/// Bounds the public admission timestamp by instants sampled around the call. +#[test] +fn observe_admission_instant() { + let fixture = Fixture::new("registry-observe-admission-instant"); + let registry = UniverseRegistry::new(HARD); + promote( + ®istry, + observer(Arc::clone(&fixture.world), 1), + Instant::now(), + ); + + let sampled_before = Instant::now(); + let observation = registry + .observe(None) + .expect("an active generation should admit"); + let sampled_after = Instant::now(); + + assert!( + observation.admitted_at() >= sampled_before, + "admission should not precede the instant sampled before the call" + ); + assert!( + observation.admitted_at() <= sampled_after, + "admission should not follow the instant sampled after the call" + ); +} + +/// Pairs a displaced requested generation with the new active generation. +#[test] +fn observe_promoted_pair() { + let fixture = Fixture::new("registry-observe-promoted-pair"); + let replacement = fixture.variant("promoted pair"); + let registry = UniverseRegistry::new(HARD); + + let opened = observer(Arc::clone(&fixture.world), 1); + let retired_publication = lifetime(&opened); + let opened_at = Instant::now(); + promote(®istry, opened, opened_at); + + let promoted = observer(Arc::clone(&replacement), 2); + let active_publication = lifetime(&promoted); + let retired_at = after(opened_at, Duration::from_secs(1)); + promote(®istry, promoted, retired_at); + assert_ne!( + active_publication, retired_publication, + "the two generations should publish independent lifetimes" + ); + + let observation = registry + .observe_at(Some(fixture.world.generation().id()), retired_at) + .expect("the displaced generation should admit within its retention"); + + assert_universe(observation.present(), &replacement); + assert_universe(observation.requested(), &fixture.world); + assert_eq!( + observation.present().epoch().reference(), + active_publication, + "the active side should carry the promoted lifetime" + ); + assert_eq!( + observation.requested().epoch().reference(), + retired_publication, + "the selected side should carry the displaced generation's lifetime" + ); +} + +/// Admits a displaced generation one nanosecond before its deadline without a cleanup pass. +#[test] +fn observe_retained_before_expiry() { + let fixture = Fixture::new("registry-observe-retained-before-expiry"); + let replacement = fixture.variant("retained before expiry"); + let (registry, retired_at) = displaced_registry(&fixture, Arc::clone(&replacement), HARD); + let age = HARD + .checked_sub(Duration::from_nanos(1)) + .expect("the retention interval should exceed one nanosecond"); + + let observation = registry + .observe_at( + Some(fixture.world.generation().id()), + after(retired_at, age), + ) + .expect("the displaced generation should admit before its retention elapses"); + + assert_universe(observation.requested(), &fixture.world); + assert_universe(observation.present(), &replacement); +} + +/// Rejects a displaced generation at its deadline without a cleanup pass. +#[test] +fn observe_retained_at_expiry() { + let fixture = Fixture::new("registry-observe-retained-at-expiry"); + let replacement = fixture.variant("retained at expiry"); + let (registry, retired_at) = displaced_registry(&fixture, replacement, HARD); + let displaced = fixture.world.generation().id(); + + let Err(error) = registry.observe_at(Some(displaced), after(retired_at, HARD)) else { + panic!("the displaced generation should not admit at its deadline") + }; + + assert_matches!(error, ObserveError::Unavailable(generation) if generation == displaced, "the refusal should name the expired generation"); +} + +/// Rejects a displaced generation at retirement when retention is zero. +#[test] +fn observe_zero_retention() { + let fixture = Fixture::new("registry-observe-zero-retention"); + let replacement = fixture.variant("zero retention"); + let (registry, retired_at) = + displaced_registry(&fixture, Arc::clone(&replacement), Duration::ZERO); + let displaced = fixture.world.generation().id(); + + let Err(error) = registry.observe_at(Some(displaced), retired_at) else { + panic!("zero retention should not admit the displaced generation") + }; + assert_matches!(error, ObserveError::Unavailable(generation) if generation == displaced, "the refusal should name the displaced generation"); + + let observation = registry + .observe_at(None, retired_at) + .expect("zero retention should admit the active generation"); + assert_universe(observation.present(), &replacement); +} + +/// Appends an expired generation to caller output and removes its registry entry. +#[test] +fn expire_appends_output() { + let fixture = Fixture::new("registry-expire-appends-output"); + let second = fixture.variant("expire second"); + let third = fixture.variant("expire third"); + let registry = UniverseRegistry::new(HARD); + + let opened_at = Instant::now(); + promote( + ®istry, + observer(Arc::clone(&fixture.world), 1), + opened_at, + ); + let first_retired_at = after(opened_at, Duration::from_secs(1)); + promote( + ®istry, + observer(Arc::clone(&second), 2), + first_retired_at, + ); + let second_retired_at = after(first_retired_at, Duration::from_secs(1)); + promote( + ®istry, + observer(Arc::clone(&third), 3), + second_retired_at, + ); + + let deadline = after(first_retired_at, HARD); + let mut expired = vec![absent()]; + registry.expire(deadline, &mut expired); + + assert_eq!( + expired, + [absent(), fixture.world.generation().id()], + "expiry should append the elapsed generation after the caller's entry" + ); + registry.expire(deadline, &mut expired); + assert_eq!( + expired, + [absent(), fixture.world.generation().id()], + "a second pass at the same deadline should find the entry already removed" + ); + + let unexpired = registry + .observe_at(Some(second.generation().id()), deadline) + .expect("the generation within its retention should still admit"); + assert_universe(unexpired.requested(), &second); + assert_universe(unexpired.present(), &third); +} + +/// Drops the retained entry on reactivation and retires the displaced generation. +#[test] +fn promote_reactivation() { + let fixture = Fixture::new("registry-promote-reactivation"); + let replacement = fixture.variant("reactivation"); + let (registry, retired_at) = displaced_registry(&fixture, Arc::clone(&replacement), HARD); + + let reactivated = observer(Arc::clone(&fixture.world), 3); + let published = lifetime(&reactivated); + promote( + ®istry, + reactivated, + after(retired_at, Duration::from_secs(1)), + ); + + let deadline = after(retired_at, HARD); + let mut expired = Vec::new(); + registry.expire(deadline, &mut expired); + assert!( + expired.is_empty(), + "reactivation should remove the reactivated generation's retained entry" + ); + + let observation = registry + .observe_at(Some(replacement.generation().id()), deadline) + .expect("the newly displaced generation should admit within its own retention"); + assert_universe(observation.present(), &fixture.world); + assert_universe(observation.requested(), &replacement); + assert_eq!( + observation.present().epoch().reference(), + published, + "the active side should carry the reactivated lifetime" + ); +} + +/// Replaces the active generation's previous delta lifetime with a fresh one. +#[test] +fn promote_same_generation() { + let fixture = Fixture::new("registry-promote-same-generation"); + let replacement = fixture.variant("same generation"); + let registry = UniverseRegistry::new(HARD); + + let opened = observer(Arc::clone(&fixture.world), 1); + let stale = lifetime(&opened); + let opened_at = Instant::now(); + promote(®istry, opened, opened_at); + + let restarted = observer(Arc::clone(&fixture.world), 2); + let current = lifetime(&restarted); + let restarted_at = after(opened_at, Duration::from_secs(1)); + promote(®istry, restarted, restarted_at); + assert_ne!( + stale, current, + "the two lifetimes should carry independent identities" + ); + + let probed_at = after(restarted_at, HARD); + let mut expired = Vec::new(); + registry.expire(probed_at, &mut expired); + assert!( + expired.is_empty(), + "the replaced lifetime should hold no retained entry under the active generation" + ); + + let retired_at = after(probed_at, Duration::from_secs(1)); + promote(®istry, observer(replacement, 3), retired_at); + let observation = registry + .observe_at(Some(fixture.world.generation().id()), retired_at) + .expect("the displaced generation should admit within its retention"); + assert_eq!( + observation.requested().epoch().reference(), + current, + "retirement should retain the generation's latest lifetime" + ); +} + +/// Refuses promotion after closure and returns the supplied observer. +#[test] +fn promote_closed() { + let fixture = Fixture::new("registry-promote-closed"); + let registry = UniverseRegistry::new(HARD); + registry.close(); + + let supplied = observer(Arc::clone(&fixture.world), 1); + let published = lifetime(&supplied); + let returned = registry + .promote(supplied, Instant::now()) + .expect_err("a closed registry should refuse promotion"); + + assert!( + Arc::ptr_eq(&returned.world, &fixture.world), + "the refusal should return the supplied world" + ); + assert_eq!( + lifetime(&returned), + published, + "the refusal should return the supplied publication" + ); +} + +/// Keeps a held observation readable after retirement, cleanup and closure. +#[test] +fn observation_outlives_registry() { + let fixture = Fixture::new("registry-observation-outlives-registry"); + let replacement = fixture.variant("outlives registry"); + let registry = UniverseRegistry::new(HARD); + let opened_at = Instant::now(); + let opened = observer(Arc::clone(&fixture.world), 1); + let published = lifetime(&opened); + promote(®istry, opened, opened_at); + + let held = registry + .observe_at(None, opened_at) + .expect("an active generation should admit"); + + let retired_at = after(opened_at, Duration::from_secs(1)); + promote(®istry, observer(replacement, 2), retired_at); + let mut expired = Vec::new(); + registry.expire(after(retired_at, HARD), &mut expired); + assert_eq!( + expired, + [fixture.world.generation().id()], + "cleanup should remove the displaced generation" + ); + registry.close(); + + assert_eq!( + held.admitted_at(), + opened_at, + "the observation should keep its supplied admission instant" + ); + assert_matching(&held, &fixture.world, published); +} diff --git a/libs/@local/graph/atlas/src/serve/runtime/tests.rs b/libs/@local/graph/atlas/src/serve/runtime/tests.rs new file mode 100644 index 00000000000..597499a1a25 --- /dev/null +++ b/libs/@local/graph/atlas/src/serve/runtime/tests.rs @@ -0,0 +1,676 @@ +//! Cases covering one generation's runtime lifecycle: opening, probing and joining its feed. + +use alloc::sync::{Arc, Weak}; +use core::{ + future::{self, Future}, + time::Duration, +}; +use std::io; + +use error_stack::{Report, ReportSink}; +use futures::FutureExt as _; +use hash_graph_postgres_store::store::{ + DatabaseConnectionInfo, DatabasePoolConfig, DatabaseType, PostgresStorePool, + PostgresStoreSettings, +}; +use hashql_core::id::Id as _; +use rand::{SeedableRng as _, TryCryptoRng, TryRng, rngs::StdRng}; +use tokio::{sync::oneshot, task::JoinHandle}; +use tokio_postgres::NoTls; +use tokio_util::sync::CancellationToken; + +use super::{Feed, FeedOptions, FeedState, Runtime, RuntimeError}; +use crate::{ + dataset::TemporalAxes, + device::Device, + file::generation::{Generation, GenerationRoot}, + identity::NodeRowId, + math::nz, + serve::{ + delta::{ + Delta, DeltaFeedTaskOptions, DeltaPlacementTaskOptions, DeltaReader, DeltaTaskError, + DeltaTaskOptions, + }, + tests::fixture::{TamperFixture, secret}, + world::World, + }, +}; + +/// The fixed seed every fixture's generator uses for repeatable arrivals. +const SEED: u64 = 0xC0FF_EE11; + +// awaitable test adapters compose the runtime's stop-and-poll operations. The manager slot cannot +// await inside its state machine. + +/// Waits for `runtime`'s runner and yields its result, or [`None`] without an unjoined runner. +/// +/// # Errors +/// +/// Returns the [`RuntimeError`] the runner reported, or the join failure of a runner that +/// panicked. +async fn join(runtime: &mut Runtime) -> Option>> { + future::poll_fn(|context| runtime.poll_join(context)).await +} + +/// Requests shutdown of `runtime` and waits for its runner. +/// +/// # Errors +/// +/// Returns the [`RuntimeError`] the runner reported. A runtime with no runner to join +/// succeeds. +async fn shutdown(runtime: &mut Runtime) -> Result<(), Report> { + runtime.stop(); + join(runtime).await.unwrap_or(Ok(())) +} + +/// Returns the common runtime-test feed configuration. +/// +/// The feed and placement loops both use five-second periods. Placement runs on the CPU with no +/// workflow. It admits one pending item and allows one attempt in each database and workflow phase. +fn options() -> FeedOptions { + FeedOptions { + task: DeltaTaskOptions { + feed: DeltaFeedTaskOptions { + tick_rate: Duration::from_secs(5), + safety_lag: Duration::from_secs(60), + }, + placement: DeltaPlacementTaskOptions { + tick_rate: Duration::from_secs(5), + tries_workflow: 1, + tries_database: 1, + minimum_projection_interval: 1, + max_pending: nz!(1), + }, + }, + device: Device::Cpu.pin(0).resolve(), + workflow: None, + } +} + +/// Builds an empty test store pool and a weak ownership probe. +/// +/// The weak handle detects whether the runtime retained its pool reference. +/// +/// # Panics +/// +/// Panics if store-pool construction fails. +async fn unconnected_pool() -> (Arc, Weak) { + let pool = Arc::new( + PostgresStorePool::new( + &DatabaseConnectionInfo::new( + DatabaseType::Postgres, + "runtime-test".to_owned(), + String::new(), + "/no-runtime-test-postgres".to_owned(), + 5432, + "runtime-test".to_owned(), + ), + &DatabasePoolConfig { + max_connections: nz!(1), + }, + NoTls, + PostgresStoreSettings::default(), + ) + .await + .expect("should construct an unconnected pool"), + ); + let weak = Arc::downgrade(&pool); + (pool, weak) +} + +/// Runs `test` under a one-second virtual timeout on a paused current-thread runtime. +/// +/// # Panics +/// +/// Panics if runtime construction fails, if `test` panics, or if the virtual timeout expires. +#[track_caller] +fn run_controlled(test: impl Future) { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .start_paused(true) + .build() + .expect("should build the runtime"); + + let result = + runtime.block_on(async { tokio::time::timeout(Duration::from_secs(1), test).await }); + result.expect("the controlled test should finish without a stalled task"); +} + +#[test] +#[should_panic(expected = "the controlled test should finish without a stalled task")] +fn controlled_stall() { + run_controlled(async { + tokio::time::advance(Duration::ZERO).await; + future::pending::<()>().await; + }); +} + +/// Builds a published generation with temporal axes. +/// +/// The returned fixture owns its directory. A feed opens only over a generation carrying axes. +/// +/// # Panics +/// +/// Panics on fixture-publication failure or a missing parent directory. Root opening, staging, +/// artifact copying, generation sealing, and generation reopening must also succeed. +fn axes_fixture(name: &str) -> (TamperFixture, Generation) { + let fixture = TamperFixture::publish(name); + let generation = fixture.generation(); + let root = GenerationRoot::new(generation.path().parent().expect("the fixture has a root")) + .expect("the fixture root should open"); + let published = { + let staging = root.stage().expect("the staging should open"); + + for file in generation.repository().files.files() { + std::fs::copy(generation.path_of(&file.name), staging.path_of(&file.name)) + .expect("the fixture artifact should copy"); + } + + let mut repository = generation.repository().clone(); + repository.metadata.snapshot.axes = Some(TemporalAxes::now()); + staging + .seal(&repository) + .expect("the generation should seal") + }; + let generation = root + .open(published.id()) + .expect("the generation should open"); + (fixture, generation) +} + +/// Builds a runtime around `feed` and returns its directory fixture. +/// +/// # Panics +/// +/// Panics on failure during fixture publication or world opening. +fn controlled(name: &str, feed: Feed) -> (TamperFixture, Runtime) { + let fixture = TamperFixture::publish(name); + let world = Arc::new( + World::open(fixture.generation().clone(), &secret()).expect("the world should open"), + ); + let delta = Delta::new(Arc::clone(&world), StdRng::seed_from_u64(SEED)) + .expect("the delta should initialize"); + let runtime = Runtime { + world, + reader: DeltaReader::from(delta), + feed: Some(feed), + }; + (fixture, runtime) +} + +/// Waits for runner completion before testing nonblocking probe state. +/// +/// This distinguishes a finished-runner verdict from scheduler timing. +/// +/// # Panics +/// +/// Panics if `runtime` has no feed. If the runner has not finished, it also panics when polled +/// outside a Tokio runtime with time enabled. +async fn feed_finished(runtime: &Runtime) { + let feed = runtime.feed.as_ref().expect("should own a feed"); + while !feed.task.is_finished() { + tokio::time::sleep(Duration::from_millis(1)).await; + } +} + +/// An entropy source that rejects every request. +/// +/// It models entropy from which runtime startup cannot seed a delta lifetime. +struct UnavailableEntropy; + +impl TryRng for UnavailableEntropy { + type Error = io::Error; + + fn try_next_u32(&mut self) -> Result { + Err(io::Error::other("entropy unavailable")) + } + + fn try_next_u64(&mut self) -> Result { + Err(io::Error::other("entropy unavailable")) + } + + fn try_fill_bytes(&mut self, dst: &mut [u8]) -> Result<(), Self::Error> { + let _: &mut [u8] = dst; + Err(io::Error::other("entropy unavailable")) + } +} + +impl TryCryptoRng for UnavailableEntropy {} + +/// Avoids starting a runner when feed options are absent. +/// +/// The returned runtime never retains the unused store pool, and its world remains readable. +#[tokio::test] +async fn open_disabled() { + let (_fixture, generation) = axes_fixture("runtime-open-disabled"); + let (pool, weak_pool) = unconnected_pool().await; + let mut runtime = Runtime::open( + generation, + &secret(), + pool, + StdRng::seed_from_u64(SEED), + None, + ) + .expect("the disabled feed should open"); + + assert!(runtime.reader().load().contains_node(NodeRowId::MIN)); + assert!(weak_pool.upgrade().is_none()); + core::assert_matches!(runtime.try_join(), Ok(FeedState::Absent)); + assert!(join(&mut runtime).await.is_none()); +} + +/// Avoids starting a feed when temporal axes are absent. +/// +/// Opening without axes drops the unused store pool. +#[tokio::test] +async fn open_without_axes() { + let fixture = TamperFixture::publish("runtime-open-without-axes"); + let (pool, weak_pool) = unconnected_pool().await; + let mut runtime = Runtime::open( + fixture.generation().clone(), + &secret(), + pool, + StdRng::seed_from_u64(SEED), + Some(options()), + ) + .expect("the generation without axes should open"); + + assert!(runtime.reader().load().contains_node(NodeRowId::MIN)); + assert!(weak_pool.upgrade().is_none()); + core::assert_matches!(runtime.try_join(), Ok(FeedState::Absent)); + assert!(join(&mut runtime).await.is_none()); +} + +/// Keeps world and reader handles usable after temporal-feed shutdown. +/// +/// A generation with axes and no projector opens with a live feed. Shutting it down +/// joins the runner, and the world, the reader clones and a snapshot taken before the +/// shutdown all stay readable afterward, with the generation directory left on disk. +#[tokio::test] +async fn open_temporal() { + let (_fixture, generation) = axes_fixture("runtime-open-temporal"); + assert!(generation.repository().files.projector.is_none()); + let (pool, weak_pool) = unconnected_pool().await; + let mut runtime = Runtime::open( + generation, + &secret(), + pool, + StdRng::seed_from_u64(SEED), + Some(options()), + ) + .expect("the temporal generation should open"); + + let world = Arc::clone(runtime.world()); + let reader = runtime.reader().clone(); + let captured = reader.load(); + assert!(runtime.feed.is_some()); + + shutdown(&mut runtime).await.expect("the feed should join"); + drop(runtime); + + assert!(weak_pool.upgrade().is_none()); + assert!(reader.load().contains_node(NodeRowId::MIN)); + assert!(captured.contains_node(NodeRowId::MIN)); + assert!(world.generation().path().is_dir()); +} + +/// Releases runtime resources when feed options are invalid. +/// +/// A zero tick rate refuses at open with a feed error, and the refusal releases the +/// pool reference rather than leaking it into a runtime that never opened. +#[tokio::test] +async fn open_invalid_interval() { + let (_fixture, generation) = axes_fixture("runtime-open-invalid-interval"); + let (pool, weak_pool) = unconnected_pool().await; + let mut options = options(); + options.task.feed.tick_rate = Duration::ZERO; + let error = Runtime::open( + generation, + &secret(), + pool, + StdRng::seed_from_u64(SEED), + Some(options), + ) + .err() + .expect("a zero interval should fail"); + + core::assert_matches!(error.current_context(), RuntimeError::Feed); + assert!(weak_pool.upgrade().is_none()); +} + +/// Refuses startup and releases the store pool when the entropy source returns an error. +#[tokio::test] +async fn start_entropy_failure() { + let fixture = TamperFixture::publish("runtime-start-entropy-failure"); + let world = Arc::new( + World::open(fixture.generation().clone(), &secret()).expect("the world should open"), + ); + let (pool, weak_pool) = unconnected_pool().await; + let error = Runtime::start(world, pool, UnavailableEntropy, None) + .err() + .expect("unavailable entropy should fail"); + + core::assert_matches!(error.current_context(), RuntimeError::Entropy); + assert!(weak_pool.upgrade().is_none()); +} + +/// Retains a signalled runner until shutdown joins it. +/// +/// The eventual join returns its result exactly once. +#[test] +fn shutdown_cancelled() { + run_controlled(async { + let (observed, observation) = oneshot::channel::<()>(); + let (release, released) = oneshot::channel::<()>(); + + let cancel = CancellationToken::new(); + let cancelled = cancel.clone().cancelled_owned(); + + let task = tokio::spawn(async move { + cancelled.await; + observed + .send(()) + .expect("the observation should remain open"); + released.await.expect("the worker should be released"); + Ok(()) + }); + + let (_fixture, mut runtime) = controlled( + "runtime-shutdown-cancelled", + Feed { + shutdown: cancel, + task, + }, + ); + + assert!(shutdown(&mut runtime).now_or_never().is_none()); + observation + .await + .expect("the worker should observe shutdown"); + assert!(runtime.feed.is_some()); + release + .send(()) + .expect("the worker should still await release"); + join(&mut runtime) + .await + .expect("the join handle should remain owned") + .expect("the worker should join"); + assert!(join(&mut runtime).await.is_none()); + }); +} + +/// Treats shutdown as idempotent after the runner joins. +/// +/// A second call succeeds, and another join yields nothing. +#[test] +fn shutdown_repeated() { + run_controlled(async { + let cancel = CancellationToken::new(); + let cancelled = cancel.clone().cancelled_owned(); + + let task = tokio::spawn(async move { + cancelled.await; + Ok(()) + }); + + let (_fixture, mut runtime) = controlled( + "runtime-shutdown-repeated", + Feed { + shutdown: cancel, + task, + }, + ); + + shutdown(&mut runtime) + .await + .expect("the worker should join"); + shutdown(&mut runtime) + .await + .expect("repeated shutdown should succeed"); + assert!(join(&mut runtime).await.is_none()); + }); +} + +/// Consumes a feed failure exactly once. +/// +/// The delta failure becomes a feed error. A later join finds no runner. +#[test] +fn join_feed_error() { + run_controlled(async { + let cancel = CancellationToken::new(); + + let task = tokio::spawn(async { + let mut sink = ReportSink::new(); + sink.attempt(Err::<(), _>(Report::new(DeltaTaskError::Feed))); + sink.finish() + }); + + let (_fixture, mut runtime) = controlled( + "runtime-join-feed-error", + Feed { + shutdown: cancel, + task, + }, + ); + + let error = join(&mut runtime) + .await + .expect("the runner should have a result") + .expect_err("the runner should report its failure"); + core::assert_matches!(error.current_context(), RuntimeError::Feed); + assert!(join(&mut runtime).await.is_none()); + }); +} + +/// Converts a runner panic into a consumed join error. +/// +/// The caller does not unwind, and a later join finds no runner. +#[test] +fn join_panic() { + run_controlled(async { + let cancel = CancellationToken::new(); + let task: JoinHandle>> = + tokio::spawn(async { panic!("controlled worker panic") }); + + let (_fixture, mut runtime) = controlled( + "runtime-join-panic", + Feed { + shutdown: cancel, + task, + }, + ); + + let error = join(&mut runtime) + .await + .expect("the runner should have a result") + .expect_err("the runner should report its panic"); + core::assert_matches!(error.current_context(), RuntimeError::Join); + assert!(join(&mut runtime).await.is_none()); + }); +} + +/// Consumes a finished runner exactly once through nonblocking probes. +/// +/// The non-blocking probe reports a running runner while the worker waits, reports it +/// finished once it ends, giving up the feed at that point, and absent from then on, +/// with a later shutdown still succeeding. +#[test] +fn try_join_pending() { + run_controlled(async { + let (release, released) = oneshot::channel::<()>(); + let task = tokio::spawn(async move { + released.await.expect("should release the worker"); + Ok(()) + }); + let (_fixture, mut runtime) = controlled( + "runtime-try-join-pending", + Feed { + shutdown: CancellationToken::new(), + task, + }, + ); + + core::assert_matches!(runtime.try_join(), Ok(FeedState::Running)); + assert!(runtime.feed.is_some()); + release.send(()).expect("should retain the waiting worker"); + feed_finished(&runtime).await; + + core::assert_matches!(runtime.try_join(), Ok(FeedState::Finished)); + assert!(runtime.feed.is_none()); + core::assert_matches!(runtime.try_join(), Ok(FeedState::Absent)); + assert!(join(&mut runtime).await.is_none()); + shutdown(&mut runtime).await.expect("should remain joined"); + }); +} + +/// Returns and consumes a finished feed error once. +/// +/// Later probes report absence, and shutdown succeeds. +#[test] +fn try_join_feed_error() { + run_controlled(async { + let task = tokio::spawn(async { Err(Report::new(DeltaTaskError::Feed).expand()) }); + let (_fixture, mut runtime) = controlled( + "runtime-try-join-feed-error", + Feed { + shutdown: CancellationToken::new(), + task, + }, + ); + feed_finished(&runtime).await; + + let error = runtime + .try_join() + .expect_err("should report the feed failure"); + core::assert_matches!(error.current_context(), RuntimeError::Feed); + assert!(runtime.feed.is_none()); + core::assert_matches!(runtime.try_join(), Ok(FeedState::Absent)); + assert!(join(&mut runtime).await.is_none()); + shutdown(&mut runtime).await.expect("should remain joined"); + }); +} + +/// Returns and consumes a finished join error once. +/// +/// Later probes report absence, and shutdown succeeds. +#[test] +fn try_join_panic() { + run_controlled(async { + let task = tokio::spawn(async { panic!("controlled worker panic") }); + let (_fixture, mut runtime) = controlled( + "runtime-try-join-panic", + Feed { + shutdown: CancellationToken::new(), + task, + }, + ); + feed_finished(&runtime).await; + + let error = runtime + .try_join() + .expect_err("should report the runner panic"); + core::assert_matches!(error.current_context(), RuntimeError::Join); + assert!(runtime.feed.is_none()); + core::assert_matches!(runtime.try_join(), Ok(FeedState::Absent)); + assert!(join(&mut runtime).await.is_none()); + shutdown(&mut runtime).await.expect("should remain joined"); + }); +} + +/// Retains a completed runner after a probe exhausts its budget. +/// +/// [`tokio::task::consume_budget`] spends the budget that the join probe also needs. A scheduler +/// yield then makes the completed result observable. +#[test] +fn try_join_exhausted_budget() { + /// Probes to spend on the drain, well past the budget one task poll starts with. + const DRAIN_PROBES: usize = 1024; + + run_controlled(async { + let task = tokio::spawn(async { Ok(()) }); + let (_fixture, mut runtime) = controlled( + "runtime-try-join-exhausted-budget", + Feed { + shutdown: CancellationToken::new(), + task, + }, + ); + feed_finished(&runtime).await; + + let exhausted = + (0..DRAIN_PROBES).any(|_| tokio::task::consume_budget().now_or_never().is_none()); + assert!( + exhausted, + "consuming budget should report pending once the drain spends it" + ); + + core::assert_matches!(runtime.try_join(), Ok(FeedState::Running)); + assert!(runtime.feed.is_some()); + + tokio::task::yield_now().await; + + core::assert_matches!(runtime.try_join(), Ok(FeedState::Finished)); + assert!(runtime.feed.is_none()); + core::assert_matches!(runtime.try_join(), Ok(FeedState::Absent)); + assert!(join(&mut runtime).await.is_none()); + shutdown(&mut runtime).await.expect("should remain joined"); + }); +} + +/// Requests graceful shutdown without aborting the runner when the runtime drops. +/// +/// Worker-owned resources remain held until completion, and reader clones remain valid. +#[test] +fn drop_graceful() { + run_controlled(async { + let resource = Arc::new(()); + let weak = Arc::downgrade(&resource); + let (started, startup) = oneshot::channel::<()>(); + let (observed, observation) = oneshot::channel::<()>(); + let (release, released) = oneshot::channel::<()>(); + let (done, finished) = oneshot::channel::<()>(); + + let cancel = CancellationToken::new(); + let cancelled = cancel.clone().cancelled_owned(); + + let task = tokio::spawn(async move { + started + .send(()) + .expect("the startup observation should remain open"); + cancelled.await; + observed + .send(()) + .expect("the observation should remain open"); + released.await.expect("the worker should be released"); + drop(resource); + done.send(()).expect("the completion should remain open"); + Ok(()) + }); + + let (_fixture, runtime) = controlled( + "runtime-drop-graceful", + Feed { + shutdown: cancel, + task, + }, + ); + let reader = runtime.reader().clone(); + startup + .await + .expect("the worker should start before its owner drops"); + drop(runtime); + + observation + .await + .expect("the worker should observe shutdown"); + assert!(weak.upgrade().is_some()); + release + .send(()) + .expect("the worker should still await release"); + finished + .await + .expect("the worker should finish without an abort"); + assert!(weak.upgrade().is_none()); + assert!(reader.load().contains_node(NodeRowId::MIN)); + }); +}