From ae48c6e8157410f3ba81c34ff19dad2680346b5b Mon Sep 17 00:00:00 2001 From: Bilal Mahmoud <7252775+indietyp@users.noreply.github.com> Date: Mon, 14 Sep 2026 15:34:11 +0200 Subject: [PATCH 1/3] refactor: remove the single-generation atlas serving layer --- apps/hash-graph/src/subcommand/atlas.rs | 245 +- apps/hash-graph/src/subcommand/mod.rs | 2 +- libs/@local/graph/atlas/package.json | 1 - .../graph/atlas/src/api/authorization.rs | 424 -- libs/@local/graph/atlas/src/api/clause.rs | 226 -- libs/@local/graph/atlas/src/api/current.rs | 55 - libs/@local/graph/atlas/src/api/edges.rs | 204 - libs/@local/graph/atlas/src/api/extract.rs | 419 -- libs/@local/graph/atlas/src/api/headers.rs | 148 - libs/@local/graph/atlas/src/api/locate.rs | 249 -- libs/@local/graph/atlas/src/api/manifest.rs | 603 --- libs/@local/graph/atlas/src/api/mod.rs | 306 -- libs/@local/graph/atlas/src/api/openapi.rs | 56 - libs/@local/graph/atlas/src/api/problem.rs | 425 -- libs/@local/graph/atlas/src/api/saltile.rs | 98 - libs/@local/graph/atlas/src/api/tile.rs | 144 - libs/@local/graph/atlas/src/api/translate.rs | 129 - libs/@local/graph/atlas/src/api/visibility.rs | 500 --- libs/@local/graph/atlas/src/cli/mod.rs | 8 +- libs/@local/graph/atlas/src/cli/serve.rs | 530 --- libs/@local/graph/atlas/src/lib.rs | 16 +- libs/@local/graph/atlas/src/salt/mod.rs | 1 - libs/@local/graph/atlas/src/salt/wire/cbor.rs | 163 - .../@local/graph/atlas/src/salt/wire/edges.rs | 191 - .../graph/atlas/src/salt/wire/envelope.rs | 210 - .../graph/atlas/src/salt/wire/fixtures.rs | 1248 ------ .../graph/atlas/src/salt/wire/locate.rs | 463 --- libs/@local/graph/atlas/src/salt/wire/mod.rs | 83 - .../@local/graph/atlas/src/salt/wire/tests.rs | 1252 ------ libs/@local/graph/atlas/src/salt/wire/tile.rs | 514 --- .../graph/atlas/src/serve/authorization.rs | 760 ---- .../@local/graph/atlas/src/serve/cache/mod.rs | 574 --- .../graph/atlas/src/serve/cache/scope.rs | 194 - .../graph/atlas/src/serve/cache/tests.rs | 978 ----- libs/@local/graph/atlas/src/serve/codec.rs | 277 -- libs/@local/graph/atlas/src/serve/colour.rs | 191 - .../graph/atlas/src/serve/delta/consumer.rs | 503 --- .../@local/graph/atlas/src/serve/delta/mod.rs | 366 -- .../graph/atlas/src/serve/delta/overlay.rs | 120 - .../graph/atlas/src/serve/delta/placement.rs | 680 ---- .../graph/atlas/src/serve/delta/register.rs | 659 --- .../graph/atlas/src/serve/delta/snapshot.rs | 351 -- .../graph/atlas/src/serve/delta/staging.rs | 749 ---- .../graph/atlas/src/serve/delta/tests.rs | 1333 ------ .../graph/atlas/src/serve/density/mod.rs | 436 -- .../graph/atlas/src/serve/density/tests.rs | 467 --- libs/@local/graph/atlas/src/serve/edges.rs | 442 -- libs/@local/graph/atlas/src/serve/error.rs | 475 --- libs/@local/graph/atlas/src/serve/grid.rs | 138 - .../graph/atlas/src/serve/hydrate/client.rs | 532 --- .../graph/atlas/src/serve/hydrate/columns.rs | 358 -- .../graph/atlas/src/serve/hydrate/compile.rs | 419 -- .../graph/atlas/src/serve/hydrate/mod.rs | 79 - .../graph/atlas/src/serve/hydrate/order.rs | 153 - .../graph/atlas/src/serve/hydrate/select.rs | 102 - ...ts__tests__bare_detail_statement_text.snap | 17 - ...__tests__masked_detail_statement_text.snap | 16 - ...ents__tests__type_urls_statement_text.snap | 11 - ...atements__tests__types_statement_text.snap | 12 - .../atlas/src/serve/hydrate/statements.rs | 406 -- .../atlas/src/serve/hydrate/type_urls.rs | 258 -- libs/@local/graph/atlas/src/serve/intern.rs | 283 -- libs/@local/graph/atlas/src/serve/locate.rs | 1015 ----- libs/@local/graph/atlas/src/serve/manifest.rs | 232 -- libs/@local/graph/atlas/src/serve/mod.rs | 422 -- .../graph/atlas/src/serve/neighbourhood.rs | 846 ---- libs/@local/graph/atlas/src/serve/open.rs | 498 --- .../graph/atlas/src/serve/schedule/cut.rs | 385 -- .../graph/atlas/src/serve/schedule/mod.rs | 806 ---- .../graph/atlas/src/serve/schedule/tests.rs | 779 ---- libs/@local/graph/atlas/src/serve/secret.rs | 63 - .../graph/atlas/src/serve/tests/arrival.rs | 1183 ------ .../atlas/src/serve/tests/authorization.rs | 394 -- .../graph/atlas/src/serve/tests/auxiliary.rs | 395 -- .../atlas/src/serve/tests/delta_edges.rs | 2475 ----------- .../graph/atlas/src/serve/tests/density.rs | 226 -- .../atlas/src/serve/tests/frame_channel.rs | 220 - .../graph/atlas/src/serve/tests/masking.rs | 1208 ------ .../atlas/src/serve/tests/metadata_channel.rs | 152 - .../@local/graph/atlas/src/serve/tests/mod.rs | 3602 ----------------- .../graph/atlas/src/serve/tests/open.rs | 697 ---- .../graph/atlas/src/serve/tests/row_codec.rs | 476 --- .../graph/atlas/src/serve/tests/schedule.rs | 1127 ------ .../graph/atlas/src/serve/tests/withdrawal.rs | 903 ----- libs/@local/graph/atlas/src/serve/tile.rs | 804 ---- .../@local/graph/atlas/src/serve/translate.rs | 385 -- libs/@local/graph/atlas/src/serve/view.rs | 331 -- .../graph/atlas/src/serve/visibility.rs | 412 -- .../graph/atlas/src/serve/walk/census.rs | 152 - .../@local/graph/atlas/src/serve/walk/full.rs | 201 - libs/@local/graph/atlas/src/serve/walk/mod.rs | 153 - .../graph/atlas/src/serve/walk/subtract.rs | 440 -- .../@local/graph/atlas/tests/route_fixture.rs | 761 ---- 93 files changed, 29 insertions(+), 42966 deletions(-) delete mode 100644 libs/@local/graph/atlas/src/api/authorization.rs delete mode 100644 libs/@local/graph/atlas/src/api/clause.rs delete mode 100644 libs/@local/graph/atlas/src/api/current.rs delete mode 100644 libs/@local/graph/atlas/src/api/edges.rs delete mode 100644 libs/@local/graph/atlas/src/api/extract.rs delete mode 100644 libs/@local/graph/atlas/src/api/headers.rs delete mode 100644 libs/@local/graph/atlas/src/api/locate.rs delete mode 100644 libs/@local/graph/atlas/src/api/manifest.rs delete mode 100644 libs/@local/graph/atlas/src/api/mod.rs delete mode 100644 libs/@local/graph/atlas/src/api/openapi.rs delete mode 100644 libs/@local/graph/atlas/src/api/problem.rs delete mode 100644 libs/@local/graph/atlas/src/api/saltile.rs delete mode 100644 libs/@local/graph/atlas/src/api/tile.rs delete mode 100644 libs/@local/graph/atlas/src/api/translate.rs delete mode 100644 libs/@local/graph/atlas/src/api/visibility.rs delete mode 100644 libs/@local/graph/atlas/src/cli/serve.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/cbor.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/edges.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/envelope.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/fixtures.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/locate.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/mod.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/tests.rs delete mode 100644 libs/@local/graph/atlas/src/salt/wire/tile.rs delete mode 100644 libs/@local/graph/atlas/src/serve/authorization.rs delete mode 100644 libs/@local/graph/atlas/src/serve/cache/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/cache/scope.rs delete mode 100644 libs/@local/graph/atlas/src/serve/cache/tests.rs delete mode 100644 libs/@local/graph/atlas/src/serve/codec.rs delete mode 100644 libs/@local/graph/atlas/src/serve/colour.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/consumer.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/overlay.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/placement.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/register.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/snapshot.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/staging.rs delete mode 100644 libs/@local/graph/atlas/src/serve/delta/tests.rs delete mode 100644 libs/@local/graph/atlas/src/serve/density/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/density/tests.rs delete mode 100644 libs/@local/graph/atlas/src/serve/edges.rs delete mode 100644 libs/@local/graph/atlas/src/serve/error.rs delete mode 100644 libs/@local/graph/atlas/src/serve/grid.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/client.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/columns.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/compile.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/order.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/select.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/statements.rs delete mode 100644 libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs delete mode 100644 libs/@local/graph/atlas/src/serve/intern.rs delete mode 100644 libs/@local/graph/atlas/src/serve/locate.rs delete mode 100644 libs/@local/graph/atlas/src/serve/manifest.rs delete mode 100644 libs/@local/graph/atlas/src/serve/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/neighbourhood.rs delete mode 100644 libs/@local/graph/atlas/src/serve/open.rs delete mode 100644 libs/@local/graph/atlas/src/serve/schedule/cut.rs delete mode 100644 libs/@local/graph/atlas/src/serve/schedule/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/schedule/tests.rs delete mode 100644 libs/@local/graph/atlas/src/serve/secret.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/arrival.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/authorization.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/auxiliary.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/delta_edges.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/density.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/frame_channel.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/masking.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/metadata_channel.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/open.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/row_codec.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/schedule.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tests/withdrawal.rs delete mode 100644 libs/@local/graph/atlas/src/serve/tile.rs delete mode 100644 libs/@local/graph/atlas/src/serve/translate.rs delete mode 100644 libs/@local/graph/atlas/src/serve/view.rs delete mode 100644 libs/@local/graph/atlas/src/serve/visibility.rs delete mode 100644 libs/@local/graph/atlas/src/serve/walk/census.rs delete mode 100644 libs/@local/graph/atlas/src/serve/walk/full.rs delete mode 100644 libs/@local/graph/atlas/src/serve/walk/mod.rs delete mode 100644 libs/@local/graph/atlas/src/serve/walk/subtract.rs delete mode 100644 libs/@local/graph/atlas/tests/route_fixture.rs diff --git a/apps/hash-graph/src/subcommand/atlas.rs b/apps/hash-graph/src/subcommand/atlas.rs index 0950574dc1a..0d1025c126e 100644 --- a/apps/hash-graph/src/subcommand/atlas.rs +++ b/apps/hash-graph/src/subcommand/atlas.rs @@ -1,27 +1,16 @@ -use alloc::sync::Arc; -use core::{net::SocketAddr, time::Duration}; +use core::time::Duration; use clap::Parser; use error_stack::{Report, ResultExt as _}; -use hash_graph_api::rest::{auth::build_authentication_provider, rate_limit::RateLimitConfig}; use hash_graph_atlas::cli::{self, PasswordString}; -use hash_graph_postgres_store::store::{ - DatabaseConnectionInfo, DatabasePoolConfig, PostgresStorePool, PostgresStoreSettings, -}; -use hash_graph_store::filter::protection::PropertyProtectionFilterConfig; -use hash_telemetry::Telemetry; -use opentelemetry::metrics::Meter; +use hash_graph_postgres_store::store::DatabaseConnectionInfo; use reqwest::Client; -use tokio::{net::TcpListener, signal, time::timeout}; -use tokio_postgres::NoTls; -use tokio_util::sync::CancellationToken; +use tokio::time::timeout; use crate::{ error::{GraphError, HealthcheckError}, subcommand::{ - HealthcheckArgs, ServerLifecycle, - server::{KratosSessionAuthConfig, TemporalConfig, create_temporal_client}, - wait_healthcheck, + HealthcheckArgs, wait_healthcheck, }, }; @@ -47,56 +36,12 @@ pub struct AtlasArgs { /// The atlas operations. #[derive(Debug, clap::Subcommand)] pub enum AtlasCommand { - /// Serves the read API over the root's active generation. - Serve(Box), /// Fits one generation over the live store and activates it on admission. Fit(Box), /// Probes the liveness endpoint of a serving atlas process. Healthcheck(AtlasHealthcheckArgs), } -/// CLI arguments for `atlas serve`. -#[derive(Debug, Parser)] -pub struct AtlasServeArgs { - #[clap(flatten)] - pub address: AtlasAddress, - - #[clap(flatten)] - pub root: cli::RootArgs, - - #[clap(flatten)] - pub serve: cli::ServeArgs, - - #[clap(flatten)] - pub db_info: DatabaseConnectionInfo, - - #[clap(flatten)] - pub db_pool_config: DatabasePoolConfig, - - #[clap(flatten)] - pub temporal: TemporalConfig, - - #[clap(flatten)] - pub session_auth: KratosSessionAuthConfig, - - /// Shared secret internal services present to act on behalf of an actor. - /// - /// Sent as the `Authorization: HASH-Service ` credential next to - /// `X-Authenticated-User-Actor-Id`. - #[clap(long, env = "HASH_GRAPH_SERVICE_SECRET", hide_env_values = true)] - pub service_secret: PasswordString, - - #[clap(flatten)] - pub rate_limit: RateLimitConfig, - - /// Disables filter protection that prevents enumeration attacks on protected properties. - /// - /// The flag matches the server subcommand's, so the embedding exclusions the atlas ensures - /// carry stay equal to the exclusions the store's own workflow starts carry. - #[clap(long, env = "HASH_GRAPH_SKIP_FILTER_PROTECTION")] - pub skip_filter_protection: bool, -} - /// CLI arguments for `atlas fit`. #[derive(Debug, Parser)] pub struct AtlasFitArgs { @@ -128,102 +73,6 @@ pub struct AtlasHealthcheckArgs { pub timeout: Option, } -struct AtlasTelemetry { - meter: Meter, -} - -/// Runs the atlas server, shutting down when `shutdown` is cancelled. -async fn run_atlas( - args: AtlasServeArgs, - telemetry: &AtlasTelemetry, - shutdown: CancellationToken, -) -> Result<(), Report> { - // Before running anything, make sure that the configuration is valid. - let session_auth = args.session_auth.into_provider_config()?; - - // The same filter-protection configuration the server subcommand parses, so the embedding - // exclusions the staging arm's ensures carry stay equal to the exclusions the store's own - // workflow starts carry. - let filter_protection = if args.skip_filter_protection { - PropertyProtectionFilterConfig::new() - } else { - PropertyProtectionFilterConfig::hash_default() - }; - let exclusions = filter_protection.embedding_exclusions().clone(); - - let service_secret = cli::SecretString::from(args.service_secret); - - // A single pool serves the whole process, so the detail trailers, the permission - // resolution, and the credential chain's actor lookups behind every request read through - // shared connections and none waits on a connection another holds. - let pool = Arc::new( - PostgresStorePool::new( - &args.db_info, - &args.db_pool_config, - NoTls, - PostgresStoreSettings { - filter_protection, - ..PostgresStoreSettings::default() - }, - ) - .await - .change_context(GraphError)?, - ); - - // Absent a configured Temporal server, arrivals stage and never ensure, which fails closed. - let workflow = - create_temporal_client(&args.temporal) - .await? - .map(|client| cli::EmbeddingWorkflow { - temporal: client, - exclusions, - }); - - // The chain the REST router authenticates with, so a credential means the same thing on - // every route of the deployment: a Kratos session, or the service secret with the actor it - // delegates. Cloudflare Access fronts the admin server's operator routes, so no JWT - // verifier enters this chain. - let provider = Arc::new(build_authentication_provider( - session_auth, - None, - service_secret.clone().into_unguarded().as_ref().to_owned(), - &pool, - &telemetry.meter, - )); - - // Every request answers under the scope of the actor it names. - let router = cli::ServeCommand::new(args.root, args.serve) - .run(cli::ServeOptions { - provider, - service_secret, - rate_limit: (&args.rate_limit).into(), - pool, - visibility: cli::VisibilityLimits::default(), - workflow, - }) - .map_err(Report::new) - .change_context(GraphError)?; - - let listener = TcpListener::bind((&*args.address.atlas_host, args.address.atlas_port)) - .await - .change_context(GraphError)?; - - tracing::info!( - "Listening on port {}", - listener.local_addr().change_context(GraphError)?.port() - ); - - axum::serve( - listener, - router.into_make_service_with_connect_info::(), - ) - .with_graceful_shutdown(shutdown.cancelled_owned()) - .await - .change_context(GraphError)?; - - Ok(()) -} - /// Renders one fit's verdict, the `atlas fit` subcommand's product. #[expect( clippy::print_stdout, @@ -234,16 +83,8 @@ fn print_verdict(verdict: &cli::FitVerdict) { } /// Standalone `atlas` subcommand entrypoint. -#[expect( - clippy::integer_division_remainder_used, - reason = "False positive on tokio::select!" -)] -#[expect( - clippy::exit, - reason = "Force shutdown on double ctrl-c is intentional" -)] -pub async fn atlas(args: AtlasArgs, telemetry: &Telemetry) -> Result<(), Report> { - let serve_args = match args.command { +pub async fn atlas(args: AtlasArgs) -> Result<(), Report> { + match args.command { AtlasCommand::Fit(fit_args) => { let mut client = cli::connect(&fit_args.db_info.url()) .await @@ -256,68 +97,18 @@ pub async fn atlas(args: AtlasArgs, telemetry: &Telemetry) -> Result<(), Report< .change_context(GraphError)?; print_verdict(&verdict); - return Ok(()); - } - AtlasCommand::Healthcheck(healthcheck_args) => { - return wait_healthcheck( - || healthcheck(healthcheck_args.address.clone()), - &HealthcheckArgs { - healthcheck: true, - wait: healthcheck_args.wait, - timeout: healthcheck_args.timeout, - }, - ) - .await - .change_context(GraphError); - } - AtlasCommand::Serve(serve_args) => serve_args, - }; - - let telemetry = AtlasTelemetry { - meter: telemetry.meter("Graph Atlas API"), - }; - - let lifecycle = ServerLifecycle::new(); - let shutdown = lifecycle.shutdown.clone(); - lifecycle.spawn("Atlas", async move { - run_atlas(*serve_args, &telemetry, shutdown).await - }); - - // Wait for shutdown signal or unexpected server exit - let aborted = tokio::select! { - result = signal::ctrl_c() => { - match result { - Ok(()) => false, - Err(error) => { - tracing::error!("Failed to install Ctrl+C handler: {error}"); - true - } - } + Ok(()) } - () = lifecycle.abort.cancelled() => { - tracing::error!("Atlas exited unexpectedly"); - true - } - }; - - // Double ctrl-c for force shutdown - tokio::select! { - () = lifecycle.shutdown_and_wait() => {} - result = signal::ctrl_c() => { - if let Err(error) = result { - tracing::error!("Failed to install Ctrl+C handler: {error}"); - } - tracing::warn!("Forced shutdown"); - std::process::exit(1); - } - } - - tracing::info!("Shutdown complete"); - - if aborted { - Err(GraphError.into()) - } else { - Ok(()) + AtlasCommand::Healthcheck(healthcheck_args) => wait_healthcheck( + || healthcheck(healthcheck_args.address.clone()), + &HealthcheckArgs { + healthcheck: true, + wait: healthcheck_args.wait, + timeout: healthcheck_args.timeout, + }, + ) + .await + .change_context(GraphError), } } @@ -342,6 +133,8 @@ async fn healthcheck(address: AtlasAddress) -> Result<(), Report block_on( - async |telemetry| atlas(*args, telemetry).await, + async |_telemetry| atlas(*args).await, "Atlas", tracing_config, worker_threads, diff --git a/libs/@local/graph/atlas/package.json b/libs/@local/graph/atlas/package.json index 57e55851c4a..48cdc0d02d4 100644 --- a/libs/@local/graph/atlas/package.json +++ b/libs/@local/graph/atlas/package.json @@ -8,7 +8,6 @@ "doc:dependency-diagram": "cargo run -p hash-repo-chores -- dependency-diagram --output docs/dependency-diagram.mmd --root hash-graph-atlas --root-deps-and-dependents --link-mode non-roots --include-dev-deps --include-build-deps --logging-console-level info", "fix:clippy": "just clippy --fix", "lint:clippy": "just clippy", - "test:integration": "cargo nextest run --package hash-graph-atlas --all-features --filterset 'kind(test)'", "test:miri": "cargo miri test --lib -- ::miri::", "test:unit": "mise run test:unit @rust/hash-graph-atlas -- --filterset 'not kind(test)'" }, diff --git a/libs/@local/graph/atlas/src/api/authorization.rs b/libs/@local/graph/atlas/src/api/authorization.rs deleted file mode 100644 index 18663561f66..00000000000 --- a/libs/@local/graph/atlas/src/api/authorization.rs +++ /dev/null @@ -1,424 +0,0 @@ -//! The caller and its authority token: the actor extractor, and the `Atlas-Authority` header's two -//! readings. -//! -//! The manifest response issues the token and every data request presents it back in the same -//! header. The judgment is [`TokenAuthority::open`]'s. Extracting [`Scope`] is the data routes' -//! reading: admission only under a fresh token naming the requesting actor, every refusal one -//! uniform `401`. Extracting `Option` is the manifest's reading: an absent token is a -//! fresh bootstrap and the reading forgives the window, while the tag and the actor stay binding. -//! Both readings resolve the caller through [`Actor`], the authentication middleware's resolution. -//! -//! [`TokenAuthority::open`]: crate::serve::authorization::TokenAuthority::open - -use std::time::SystemTime; - -use aide::{generate::GenContext, openapi, operation::OperationInput}; -use axum::{ - extract::{FromRequestParts, OptionalFromRequestParts}, - http::request::Parts, -}; -use hash_middleware::authentication::{AuthenticatedActorId, AuthenticationRejection}; -use type_system::principal::actor::ActorId; - -use super::{ - AppState, headers, - problem::{Problem, unauthorized}, -}; -use crate::{ - integrity::HexBytes, - serve::authorization::{Scope, TOKEN_BYTES, TokenAuthority}, -}; - -/// A request's caller resolution, cached after the first extraction. -/// -/// A test inserts one ahead of the middleware, whose own resolution type is private to its crate. -#[derive(Clone)] -struct ActorCache(Result); - -/// The authenticated caller. -/// -/// The authentication middleware resolves the caller ahead of this router, and this extractor -/// reads that resolution, rejecting an anonymous one rather than letting it reach a handler. -#[derive(Debug, Clone, Copy)] -pub(super) struct Actor(pub ActorId); - -impl FromRequestParts for Actor -where - S: Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request_parts(parts: &mut Parts, state: &S) -> Result { - let resolution = if let Some(ActorCache(cached)) = parts.extensions.get::() { - cached.clone() - } else { - let resolution = - >::from_request_parts(parts, state) - .await - .map(|AuthenticatedActorId(actor)| Self(actor)); - - parts.extensions.insert(ActorCache(resolution.clone())); - resolution - }; - - match resolution { - Ok(actor) => Ok(actor), - Err(AuthenticationRejection::Misconfigured) => Err(Problem::internal( - "`Actor` extracted on a route without the authentication middleware", - "the caller's authentication was never resolved", - )), - Err(AuthenticationRejection::Authentication { - metrics: _, - ref report, - recorded: _, - }) => Err(report.current_context().into()), - } - } -} - -/// Adds nothing per operation: the document root declares the credential surface. -/// -/// The authentication middleware resolves credentials ahead of this router and cannot write into -/// this document, so [`router`](super::router) declares the schemes it accepts once, as the -/// document's own security requirements. -impl OperationInput for Actor {} - -/// The token authority slice of a router state. -/// -/// Both [`Scope`] readings judge presentations against the authority alone, so they extract from -/// any state that supplies one, the full application state or a bare authority. -pub(super) trait TokenState { - /// The authority's randomness source. - type Rng; - - /// The authority judging every presentation. - fn tokens(&self) -> &TokenAuthority; -} - -impl TokenState for AppState { - type Rng = R; - - fn tokens(&self) -> &TokenAuthority { - &self.tokens - } -} - -/// Admits one data request: the sealed [`Scope`] under a fresh, actor-matching token. -/// -/// Admission precedes every resolution (an unauthorized request costs one AEAD open and never a -/// store round trip) and precedes the handler body's generation check. A retired generation's -/// token fails its tag under the current key and answers `401` here, while a current token -/// presented at a route naming a retired generation passes admission and finds the `404` in the -/// handler. A `404` from a data route therefore means the caller's pin is stale and its token is -/// good. -/// -/// Every refusal is the one uniform `401` problem. An absent header and a value outside the -/// codec refuse silently. A refusal from [`TokenAuthority::open`] also reaches the server log. -/// -/// [`TokenAuthority::open`]: crate::serve::authorization::TokenAuthority::open -impl FromRequestParts for Scope -where - S: TokenState + Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request_parts(parts: &mut Parts, state: &S) -> Result { - let Actor(actor) = Actor::from_request_parts(parts, state).await?; - - parts - .headers - .get(headers::AUTHORITY) - .and_then(|header| header.to_str().ok()?.parse::>().ok()) - .and_then(|token| { - state - .tokens() - .open(&token.into_inner(), actor, SystemTime::now()) - .inspect_err(|error| tracing::warn!(%error, "unable to open token")) - .ok() - }) - .ok_or_else(unauthorized) - } -} - -/// Reads one presentation for the manifest's renewal: window forgiven, tag and actor binding. -/// -/// An absent header is [`None`], a fresh bootstrap. A refused presentation (outside the codec, -/// failing the tag, or naming another actor) is the uniform `401` problem rather than a fresh -/// bootstrap, so a corrupted retention or an actor switch cannot become another view without -/// notice. [`TokenAuthority::continuity`] is the judgment, so an expired token still carries its -/// sealed view into the fresh token the manifest issues. -/// -/// Refusal fires at extraction, ahead of the handler body's generation check. A stale pin with a -/// refused token answers `401`, and the token-less follow-up finds the generation's `404` and the -/// re-pin there. -/// -/// [`TokenAuthority::continuity`]: crate::serve::authorization::TokenAuthority::continuity -impl OptionalFromRequestParts for Scope -where - S: TokenState + Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request_parts( - parts: &mut Parts, - state: &S, - ) -> Result, Self::Rejection> { - let Actor(actor) = Actor::from_request_parts(parts, state).await?; - - let Some(header) = parts.headers.get(headers::AUTHORITY) else { - return Ok(None); - }; - - header - .to_str() - .ok() - .and_then(|text| text.parse::>().ok()) - .and_then(|token| { - state - .tokens() - .continuity(&token.into_inner(), actor) - .inspect_err(|error| tracing::warn!(%error, "unable to carry token")) - .ok() - }) - .map_or_else(|| Err(unauthorized()), |scope| Ok(Some(scope))) - } -} - -impl OperationInput for Scope { - /// Documents the presented token as a required request header. - /// - /// Required is the data routes' contract: extraction admits before anything resolves, so a - /// route taking the sealed scope answers nothing without a token. The manifest takes - /// `Option`, and aide derives its documentation from this impl, flipping the parameter - /// to optional - the manifest's whole distinction from a data route. - fn operation_input(_ctx: &mut GenContext, operation: &mut openapi::Operation) { - operation - .parameters - .push(openapi::ReferenceOr::Item(headers::presented_authority())); - } -} - -#[cfg(test)] -mod tests { - use core::{assert_matches, time::Duration}; - use std::time::SystemTime; - - use aide::{openapi::Operation, transform::TransformOperation}; - use axum::{ - extract::{FromRequestParts, OptionalFromRequestParts}, - http::{HeaderValue, Request, request::Parts}, - }; - use futures::executor::block_on; - use rand::{SeedableRng as _, rngs::ChaCha20Rng}; - use type_system::principal::actor::{ActorId, UserId}; - use uuid::Uuid; - - use super::{Actor, ActorCache, Problem, TokenState, headers}; - use crate::{ - api::visibility::Visibility, - integrity::{HexBytes, SecretHexBytes}, - serve::{ - CutOffset, - authorization::{Scope, TOKEN_BYTES, TokenAuthority}, - }, - }; - - /// The bare authority as a test's whole state. - impl TokenState for TokenAuthority { - type Rng = ChaCha20Rng; - - fn tokens(&self) -> &Self { - self - } - } - - /// Renders the operation input one extractor documents. - fn emitted_input() -> serde_json::Value { - let mut operation = Operation::default(); - let _documented = TransformOperation::new(&mut operation).input::(); - - serde_json::to_value(&operation).expect("an operation serializes") - } - - /// The fixture issue time, a round wall-clock second. - fn issued_at() -> SystemTime { - SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000) - } - - /// The fixture authority, accepting tokens for ten minutes. - fn authority() -> TokenAuthority { - TokenAuthority::new( - "07".repeat(32) - .parse() - .expect("64 hexadecimal digits name a generation"), - &SecretHexBytes::new([0x5A; 32]), - Duration::from_mins(10), - None, - ChaCha20Rng::from_seed([7; 32]), - ) - } - - /// The actor identity `actor` names. - fn actor(actor: u128) -> ActorId { - ActorId::User(UserId::new(Uuid::from_u128(actor))) - } - - /// The authority header presenting a token for `presenter`, issued at `issued_at`. - fn issued( - tokens: &TokenAuthority, - presenter: ActorId, - issued_at: SystemTime, - ) -> HeaderValue { - let token = tokens - .issue(Scope::new(presenter, None, CutOffset::ZERO), issued_at) - .expect("the seeded generator is infallible"); - - HeaderValue::try_from(HexBytes::new(token).to_string()) - .expect("hexadecimal is a valid header value") - } - - /// One request's parts, with `resolution` cached as the middleware's outcome and `token` - /// presented in the authority header. - fn parts(resolution: ActorId, token: Option<&HeaderValue>) -> Parts { - let mut request = Request::builder().uri("/"); - if let Some(token) = token { - request = request.header(headers::AUTHORITY, token.clone()); - } - - let (mut parts, ()) = request - .body(()) - .expect("a static URI and a hexadecimal header form a valid request") - .into_parts(); - parts.extensions.insert(ActorCache(Ok(Actor(resolution)))); - - parts - } - - /// Drives a data route's admission reading with the authority as the whole state. - fn admit( - tokens: &TokenAuthority, - resolution: ActorId, - token: Option<&HeaderValue>, - ) -> Result> { - block_on(>::from_request_parts( - &mut parts(resolution, token), - tokens, - )) - } - - /// Drives the manifest's continuity reading with the authority as the whole state. - fn read( - tokens: &TokenAuthority, - resolution: ActorId, - token: Option<&HeaderValue>, - ) -> Result, Problem<'static>> { - block_on(>::from_request_parts( - &mut parts(resolution, token), - tokens, - )) - } - - /// An absent header reads as a fresh bootstrap, never as a refusal. - #[test] - fn absent_token_reads_as_a_fresh_bootstrap() { - assert_matches!(read(&authority(), actor(11), None), Ok(None)); - } - - /// A header outside the codec reads as refused, never as a fresh bootstrap. - /// - /// The refused answer tells a client its retention corrupted, where a bootstrap would - /// silently hand it another view. - #[test] - fn garbage_header_reads_as_refused() { - let tokens = authority(); - - for garbage in ["", "zz", &"ab".repeat(TOKEN_BYTES)] { - let header = garbage - .parse::() - .expect("a visible ASCII string"); - - assert_matches!( - read(&tokens, actor(11), Some(&header)), - Err(_), - "a garbage header did not read as refused" - ); - } - } - - /// Another actor's authentic token reads as refused, never as a fresh bootstrap. - /// - /// An actor switch that leaves a stale token behind answers with a refusal rather than a - /// fresh view under a `200` the client would read as continuity. - #[test] - fn foreign_actors_token_reads_as_refused() { - let tokens = authority(); - let header = issued(&tokens, actor(11), issued_at()); - - assert_matches!(read(&tokens, actor(12), Some(&header)), Err(_)); - } - - /// Admission enforces the window the continuity reading forgives. - /// - /// The same presentation diverges at the two readings: past the enforced window a data - /// request refuses while the manifest still carries the sealed view into the fresh token it - /// issues. Issue times pin to the wall clock admission reads, so the fresh token stays inside - /// the ten-minute window and the expired one stays outside it. - #[test] - fn expired_token_refuses_at_admission_yet_reads_as_carried() { - let tokens = authority(); - let now = SystemTime::now(); - let expired = issued(&tokens, actor(11), now - Duration::from_mins(11)); - let fresh = issued(&tokens, actor(11), now); - - assert_matches!( - admit(&tokens, actor(11), Some(&expired)), - Err(_), - "an expired token admitted a data request" - ); - assert_matches!(read(&tokens, actor(11), Some(&expired)), Ok(Some(_))); - assert_matches!(admit(&tokens, actor(11), Some(&fresh)), Ok(_)); - } - - /// The emitted OpenAPI documents the token optional on the manifest, required on a data route. - /// - /// The readings differ in exactly one emitted field, and a generated client's behaviour follows - /// it. A required header makes the token a precondition of the call, while an optional one - /// leaves the bootstrap reachable. Asserted on the serialized parameter rather than on the - /// builder call, because that is what a generator reads. - #[test] - fn presented_token_is_documented_by_reading() { - let manifest = emitted_input::>(); - let data_route = emitted_input::(); - - for (route, emitted) in [("manifest", &manifest), ("data route", &data_route)] { - let parameter = &emitted["parameters"][0]; - - assert_eq!( - parameter["in"], "header", - "the {route} parameter is not a header" - ); - assert_eq!( - parameter["name"], - headers::AUTHORITY_DOCUMENTED, - "the {route} parameter names another header" - ); - assert!( - parameter["schema"].is_object(), - "the {route} parameter carries no schema" - ); - } - - // The emitted form omits `required` when it is false. - assert!( - !manifest["parameters"][0]["required"] - .as_bool() - .unwrap_or(false), - "the manifest documents the token as required, refusing its own bootstrap" - ); - assert_eq!( - data_route["parameters"][0]["required"], - serde_json::Value::Bool(true), - "a data route documents the token as optional" - ); - } -} diff --git a/libs/@local/graph/atlas/src/api/clause.rs b/libs/@local/graph/atlas/src/api/clause.rs deleted file mode 100644 index 840b2c0230a..00000000000 --- a/libs/@local/graph/atlas/src/api/clause.rs +++ /dev/null @@ -1,226 +0,0 @@ -//! The OpenAPI response clauses that more than one route states. -//! -//! A documented response is part of the contract, so two routes that answer one problem the same -//! way state it from one place. Divergent wording for an identical refusal reads as a difference -//! the server does not have. Each helper transforms one operation and composes through aide's -//! `with`, which keeps a route's documentation one chain naming the clauses it states. A clause -//! only one route answers stays in that route's own module. - -use aide::{ - openapi, - transform::{TransformOpenApi, TransformOperation}, - util::iter_operations_mut, -}; - -use super::{headers, problem::Problem}; - -/// The `401` the authentication middleware answers before any route runs. -const UNAUTHENTICATED: &str = "`unauthenticated`: the call names no valid actor"; - -/// States on every operation what the middleware in front of the router answers. -/// -/// The request budgets and the authentication layer wrap the whole router, so their `429` and -/// `401` belong to every operation and are stated once here rather than by each route. An -/// operation that documents a `401` of its own keeps it and gains the middleware's cause beside -/// it, because one operation documents one `401`. -pub(super) fn middleware(mut api: TransformOpenApi<'_>) -> TransformOpenApi<'_> { - let Some(paths) = api.inner_mut().paths.as_mut() else { - return api; - }; - for path in paths.paths.values_mut() { - let openapi::ReferenceOr::Item(path) = path else { - continue; - }; - for (_, operation) in iter_operations_mut(path) { - let mut operation = TransformOperation::new(operation).with(too_many_requests); - let existing = operation - .inner_mut() - .responses - .as_mut() - .and_then(|responses| responses.responses.get_mut(&openapi::StatusCode::Code(401))); - match existing { - Some(openapi::ReferenceOr::Item(response)) => { - response.description = - format!("{UNAUTHENTICATED}, or {}", response.description); - } - Some(openapi::ReferenceOr::Reference { .. }) => {} - None => { - let _: TransformOperation<'_> = operation.with(unauthenticated); - } - } - } - } - api -} - -/// States the middleware's `401` on an operation that documents no `401` of its own. -fn unauthenticated(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation - .response_with::<401, Problem<'static>, _>(|response| response.description(UNAUTHENTICATED)) -} - -/// States the `429` the request budgets answer, with the `Retry-After` header it carries. -fn too_many_requests(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation.response_with::<429, Problem<'static>, _>(|mut response| { - response - .inner() - .headers - .insert("Retry-After".to_owned(), headers::retry_after()); - response.description( - "`too-many-requests`: the caller is over its per-address or per-actor budget; \ - `Retry-After` states whole seconds until it admits again", - ) - }) -} - -/// States the `401` every authority-bearing route answers alike. -/// -/// One uniform refusal covers every cause, so the documented remedy is all a caller learns from it. -pub(super) fn unauthorized(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation.response_with::<401, Problem<'static>, _>(|response| { - response.description("`unauthorized`: no valid authority token; re-fetch the manifest") - }) -} - -/// States the `422` every body-bearing route answers alike. -/// -/// The body extractor keeps the framework's own status, so well-formed JSON that is not the -/// operation's shape - a mistyped member or an unknown one - answers `invalid-body` at 422, -/// while a body that is not JSON at all answers it at 400. -pub(super) fn invalid_body_data(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation.response_with::<422, Problem<'static>, _>(|response| { - response.description( - "`invalid-body`: well-formed JSON that is not this operation's shape - a mistyped \ - member or an unknown one", - ) - }) -} - -/// States the catch-all clause covering what the enumerated responses do not. -pub(super) fn any_problem(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation.default_response_with::, _>(|response| { - response.description("any other problem document; `internal` marks a server-side failure") - }) -} - -/// Marks the operation's request body optional. -/// -/// A declared body documents as required, including the `Option>` a route reading an absent -/// body as the all-defaults request accepts. -pub(super) fn optional_body(mut operation: TransformOperation<'_>) -> TransformOperation<'_> { - if let Some(body) = body_mut(&mut operation) { - body.required = false; - } - - operation -} - -/// Describes the operation's request body, naming what the body is for. -pub(super) fn describe_body( - description: &'static str, -) -> impl FnOnce(TransformOperation<'_>) -> TransformOperation<'_> { - move |mut operation| { - if let Some(body) = body_mut(&mut operation) { - body.description = Some(description.to_owned()); - } - - operation - } -} - -/// The operation's declared request body, when it declares one inline. -/// -/// A body reaching the document as a component reference carries no per-operation text to edit, so -/// there is nothing to describe and nothing to mark. -fn body_mut<'body>( - operation: &'body mut TransformOperation<'_>, -) -> Option<&'body mut openapi::RequestBody> { - operation.inner_mut().request_body.as_mut()?.as_item_mut() -} - -#[cfg(test)] -mod tests { - use aide::{ - openapi, - transform::{TransformOpenApi, TransformOperation}, - }; - - use super::{UNAUTHENTICATED, middleware, unauthorized}; - - /// One path with a bare `get` and a `post` that already states its own `401`. - fn document() -> openapi::OpenApi { - let mut post = openapi::Operation::default(); - let _: TransformOperation<'_> = TransformOperation::new(&mut post).with(unauthorized); - let mut api = openapi::OpenApi { - paths: Some(openapi::Paths { - paths: core::iter::once(( - "/route".to_owned(), - openapi::ReferenceOr::Item(openapi::PathItem { - get: Some(openapi::Operation::default()), - post: Some(post), - ..openapi::PathItem::default() - }), - )) - .collect(), - ..openapi::Paths::default() - }), - ..openapi::OpenApi::default() - }; - let _: TransformOpenApi<'_> = middleware(TransformOpenApi::new(&mut api)); - api - } - - fn response<'document>( - api: &'document openapi::OpenApi, - method: &str, - status: u16, - ) -> Option<&'document openapi::Response> { - let paths = api.paths.as_ref()?; - let openapi::ReferenceOr::Item(path) = paths.paths.get("/route")? else { - return None; - }; - let operation = match method { - "get" => path.get.as_ref()?, - "post" => path.post.as_ref()?, - _ => return None, - }; - operation - .responses - .as_ref()? - .responses - .get(&openapi::StatusCode::Code(status))? - .as_item() - } - - #[test] - fn too_many_requests_on_every_method() { - let api = document(); - for method in ["get", "post"] { - let response = response(&api, method, 429).expect("every method states 429"); - assert!( - response.headers.contains_key("Retry-After"), - "{method}: the 429 documents its Retry-After header" - ); - } - } - - #[test] - fn unauthenticated_inserted_where_absent() { - let api = document(); - let response = response(&api, "get", 401).expect("a route with no 401 gains one"); - assert_eq!(response.description, UNAUTHENTICATED); - } - - #[test] - fn unauthenticated_merged_into_own_401() { - let api = document(); - let response = response(&api, "post", 401).expect("the route's own 401 stays"); - assert_eq!( - response.description, - format!( - "{UNAUTHENTICATED}, or `unauthorized`: no valid authority token; re-fetch the \ - manifest" - ) - ); - } -} diff --git a/libs/@local/graph/atlas/src/api/current.rs b/libs/@local/graph/atlas/src/api/current.rs deleted file mode 100644 index 39fc49c90ac..00000000000 --- a/libs/@local/graph/atlas/src/api/current.rs +++ /dev/null @@ -1,55 +0,0 @@ -//! `GET /v1/atlas/current`: the one mutable read. - -use aide::{axum::IntoApiResponse, transform::TransformOperation}; -use axum::{Json, extract::State, http::header}; - -use super::{AppState, headers}; -use crate::file::generation::GenerationId; - -/// The operation's description. -const DESCRIPTION: &str = "Returns the generation this server serves, pinned at startup. - -Every other route's geometry and configuration are pinned per generation. Detail provenance varies \ - by route: tile detail is generation-local, while edges and locate \ - combine generation payloads with request-time store state. This \ - pointer is the only generation read that changes. Re-read it whenever \ - any route answers `unknown-generation`, then retry against the \ - returned generation."; - -/// The `current` document: the one mutable read. -#[derive( - Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize, schemars::JsonSchema, -)] -struct CurrentResponse { - /// The active generation's sha256 identity. - generation: GenerationId, -} - -/// `GET /v1/atlas/current`: the one mutable read. -pub(super) async fn handler(State(state): State>) -> impl IntoApiResponse { - ( - [(header::CACHE_CONTROL, headers::REVALIDATE)], - Json(CurrentResponse { - generation: state.atlas.generation(), - }), - ) -} - -/// Documents the operation. -pub(super) fn document(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation - .id("current") - .summary("The active generation") - .description(DESCRIPTION) - .response_with::<200, Json, _>(|mut response| { - response.inner().headers.insert( - "Cache-Control".to_owned(), - headers::cache_control( - headers::REVALIDATE, - "the one mutable read revalidates on every use - a stale pointer is exactly \ - the failure this route exists to prevent", - ), - ); - response.description("the generation this process serves") - }) -} diff --git a/libs/@local/graph/atlas/src/api/edges.rs b/libs/@local/graph/atlas/src/api/edges.rs deleted file mode 100644 index 9f5787c17b3..00000000000 --- a/libs/@local/graph/atlas/src/api/edges.rs +++ /dev/null @@ -1,204 +0,0 @@ -//! `POST /v1/atlas/edges/{generation}/{variant}`. -//! -//! The edges among the listed tiles' delivered rows, as `SALTILEE` bytes. - -use alloc::sync::Arc; -use core::panic::AssertUnwindSafe; - -use aide::transform::TransformOperation; -use axum::{extract::State, http::StatusCode}; -use hashql_core::{ - collections::FastHashMap, - id::{IdSlice, IdVec}, -}; -use tokio::sync::oneshot; -use type_system::ontology::id::VersionedUrl; - -use super::{ - AppState, clause, - extract::{Body, Generation, VariantPath}, - problem::{Problem, ProblemType, reject_generation, reject_variant}, - saltile::{Saltile, spawn}, - visibility::{Visibility, view_problem}, -}; -use crate::{ - postgres::id::ArchivedOntologyTypeUuid, - serve::{ - EdgesError, EdgesRequest, - hydrate::{DetailError, EdgesStore, TypeSlot, TypeUrlResolver as _}, - }, -}; - -/// The operation's description. -const DESCRIPTION: &str = - "Returns the edges among the points the listed tiles deliver, as a `SALTILEE` binary envelope. - -An edge is included exactly when both of its endpoints lie in the union of the listed tiles' \ - delivered sets, so one request listing the whole viewport also returns the edges that cross \ - between its tiles. Delivery is the tile route's level-of-detail delivery, not spatial \ - containment: an edge is absent while either endpoint sits below the listed zoom's cut, and \ - appears once the viewport or zoom reaches that endpoint. - -The JSON body is required; the manifest's `limits.edgesTiles` caps the tile list. - -The response's three columns - sources, targets, and `EDGE_IDS` (each edge's 32-byte link entity \ - id, web uuid then entity uuid) - are ordered ascending by identity bytes, independent of the \ - tile list's order: identical requests yield identical geometry bytes, and the order is \ - verifiable from the `EDGE_IDS` column alone. Edges have no row id of their own; the entity \ - id is an edge's identity on every route. - -When the server's edge cap truncates the set, the response keeps the edges whose worse endpoint \ - ranks best, and the HEAD's `complete` key reads `false`. - -`detail: \"auxiliary\"` adds the detail trailer (`\"minimal\"`, the default, sends the columns \ - alone). Every trailer value reads from its entity's currently served edition, which may \ - trail the newest edition by up to 65 seconds. - -Entities and links that arrive after the serving generation's fit can also appear. A post-fit link \ - is an ordinary edge row wherever both of its endpoints deliver, merged into the same \ - identity order, and its label and representative type come from the display the server \ - captured at the link's currently served edition rather than the generation. An endpoint \ - placed since the fit takes a session-scoped row id, and such ids die with the serving \ - session that minted them, exactly as locate and translate describe. - -Filtering binds at the manifest. This body has no `filter` field, and an unknown member is \ - rejected as `invalid-body`. -"; - -/// `POST /v1/atlas/edges/{generation}/{variant}`. -/// -/// The edges among the listed tiles' delivered rows, as `SALTILEE` bytes. The tiles list is the -/// request's subject, so the route requires the body. -pub(super) async fn handler( - State(state): State>, - visibility: Visibility, - Generation(VariantPath { - generation, - variant, - }): Generation, - Body(request): Body, -) -> Result> { - reject_generation(&state, generation)?; - reject_variant(&variant)?; - - // The whole pipeline is one synchronous call on a rayon worker; only the store order and its - // answer cross back here, where the connections live. - let atlas = Arc::clone(&state.atlas); - let limits = state.limits.edges; - let (order_sender, order_receiver) = oneshot::channel(); - let (answer_sender, answer_receiver) = oneshot::channel(); - let store = ChannelEdgesStore { - order: order_sender, - answer: answer_receiver, - }; - - let (result, ()) = tokio::join!( - spawn(AssertUnwindSafe(move || { - let view = visibility.view(&atlas)?; - - atlas.edges(&request, limits, view, store) - })), - async { - // An order never arrives when the request skips the trailer, rejects, or panics first. - let Ok(types) = order_receiver.await else { - return; - }; - - let types = ArchivedOntologyTypeUuid::into_slice(types.as_raw()); - - let answer = state - .type_urls - .resolve(types.iter().copied()) - .await - .map(|pairs| { - let resolved: FastHashMap<_, _> = pairs.into_iter().collect(); - - types - .iter() - .map(|uuid| resolved.get(uuid).cloned()) - .collect() - }); - - let _: Result<(), _> = answer_sender.send(answer); - }, - ); - - let bytes = match result? { - Ok(bytes) => bytes, - Err(error @ EdgesError::Tiles { .. }) => { - return Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::TooManyTiles, - error.to_string(), - )); - } - Err(error @ (EdgesError::Depth { .. } | EdgesError::Grid { .. })) => { - return Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidCoordinate, - error.to_string(), - )); - } - // A stale sealed offset answers the uniform refusal. A mismatched pair or width names - // an input this process produced and answers the internal problem. - Err(EdgesError::View(error)) => return Err(view_problem(error)), - Err(error @ EdgesError::Details(_)) => { - return Err(Problem::internal(error, "the detail hydration failed")); - } - }; - - Ok(Saltile::new(bytes)) -} - -/// The transport's edges store, carrying one order out to the handler and one answer back in. -struct ChannelEdgesStore { - order: oneshot::Sender>, - answer: oneshot::Receiver>, DetailError>>, -} - -impl EdgesStore for ChannelEdgesStore { - fn hydrate( - self, - types: &IdSlice, - ) -> Result>, DetailError> { - // The channel is the one boundary that owns the identities, so the list materializes - // here and nowhere earlier. - self.order - .send(types.iter().copied().collect()) - .map_err(|_order| DetailError::Disconnected)?; - - self.answer - .blocking_recv() - .map_err(|_closed| DetailError::Disconnected)? - } -} - -/// Documents the operation. -pub(super) fn document(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation - .id("edges") - .summary("The edges among the listed tiles' delivered rows, as SALTILEE envelope bytes") - .description(DESCRIPTION) - .with(clause::describe_body( - "the edges request; the `tiles` list is the request's subject", - )) - .response_with::<200, Saltile, _>(|response| { - response.description( - "a `SALTILEE` envelope: the qualifying edges, ascending by link-entity identity", - ) - }) - .response_with::<400, Problem<'static>, _>(|response| { - response.description( - "`too-many-tiles`, `invalid-generation`, `missing-body`, `invalid-body` (a body \ - that is not JSON), or `invalid-coordinate`", - ) - }) - .with(clause::invalid_body_data) - .with(clause::unauthorized) - .response_with::<404, Problem<'static>, _>(|response| { - response.description( - "`unknown-generation` or `unknown-variant`: re-read `current` and retry", - ) - }) - .with(clause::any_problem) -} diff --git a/libs/@local/graph/atlas/src/api/extract.rs b/libs/@local/graph/atlas/src/api/extract.rs deleted file mode 100644 index b7228d49a52..00000000000 --- a/libs/@local/graph/atlas/src/api/extract.rs +++ /dev/null @@ -1,419 +0,0 @@ -//! Extractors whose rejections are problem documents. -//! -//! The framework's own extractors answer plain-text rejections, which breaks the API's -//! every-error-is-a-problem-document contract. [`Body`] wraps JSON request bodies - an absent body -//! answers the `missing-body` problem, a body that is not the operation's JSON answers -//! `invalid-body`. [`Coordinates`] wraps the numeric tile-address segments so an unparsable `z/x/y` -//! answers `invalid-coordinate`, and [`Generation`] wraps the generation-bearing segments so a -//! malformed generation id answers `invalid-generation`. All three delegate their OpenAPI schemas -//! to the extractor they wrap, so the documented contract stays unchanged. -//! -//! [`VariantPath`] lives here beside them: the generation/variant pair addressing a fitted layout, -//! defined once so the routes that take it document one shape. - -#![expect( - clippy::field_scoped_visibility_modifiers, - reason = "handlers in sibling modules destructure the wrappers - the axum extractor pattern - \ - and pub(super) is the narrowest visibility that permits it" -)] - -use aide::{OperationInput, generate::GenContext, openapi}; -use axum::{ - Json, - extract::{ - FromRequest, FromRequestParts, OptionalFromRequest, Path, Request, rejection::JsonRejection, - }, - http::{StatusCode, request::Parts}, -}; -use schemars::JsonSchema; -use serde::de::DeserializeOwned; - -use super::problem::{Problem, ProblemType}; -use crate::file::generation::GenerationId; - -/// The generation/variant pair addressing one fitted layout. -/// -/// Extracted through [`Generation`]: a malformed generation id answers the `invalid-generation` -/// problem before the handler runs. -#[derive(Debug, serde::Deserialize, schemars::JsonSchema)] -pub(super) struct VariantPath { - /// The sha256 generation id, as returned by `current`. - pub(super) generation: GenerationId, - /// The fitted variant name. - /// - /// The manifest lists what this generation serves. - pub(super) variant: String, -} - -/// A JSON request body whose rejections are problem documents. -/// -/// [`Json`] with the failure paths routed into the problem surface. A request without a body -/// answers `missing-body` when the operation requires one, and `Option>` reads an absent -/// body as `None`. A present body that is not the operation's JSON - whether wrong content type, -/// syntax error, shape mismatch, or oversize - answers `invalid-body` with the framework's parse -/// failure as its detail and status. The detail is a request echo, never server state. -#[derive(Debug)] -pub(super) struct Body(pub(super) T); - -impl FromRequest for Body -where - T: DeserializeOwned, - S: Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request(req: Request, state: &S) -> Result { - match as OptionalFromRequest>::from_request(req, state).await { - Ok(Some(Json(body))) => Ok(Self(body)), - Ok(None) => Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::MissingBody, - "the operation's subject rides a required JSON body", - )), - Err(rejection) => Err(invalid_body(&rejection)), - } - } -} - -impl OptionalFromRequest for Body -where - T: DeserializeOwned, - S: Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request(req: Request, state: &S) -> Result, Self::Rejection> { - match as OptionalFromRequest>::from_request(req, state).await { - Ok(body) => Ok(body.map(|Json(body)| Self(body))), - Err(rejection) => Err(invalid_body(&rejection)), - } - } -} - -impl OperationInput for Body { - fn operation_input(ctx: &mut GenContext, operation: &mut openapi::Operation) { - Json::::operation_input(ctx, operation); - } -} - -/// The `invalid-body` problem for one JSON rejection. -/// -/// The framework's status survives - a syntax error stays 400, a wrong content type 415, an -/// oversize body 413 - and its message rides as the detail: parse positions and expected shapes are -/// the crate's contract-safe request echoes. -fn invalid_body(rejection: &JsonRejection) -> Problem<'static> { - Problem::new( - rejection.status(), - ProblemType::InvalidBody, - rejection.body_text(), - ) -} - -/// The numeric tile-address segments, whose parse failure answers `invalid-coordinate`. -/// -/// [`Path`] with the rejection routed into the problem surface. -#[derive(Debug)] -pub(super) struct Coordinates(pub(super) T); - -impl FromRequestParts for Coordinates -where - T: DeserializeOwned + Send, - S: Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request_parts(parts: &mut Parts, state: &S) -> Result { - Path::::from_request_parts(parts, state) - .await - .map(|Path(inner)| Self(inner)) - .map_err(|rejection| { - Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidCoordinate, - rejection.body_text(), - ) - }) - } -} - -impl OperationInput for Coordinates { - fn operation_input(ctx: &mut GenContext, operation: &mut openapi::Operation) { - Path::::operation_input(ctx, operation); - } -} - -/// The generation-bearing path segments, whose parse failure answers `invalid-generation`. -/// -/// [`Path`] with the rejection routed into the problem surface. A generation segment that is not a -/// sha256 generation id answers 400 with the parse failure as its detail, while a well-formed id -/// the process does not serve stays the handler's 404. -#[derive(Debug)] -pub(super) struct Generation(pub(super) T); - -impl FromRequestParts for Generation -where - T: DeserializeOwned + Send, - S: Send + Sync, -{ - type Rejection = Problem<'static>; - - async fn from_request_parts(parts: &mut Parts, state: &S) -> Result { - Path::::from_request_parts(parts, state) - .await - .map(|Path(inner)| Self(inner)) - .map_err(|rejection| { - Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidGeneration, - rejection.body_text(), - ) - }) - } -} - -impl OperationInput for Generation { - fn operation_input(ctx: &mut GenContext, operation: &mut openapi::Operation) { - Path::::operation_input(ctx, operation); - } -} - -#[cfg(test)] -mod tests { - use axum::{ - Router, - body::{Body as RequestBody, to_bytes}, - extract::FromRequest as _, - http::{Request, StatusCode, header}, - routing::{get, post}, - }; - use tower::ServiceExt as _; - - use super::{Body, Coordinates, Generation}; - use crate::file::generation::GenerationId; - - /// A minimal operation body for the extraction tests. - #[derive(Debug, PartialEq, Eq, serde::Deserialize, schemars::JsonSchema)] - struct Subject { - name: String, - } - - fn json_request(body: &str) -> Request { - Request::builder() - .method("POST") - .header(header::CONTENT_TYPE, "application/json") - .body(RequestBody::from(body.to_owned())) - .expect("the request builds") - } - - fn bare_request() -> Request { - Request::builder() - .method("POST") - .body(RequestBody::empty()) - .expect("the request builds") - } - - async fn problem_json(problem: crate::api::problem::Problem<'static>) -> serde_json::Value { - use axum::response::IntoResponse as _; - - let response = problem.into_response(); - let bytes = to_bytes(response.into_body(), usize::MAX) - .await - .expect("the problem body reads"); - serde_json::from_slice(&bytes).expect("the problem body is JSON") - } - - #[tokio::test] - async fn valid_body_extracts() { - let Body(subject) = Body::::from_request(json_request(r#"{"name": "n"}"#), &()) - .await - .expect("a well-formed body extracts"); - - assert_eq!(subject.name, "n"); - } - - #[tokio::test] - async fn absent_body_answers_the_missing_body_problem() { - let problem = Body::::from_request(bare_request(), &()) - .await - .expect_err("a required body must arrive"); - let document = problem_json(problem).await; - - assert_eq!(document["type"], "/problems/atlas/missing-body"); - assert_eq!(document["status"], 400); - } - - #[tokio::test] - async fn malformed_body_answers_the_invalid_body_problem() { - let problem = Body::::from_request(json_request("{ not json"), &()) - .await - .expect_err("a malformed body must refuse"); - let document = problem_json(problem).await; - - assert_eq!(document["type"], "/problems/atlas/invalid-body"); - assert_eq!(document["status"], 400); - } - - #[tokio::test] - async fn mistyped_body_answers_the_invalid_body_problem() { - // Well-formed JSON of the wrong shape reads as a data error (422) rather than a syntax - // error (400). - let problem = Body::::from_request(json_request(r#"{"name": 7}"#), &()) - .await - .expect_err("a mistyped body must refuse"); - let document = problem_json(problem).await; - - assert_eq!(document["type"], "/problems/atlas/invalid-body"); - assert_eq!(document["status"], 422); - } - - #[tokio::test] - async fn optional_body_reads_absent_as_none_and_refuses_malformed() { - use axum::extract::OptionalFromRequest; - - let absent = as OptionalFromRequest<()>>::from_request(bare_request(), &()) - .await - .expect("an absent optional body extracts"); - assert!(absent.is_none(), "an absent body reads as None"); - - let problem = as OptionalFromRequest<()>>::from_request( - json_request("{ not json"), - &(), - ) - .await - .expect_err("a present malformed body must refuse even when optional"); - let document = problem_json(problem).await; - - assert_eq!(document["type"], "/problems/atlas/invalid-body"); - } - - /// The tile-address shape whose numeric segments can fail to parse. - #[derive(Debug, serde::Deserialize, schemars::JsonSchema)] - struct Cell { - z: u8, - x: u32, - } - - /// Routes a coordinate pair through a real router, where path extraction runs. - async fn get_cell(uri: &str) -> (StatusCode, serde_json::Value) { - let router: Router = - Router::new().route( - "/{z}/{x}", - get(|Coordinates(cell): Coordinates| async move { - format!("{}/{}", cell.z, cell.x) - }), - ); - let response = router - .oneshot( - Request::builder() - .uri(uri) - .body(RequestBody::empty()) - .expect("the request builds"), - ) - .await - .expect("the router answers"); - let status = response.status(); - let bytes = to_bytes(response.into_body(), usize::MAX) - .await - .expect("the response body reads"); - let value = serde_json::from_slice(&bytes) - .unwrap_or_else(|_| serde_json::Value::String(String::from_utf8_lossy(&bytes).into())); - (status, value) - } - - #[tokio::test] - async fn parsable_coordinates_extract() { - let (status, body) = get_cell("/3/7").await; - - assert_eq!(status, StatusCode::OK); - assert_eq!(body, serde_json::Value::String("3/7".to_owned())); - } - - #[tokio::test] - async fn unparsable_coordinates_answer_the_invalid_coordinate_problem() { - let (status, document) = get_cell("/deep/7").await; - - assert_eq!(status, StatusCode::BAD_REQUEST); - assert_eq!(document["type"], "/problems/atlas/invalid-coordinate"); - assert_eq!(document["status"], 400); - } - - /// A generation-bearing path shape. - #[derive(Debug, serde::Deserialize, schemars::JsonSchema)] - struct Layout { - generation: GenerationId, - } - - /// Routes a generation id through a real router, where path extraction runs. - async fn get_layout(uri: &str) -> (StatusCode, serde_json::Value) { - let router: Router = - Router::new().route( - "/{generation}", - get(|Generation(layout): Generation| async move { - layout.generation.to_string() - }), - ); - let response = router - .oneshot( - Request::builder() - .uri(uri) - .body(RequestBody::empty()) - .expect("the request builds"), - ) - .await - .expect("the router answers"); - let status = response.status(); - let bytes = to_bytes(response.into_body(), usize::MAX) - .await - .expect("the response body reads"); - let value = serde_json::from_slice(&bytes) - .unwrap_or_else(|_| serde_json::Value::String(String::from_utf8_lossy(&bytes).into())); - (status, value) - } - - #[tokio::test] - async fn well_formed_generation_id_extracts() { - let id = "a".repeat(64); - let (status, body) = get_layout(&format!("/{id}")).await; - - assert_eq!(status, StatusCode::OK); - assert_eq!(body, serde_json::Value::String(id)); - } - - #[tokio::test] - async fn malformed_generation_id_answers_the_invalid_generation_problem() { - let (status, document) = get_layout("/not-a-generation").await; - - assert_eq!(status, StatusCode::BAD_REQUEST); - assert_eq!(document["type"], "/problems/atlas/invalid-generation"); - assert_eq!(document["status"], 400); - } - - #[tokio::test] - async fn wrong_content_type_answers_the_invalid_body_problem() { - // The handler is irrelevant because the extractor refuses first. - let router: Router = Router::new().route( - "/subject", - post(|Body(subject): Body| async move { subject.name }), - ); - let response = router - .oneshot( - Request::builder() - .method("POST") - .uri("/subject") - .header(header::CONTENT_TYPE, "text/plain") - .body(RequestBody::from("name=n")) - .expect("the request builds"), - ) - .await - .expect("the router answers"); - - assert_eq!(response.status(), StatusCode::UNSUPPORTED_MEDIA_TYPE); - let bytes = to_bytes(response.into_body(), usize::MAX) - .await - .expect("the response body reads"); - let document: serde_json::Value = - serde_json::from_slice(&bytes).expect("the rejection is a problem document"); - assert_eq!(document["type"], "/problems/atlas/invalid-body"); - assert_eq!(document["status"], 415); - } -} diff --git a/libs/@local/graph/atlas/src/api/headers.rs b/libs/@local/graph/atlas/src/api/headers.rs deleted file mode 100644 index 4d369375670..00000000000 --- a/libs/@local/graph/atlas/src/api/headers.rs +++ /dev/null @@ -1,148 +0,0 @@ -//! The routes' `Cache-Control` postures: sent and documented from one constant each. -//! -//! Each handler sends its posture from these constants and each operation documents the same -//! constant through [`cache_control`], so the OpenAPI document and the wire cannot drift apart. - -use aide::openapi; - -/// The `current` posture: cached copies revalidate on every read. -/// -/// The pointer is the API's one mutable read. A stale copy is exactly the failure the route exists -/// to prevent. -pub(super) const REVALIDATE: &str = "private, no-cache"; - -/// The authority token header. -/// -/// The manifest response issues it, and data requests present it back. -/// -/// The canonical spelling is `Atlas-Authority`. The constant is lowercase because static header -/// names are, and header matching is case-insensitive either way. -pub(super) const AUTHORITY: &str = "atlas-authority"; - -/// The same header in its canonical spelling, for the documents that name it. -/// -/// People and generators that echo them verbatim read the OpenAPI parameter and header keys, so the -/// document carries the canonical form while the wire carries [`AUTHORITY`]. -pub(super) const AUTHORITY_DOCUMENTED: &str = "Atlas-Authority"; - -/// The query-response posture. -/// -/// The client's application-layer cache is the cache. Binary envelopes and translate maps key on -/// (authorization context, generation, route, canonical body), which shared caches cannot see. -/// `no-store` keeps them out of the way. -pub(super) const NO_STORE: &str = "private, no-store"; - -/// Documents the presented authority token's request header, in its required reading. -/// -/// The data routes document this parameter as written. The manifest extracts the scope optionally, -/// and aide derives its documentation from this same parameter, flipping `required`. The -/// description therefore covers both readings and the flag separates them. The schema pins the -/// alphabet and not the width, and names no sealed field - the token is opaque to every caller. -#[expect( - clippy::default_trait_access, - reason = "we do not want to pull in a dependency just to pin its default" -)] -pub(super) fn presented_authority() -> openapi::Parameter { - let description = "the authority token present in the response's `Atlas-Authority` header, \ - replayed verbatim. Required for any data route. The manifest may supply a \ - previously admitted token, in which case certain properties of the token \ - are carried forward."; - - openapi::Parameter::Header { - parameter_data: openapi::ParameterData { - name: AUTHORITY_DOCUMENTED.to_owned(), - description: Some(description.to_owned()), - required: true, - deprecated: None, - format: openapi::ParameterSchemaOrContent::Schema(openapi::SchemaObject { - json_schema: schemars::json_schema!({"type": "string", "pattern": "^[0-9a-f]+$"}), - example: None, - external_docs: None, - }), - example: None, - examples: Default::default(), - explode: None, - extensions: Default::default(), - }, - style: openapi::HeaderStyle::Simple, - } -} - -/// Documents the issued authority token's response header. -/// -/// The pattern fixes the alphabet and not the width: the width follows from the envelope's -/// construction, and no client may depend on it. -#[expect( - clippy::default_trait_access, - reason = "we do not want to pull in a dependency just to pin its default" -)] -pub(super) fn authority() -> openapi::ReferenceOr { - openapi::ReferenceOr::Item(openapi::Header { - description: Some( - "a freshly issued per-caller authority token, lowercase hexadecimal. Present it back \ - in this same header on every data request" - .to_owned(), - ), - style: openapi::HeaderStyle::Simple, - required: true, - deprecated: None, - format: openapi::ParameterSchemaOrContent::Schema(openapi::SchemaObject { - json_schema: schemars::json_schema!({"type": "string", "pattern": "^[0-9a-f]+$"}), - example: None, - external_docs: None, - }), - example: None, - examples: Default::default(), - extensions: Default::default(), - }) -} - -/// Documents the `Retry-After` header every `429` carries. -#[expect( - clippy::default_trait_access, - reason = "we do not want to pull in a dependency just to pin its default" -)] -pub(super) fn retry_after() -> openapi::ReferenceOr { - openapi::ReferenceOr::Item(openapi::Header { - description: Some( - "whole seconds until the crossed budget admits the request again, at least one" - .to_owned(), - ), - style: openapi::HeaderStyle::Simple, - required: true, - deprecated: None, - format: openapi::ParameterSchemaOrContent::Schema(openapi::SchemaObject { - json_schema: schemars::json_schema!({"type": "integer", "minimum": 1}), - example: None, - external_docs: None, - }), - example: None, - examples: Default::default(), - extensions: Default::default(), - }) -} - -/// Documents a response header that always carries `value`. -#[expect( - clippy::default_trait_access, - reason = "we do not want to pull in a dependency just to pin its default" -)] -pub(super) fn cache_control( - value: &str, - description: &str, -) -> openapi::ReferenceOr { - openapi::ReferenceOr::Item(openapi::Header { - description: Some(format!("always `{value}`; {description}")), - style: openapi::HeaderStyle::Simple, - required: false, - deprecated: None, - format: openapi::ParameterSchemaOrContent::Schema(openapi::SchemaObject { - json_schema: schemars::json_schema!({"type": "string"}), - example: None, - external_docs: None, - }), - example: Some(serde_json::Value::String(value.to_owned())), - examples: Default::default(), - extensions: Default::default(), - }) -} diff --git a/libs/@local/graph/atlas/src/api/locate.rs b/libs/@local/graph/atlas/src/api/locate.rs deleted file mode 100644 index 2c6f8e06851..00000000000 --- a/libs/@local/graph/atlas/src/api/locate.rs +++ /dev/null @@ -1,249 +0,0 @@ -//! `POST /v1/atlas/locate/{generation}/{variant}`. -//! -//! The source entity's ego-graph - its edges and the partners they connect - as `SALTILEL` bytes. - -use alloc::sync::Arc; -use core::panic::AssertUnwindSafe; - -use aide::transform::TransformOperation; -use axum::{extract::State, http::StatusCode}; -use hashql_core::id::IdVec; -use tokio::sync::oneshot; -use tracing::Instrument as _; - -use super::{ - AppState, clause, - extract::{Body, Generation, VariantPath}, - problem::{Problem, ProblemType, reject_generation, reject_variant}, - saltile::{Saltile, spawn}, - visibility::{Visibility, view_problem}, -}; -use crate::{ - postgres::id::ArchivedEntityId, - serve::{ - LocateError, LocateRequest, - hydrate::{ - DetailError, EdgeSlot, LocateHydration, LocateOrder, LocateStore, MaskingActor, - NodeSlot, - }, - }, -}; - -/// The operation's description. -const DESCRIPTION: &str = - "Returns one entity's ego-graph: the source point, every neighbour it links to, and the edges \ - joining them, as a `SALTILEL` binary envelope. - -The JSON body is required and names the source in exactly one of two fields: `entityId` (the \ - upstream `webId~entityUuid` id a search result or deep link carries) or `row` (a row id a \ - tile's `ROW_IDS` column delivered - if you hold one, no translate round trip is needed). A \ - body carrying both or neither answers `invalid-source`. Either field reaches identical \ - geometry bytes for the same source. - -A source may also name an entity placed since the generation was fitted: it resolves through the \ - serving session's own records in either field form, answers its frozen coordinate and \ - captured display, and delivers alone - the generation's adjacency never names an entity \ - placed after the fit, so its ego-graph is empty and complete. The row ids such entities \ - carry die with the serving session that minted them, exactly as translate describes. - -The source is delivered first, partners follow ascending by row id, and edges are ordered \ - ascending by link-entity identity bytes (the `EDGE_IDS` column: each edge's 32-byte entity \ - id, web uuid then entity uuid). The manifest's `limits.locateEdges` caps the edge set; under \ - truncation the response keeps the edges whose partners lie nearest the source, the HEAD's \ - `complete` key reads `false`, and a partner whose every edge was truncated is not delivered. - -The HEAD also carries the source's first visible zoom and its tile there (the fly-to target), the \ - source's entity id as 32 raw bytes, and two completeness flags: `typeIdsComplete` (the \ - request's `coloredTypeIds` cover every direct type of the source) and `propertiesComplete` \ - (the trailer's source property map is the entity's whole deliverable set - every property no \ - protection withholds from the requesting actor). `coloredTypeIds` behaves exactly as on the \ - tile route. - -The response always carries the detail trailer. Labels come from the generation (for an entity \ - placed since the fit, from its placement's captured display) and are admitted only when the \ - request-time store read resolves the corresponding entity. The store also supplies each \ - node's representative type, the source's properties capped by `limits.locateProperties`, and \ - each edge's direct types and properties capped by `limits.locateLinkTypeIds` and \ - `limits.locateLinkProperties`. Each edge cap has a completeness flag. Type and property \ - references are integer indexes into the trailer's two sorted URL tables. - -A source that does not name a visible node answers `unknown-entity`: nonexistent, inaccessible, \ - unparsable, and out-of-range `row` values are indistinguishable by design. - -Filtering binds at the manifest. This body has no `filter` field, and an unknown member is \ - rejected as `invalid-body`. -"; - -/// `POST /v1/atlas/locate/{generation}/{variant}`. -/// -/// The source's spotlight subgraph, as `SALTILEL` bytes. The source id is the request's subject, so -/// the route requires the body. -pub(super) async fn handler( - State(state): State>, - visibility: Visibility, - Generation(VariantPath { - generation, - variant, - }): Generation, - Body(request): Body, -) -> Result> { - reject_generation(&state, generation)?; - reject_variant(&variant)?; - - // The whole pipeline is one synchronous call on a rayon worker; only the store order and its - // answer cross back here, where the connections live and the two queries run concurrently. - let atlas = Arc::clone(&state.atlas); - let limits = state.limits; - let masking = visibility.masking(); - let (order_sender, order_receiver) = oneshot::channel(); - let (answer_sender, answer_receiver) = oneshot::channel(); - let store = ChannelLocateStore { - order: order_sender, - answer: answer_receiver, - }; - - let (result, ()) = tokio::join!( - spawn(AssertUnwindSafe(move || { - let view = visibility.view(&atlas)?; - - atlas.locate(&request, limits, view, store) - })), - async { - // An order never arrives when the pipeline rejects the request or panics first. - let Ok(order) = order_receiver.await else { - return; - }; - let _: Result<(), _> = answer_sender.send(hydrate(&state, order, masking).await); - }, - ); - - let bytes = match result? { - Ok(bytes) => bytes, - Err(error @ LocateError::UnknownEntity) => { - return Err(Problem::new( - StatusCode::NOT_FOUND, - ProblemType::UnknownEntity, - error.to_string(), - )); - } - Err(error @ LocateError::Types { .. }) => { - return Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::TooManyTypes, - error.to_string(), - )); - } - Err(error @ LocateError::Source { .. }) => { - return Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidSource, - error.to_string(), - )); - } - // A stale sealed offset answers the uniform refusal. A mismatched pair or width names - // an input this process produced and answers the internal problem. - Err(LocateError::View(error)) => return Err(view_problem(error)), - Err(error @ LocateError::Details(_)) => { - return Err(Problem::internal(error, "the detail hydration failed")); - } - }; - - Ok(Saltile::new(bytes)) -} - -/// One locate hydration order, owned for the trip between the pipeline and the store side. -struct LocateOrderMessage { - /// The delivered node identities, source first. - nodes: IdVec, - /// The delivered link-entity identities, ascending identity bytes. - links: IdVec, - /// Most properties the source's map delivers. - properties: u32, - /// Most direct-type URLs each link delivers. - link_type_ids: u32, - /// Most properties each link's map delivers. - link_properties: u32, -} - -/// The transport's locate store, carrying one order out to the handler and one answer back in. -struct ChannelLocateStore { - order: oneshot::Sender, - answer: oneshot::Receiver>, -} - -impl LocateStore for ChannelLocateStore { - fn hydrate(self, order: LocateOrder<'_>) -> Result { - self.order - .send(LocateOrderMessage { - // The channel is the one boundary that owns the identities, so the views - // materialize here and nowhere earlier. - nodes: order.nodes.iter().collect(), - links: order.links.iter().copied().collect(), - properties: order.properties, - link_type_ids: order.link_type_ids, - link_properties: order.link_properties, - }) - .map_err(|_order| DetailError::Disconnected)?; - - self.answer - .blocking_recv() - .map_err(|_closed| DetailError::Disconnected)? - } -} - -/// Answers one order against the serving store, both halves concurrently. -async fn hydrate( - state: &AppState, - order: LocateOrderMessage, - masking: MaskingActor, -) -> Result { - let (nodes, links) = tokio::try_join!( - state - .remote - .locate_node_hydration(&order.nodes, order.properties, masking) - .in_current_span(), - state - .remote - .locate_link_hydration( - &order.links, - order.link_type_ids, - order.link_properties, - masking - ) - .in_current_span(), - )?; - - Ok(LocateHydration { nodes, links }) -} - -/// Documents the operation. -pub(super) fn document(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation - .id("locate") - .summary("The source entity's ego-graph, as SALTILEL envelope bytes") - .description(DESCRIPTION) - .with(clause::describe_body( - "the locate request; exactly one of `entityId` and `row` names the subject", - )) - .response_with::<200, Saltile, _>(|response| { - response.description( - "a `SALTILEL` envelope: the source first, its linked partners, and the edges \ - joining them", - ) - }) - .response_with::<400, Problem<'static>, _>(|response| { - response.description( - "`too-many-types`, `invalid-source`, `invalid-generation`, `missing-body`, or \ - `invalid-body` (a body that is not JSON)", - ) - }) - .with(clause::invalid_body_data) - .with(clause::unauthorized) - .response_with::<404, Problem<'static>, _>(|response| { - response.description( - "`unknown-generation`, `unknown-variant`, or `unknown-entity` (identical for \ - nonexistent, inaccessible, unparsable, and out-of-range sources)", - ) - }) - .with(clause::any_problem) -} diff --git a/libs/@local/graph/atlas/src/api/manifest.rs b/libs/@local/graph/atlas/src/api/manifest.rs deleted file mode 100644 index 34348c081b6..00000000000 --- a/libs/@local/graph/atlas/src/api/manifest.rs +++ /dev/null @@ -1,603 +0,0 @@ -//! `POST /v1/atlas/generation/{generation}/manifest`. -//! -//! Bootstrap data - configuration, snapshot provenance, and the delivery schedule resolved for this -//! caller - delivered beside a fresh per-caller authority token in the `Atlas-Authority` response -//! header. An optional body carries the filter document that binds the view. - -use alloc::sync::Arc; -use std::time::SystemTime; - -use aide::{axum::IntoApiResponse, transform::TransformOperation}; -use axum::{ - Json, - body::Bytes, - extract::State, - http::{HeaderName, HeaderValue, StatusCode, header}, -}; -use hash_graph_store::filter::Filter; -use rand::TryCryptoRng; -use type_system::knowledge::Entity; - -use super::{ - AppState, - authorization::Actor, - clause, - extract::Generation, - headers, - problem::{Problem, ProblemType, reject_generation}, - visibility::{self, view_problem}, -}; -use crate::{ - file::generation::GenerationId, - integrity::HexBytes, - serve::{ - CutOffset, DensityPolicy, Manifest, ViewOccupancy, - authorization::{Scope, ScopeFilter}, - cache::scope::FilterDigest, - }, -}; - -/// The rule by which one issuance seals its delivery-cut offset. -#[derive(Debug, Copy, Clone)] -enum OffsetRule { - /// A first token for this actor, which resolves the wanted view's own offset. - Bootstrap, - /// A token for the view its predecessor sealed. - /// - /// A scoped view keeps the offset it sealed, so the detail a tile carries at a fixed zoom does - /// not move across a renewal. Zero is what an operator view seals here as everywhere, which - /// normalizes a token issued under an older contract instead of carrying its value forward. - Carry(CutOffset), - /// A token for another view, which keeps the sealed offset unless that view resolves coarser. - Rebind(CutOffset), -} - -/// The delivery-cut offset one issuance seals. -/// -/// [`CutOffset::ZERO`] whenever no offset is servable. A deployment without a density policy serves -/// every scope at its recorded cut. An operator view serves the corpus schedule, and an absent -/// `view` is what says that, so no route can serve corpus bytes while its manifest declares a -/// deeper cut. -/// -/// With a policy and a scoped view, [`OffsetRule`] states which question this issuance asks, and -/// the arithmetic of every answer lives in [`DensityPolicy`]. Every handler path issues through -/// here, so no branch can seal an offset by a rule of its own. `view` is the scope's entry-held -/// aggregate, taken from the store's answer alone, so no issuance pays an occupancy pass and no -/// snapshot moves the sealed offset. [`OffsetRule::Carry`] never reads it: a session keeping its -/// own view -/// keeps the offset it sealed. -fn sealed_offset( - density: Option, - rule: OffsetRule, - view: Option<&ViewOccupancy>, -) -> CutOffset { - let (Some(policy), Some(view)) = (density, view) else { - return CutOffset::ZERO; - }; - - match rule { - OffsetRule::Bootstrap => policy.resolve(view), - OffsetRule::Carry(carried) => carried, - OffsetRule::Rebind(carried) => policy.rebind(carried, view), - } -} - -/// The operation's description. -const DESCRIPTION: &str = - "Returns the bootstrap document for one generation: everything a client needs before its \ - first tile. - -The wire version the binary envelopes speak, the served variant names, the bucket schedule the \ - tile grid follows, the serving limits the handlers enforce, and the snapshot's decision-time \ - point when the source data carried one. Those blocks hold for the generation's lifetime. One \ - block does not: `scopeSchedule` states the delivery cut resolved for this caller and, as \ - `maxZoom`, the deepest zoom at which that view still delivers new points, so two callers of \ - one generation can read different documents, and a client reads its own rather than a shared \ - one. The response is not cached either: the `Atlas-Authority` header carries a fresh \ - authority token the data routes require, valid for `authorityHardSeconds`. Re-fetch at the \ - `authorityRefreshSeconds` cadence, presenting the current token - even expired - in the same \ - header: a scoped view's sealed delivery depth carries into the fresh token, so renewing \ - authority does not change the detail a tile carries, and a full-visibility view renews at \ - the corpus cut it serves. There is no separate renewal mode: every request states the view \ - it wants, so a caller that wants its filter must send that filter's exact bytes again. A \ - presented token that is invalid or names another actor answers `401`; a request without a \ - token bootstraps."; - -/// What the optional filter document does. -const FILTER: &str = - "The body states the view this request wants, which is why this read is a `POST`: the filter \ - is part of the view's identity. A body carries a filter document, and no body asks for the \ - unfiltered view. The digest - taken over the bytes exactly as presented - seals into the \ - token, and the visibility proof compiles over the document itself. - -A request whose wanted filter is the one its token already seals keeps a scoped session's delivery \ - depth, so the detail a tile carries at a fixed zoom does not move; its document is still \ - resolved from the resent bytes, because a filter the server has already purged can be \ - rebuilt only from them. A request wanting a different filter - including no filter at all, \ - which removes one - resolves the wanted view and keeps the session's depth unless that view \ - resolves coarser, which clamps the depth down to it. A request without a token bootstraps \ - the view it asks for."; - -/// The route's path parameters. -/// -/// Extracted through [`Generation`]: a malformed generation id answers the `invalid-generation` -/// problem before the handler runs. -#[derive(Debug, serde::Deserialize, schemars::JsonSchema)] -pub(super) struct GenerationPath { - /// The sha256 generation id, as returned by `current`. - generation: GenerationId, -} - -/// `POST /v1/atlas/generation/{generation}/manifest`. -/// -/// Bootstrap data for one generation and one caller. -/// -/// Every block but one holds for the generation's lifetime. `scopeSchedule` states the delivery cut -/// this issuance resolved and sealed, so the document a caller reads describes the bytes its own -/// routes answer with. The response carries a freshly issued authority token in the -/// `Atlas-Authority` -/// header, which is the second reason it sends `no-store`. Fetching it also resolves the caller's -/// scope. A client bootstraps here, so the resolution costs the request that expects a wait rather -/// than the first tile. -/// -/// Extraction judges the presented token before the handler judges the generation. An unacceptable -/// token answers `401` whatever generation the route names, and a retired generation answers `404` -/// to an absent or accepted token, where a re-fetching client discovers the re-pin. -/// -/// The body states the view the request wants, either a filter document or nothing for the -/// unfiltered view. No renewal mode leaves the wanted view unstated, because the server purges a -/// filter document with its cache entry and a token cannot rebuild it. The token seals the filter's -/// digest, and a digest names no document. -/// -/// The wanted view therefore decides. When it equals the sealed one, a scoped session keeps its -/// delivery depth `k` while an operator one renews at the corpus cut, and the handler resolves the -/// view either way, so the fresh token carries current authorization and a purged filter document -/// is rebuilt from the resent bytes. When it differs from the sealed one - a changed filter, or its -/// removal - the handler resolves the wanted view and keeps `k` unless that view resolves coarser, -/// which clamps it down through [`DensityPolicy::rebind`]; an empty wanted view therefore seals -/// zero. Without a density policy the seal is [`CutOffset::ZERO`]. A bootstrap resolves both the -/// wanted view and its depth. -/// -/// [`DensityPolicy::rebind`]: crate::serve::DensityPolicy::rebind -pub(super) async fn handler( - State(state): State>, - Actor(actor): Actor, - carried: Option, - Generation(GenerationPath { generation }): Generation, - body: Bytes, -) -> Result> -where - R: TryCryptoRng, -{ - reject_generation(&state, generation)?; - - // The digest is over the bytes exactly as presented; the parse is the edge validation, and the - // resolution recompiles the filter from the same bytes. - let filter = (!body.is_empty()) - .then(|| { - serde_json::from_slice::>(&body) - .map(|_document| (FilterDigest::of(&body), Arc::<[u8]>::from(body.as_ref()))) - .map_err(|error| { - Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidBody, - format!("the filter document does not parse: {error}"), - ) - }) - }) - .transpose()?; - - // The body states the wanted view, so an absent one wants the unfiltered view rather than - // whatever a token happens to seal. - let (wanted, document) = filter.map_or((None, None), |(digest, document)| { - (Some(digest), Some(document)) - }); - - // Resolve the wanted view rather than trust what a token seals: the authorization behind - // the view may have changed, and a wanted filter is rebuilt from the resent bytes because - // the server purges a document with its entry. An unfiltered renewal has no document to - // rebuild and resolves the same way, and a bootstrap resolves the view it asks for. - let visibility = visibility::resolve(&state, actor, wanted, document).await?; - - let scope = match carried { - // The wanted view equals the sealed one, so a scoped session keeps its delivery depth. - Some(scope) if scope.filter.digest() == wanted => Scope { - actor: scope.actor, - filter: scope.filter, - k: sealed_offset( - state.density, - OffsetRule::Carry(scope.k), - visibility.occupancy(), - ), - }, - // A different wanted view, removal included: the session keeps its delivery depth - // unless the wanted view resolves coarser. - Some(scope) => Scope { - actor: scope.actor, - filter: ScopeFilter::from(wanted), - k: sealed_offset( - state.density, - OffsetRule::Rebind(scope.k), - visibility.occupancy(), - ), - }, - // A bootstrap resolves the depth it will serve at. - None => Scope::new( - actor, - wanted, - sealed_offset(state.density, OffsetRule::Bootstrap, visibility.occupancy()), - ), - }; - - let token = state - .tokens - .issue(scope, SystemTime::now()) - .map_err(|error| Problem::internal(error, "issuing the authority token failed"))?; - - Ok(( - [ - ( - header::CACHE_CONTROL, - HeaderValue::from_static(headers::NO_STORE), - ), - ( - HeaderName::from_static(headers::AUTHORITY), - HeaderValue::try_from(HexBytes::new(token).to_string()) - .unwrap_or_else(|_| unreachable!("hexadecimal is a valid header value")), - ), - ], - Json( - state.atlas.manifest( - state.limits.manifest_limits(state.visibility), - scope.k, - visibility - .view(&state.atlas, scope.k) - .map_err(view_problem)? - .min_resolution(), - ), - ), - )) -} - -/// The filter document's schema. -/// -/// One JSON object in the graph's entity-query filter grammar, which lives in that surface rather -/// than in a restatement here: the server validates the document against it and answers -/// `invalid-body` when it does not parse. The digest that names the view hashes the bytes exactly -/// as presented, so a client re-presenting a filter sends the same bytes it sent before. -struct FilterDocument; - -impl schemars::JsonSchema for FilterDocument { - fn schema_name() -> alloc::borrow::Cow<'static, str> { - "FilterDocument".into() - } - - fn json_schema(_generator: &mut schemars::SchemaGenerator) -> schemars::Schema { - schemars::json_schema!({ - "type": "object", - "description": "an entity-query filter document, in the graph's structural-query \ - filter grammar", - }) - } -} - -/// Documents the operation. -/// -/// The default response is the catch-all each of the four data routes already declares. The -/// manifest resolves a caller's scope too, so it answers the same visibility and internal problems -/// and owes the same declaration. Without it the document would promise four statuses for an -/// operation that has five. -pub(super) fn document(operation: TransformOperation<'_>) -> TransformOperation<'_> { - // A bodyless request states the unfiltered view (a bootstrap, or a renewal that removes its - // filter), so the declared body is optional. The input declaration marks it required. - operation - .id("manifest") - .summary("The generation's bootstrap manifest") - .description(&format!("{DESCRIPTION}\n\n{FILTER}")) - .input::>() - .with(clause::optional_body) - .response_with::<200, Json, _>(|mut response| { - response.inner().headers.insert( - "Cache-Control".to_owned(), - headers::cache_control( - headers::NO_STORE, - "the response carries a per-caller authority token, and its document states \ - the delivery schedule resolved for that caller", - ), - ); - response.inner().headers.insert( - headers::AUTHORITY_DOCUMENTED.to_owned(), - headers::authority(), - ); - response.description( - "the manifest, with a fresh authority token in the `Atlas-Authority` header", - ) - }) - .response_with::<400, Problem<'static>, _>(|response| { - response.description( - "`invalid-generation`: a malformed generation id, or `invalid-body`: a body that \ - is not JSON, or a filter the entity query surface cannot compile", - ) - }) - .with(clause::invalid_body_data) - .response_with::<401, Problem<'static>, _>(|response| { - response.description( - "`unauthorized`: the presented token is invalid or names another actor; a \ - bootstrap without a token succeeds", - ) - }) - .response_with::<404, Problem<'static>, _>(|response| { - response.description("`unknown-generation`: re-read `current` and retry") - }) - .default_response_with::, _>(|response| { - response.description( - "any other problem document: `visibility-unavailable` marks a scope the store \ - could not resolve, `internal` a server-side failure", - ) - }) -} - -#[cfg(test)] -mod tests { - use core::num::NonZero; - - use aide::{openapi::Operation, transform::TransformOperation}; - - use super::{OffsetRule, document, sealed_offset}; - use crate::{ - math::Log2, - morton::{Depth, MortonCell, MortonKey}, - serve::{CutOffset, DensityBand, DensityPolicy, ViewOccupancy}, - }; - - /// The fixture policy. - /// - /// The span exponent is 1, so offset `k` cuts at depth `1 + k`. The band is `[2, 3]` occupied - /// cells, and the deepest served zoom is 4. - /// - /// The band is narrow and small on purpose. Under the default `[2_000, 4_000]` band every - /// fixture view small enough to hand-derive sits below the band at every depth, so the argmin - /// ties everywhere and resolves to zero - which is the one value that cannot distinguish - /// `resolve` from `rebind` from the no-policy branch. - fn policy() -> DensityPolicy { - DensityPolicy::new( - DensityBand::new( - NonZero::new(2).expect("the fixture band's lower bound is positive"), - NonZero::new(3).expect("the fixture band's upper bound is positive"), - ) - .expect("the fixture band is ordered"), - Log2::new(1).expect("the fixture span lies below the shift width"), - 4, - ) - .expect("the fixture schedule admits an offset") - } - - /// The absent occupancy aggregate, which is what an operator proof's entry holds. - const NO_VIEW: Option<&ViewOccupancy> = None; - - /// The fixture view, hand-derived: four points on one row of the depth-3 grid. - /// - /// Cells `(0, 0)`, `(1, 0)`, `(2, 0)`, `(3, 0)` at depth 3. Folding the grid coarser: at depth - /// 2 they pair into `(0, 0)` and `(1, 0)`, at depth 1 they all fall in `(0, 0)`, so `C(1, V) = - /// 1`, `C(2, V) = 2`, `C(3, V) = 4`, constant below that, and the saturation depth is 3 - every - /// occupied cell holds one key there. - /// - /// Against the band `[2, 3]`, the candidate offsets `0..=2` (the saturation cap, below the - /// schedule's ceiling of 27) sit at distances 1, 0, 1. The argmin is **offset 1**, and it is a - /// strict minimum rather than a tie, so a resolution and a clamp read differently on it. - fn view() -> ViewOccupancy { - let mut keys: Vec = (0..4) - .map(|x| { - MortonCell::new( - Depth::new(3).expect("the fixture depth lies within the key width"), - x, - 0, - ) - .expect("the fixture cell lies on the depth's grid") - .min_key() - }) - .collect(); - - ViewOccupancy::of(&mut keys) - } - - /// A generation whose schedule admits no offset seals zero, whatever a session carried. - #[test] - fn no_density_policy_seals_zero() { - assert_eq!( - sealed_offset(None, OffsetRule::Bootstrap, Some(&view())), - CutOffset::ZERO - ); - assert_eq!( - sealed_offset(None, OffsetRule::Rebind(CutOffset::new(2)), Some(&view())), - CutOffset::ZERO - ); - } - - /// A bootstrap - nothing carried - resolves the wanted view's own offset. - /// - /// The fixture's argmin is 1, which is neither `CutOffset::ZERO` nor either carried value the - /// tests below use, so this fails if the branch resolves nothing or clamps something. - #[test] - fn bootstrap_resolves_the_wanted_views_own_offset() { - assert_eq!( - sealed_offset(Some(policy()), OffsetRule::Bootstrap, Some(&view())), - CutOffset::new(1) - ); - } - - /// The rebind keeps a carried offset coarser than the wanted view's resolution. - #[test] - fn rebind_keeps_a_carried_offset_coarser_than_the_views_resolution() { - assert_eq!( - sealed_offset( - Some(policy()), - OffsetRule::Rebind(CutOffset::ZERO), - Some(&view()) - ), - CutOffset::ZERO, - "the wanted view resolves to 1 and a session at 0 must not be deepened into it" - ); - } - - /// A carried offset deeper than the wanted view's resolution clamps down to it. - #[test] - fn rebind_clamps_a_carried_offset_deeper_than_the_views_resolution() { - assert_eq!( - sealed_offset( - Some(policy()), - OffsetRule::Rebind(CutOffset::new(2)), - Some(&view()) - ), - CutOffset::new(1), - "a session at 2 must clamp to the wanted view's resolution of 1" - ); - } - - /// An operator view seals zero at every issuance, over a fixture whose argmin is not zero. - /// - /// The absent occupancy is what an operator proof answers, and the fixture's own argmin is 1, - /// so an issuance that consulted the policy anyway would seal 1 here and fail all three - /// assertions. The carried case guards the change boundary, where a token issued before this - /// rule seals a nonzero offset and its renewal has to come back at zero rather than carry the - /// bad value forward. - #[test] - fn operator_view_seals_zero_at_every_mint() { - assert_eq!( - sealed_offset(Some(policy()), OffsetRule::Bootstrap, NO_VIEW), - CutOffset::ZERO, - ); - assert_eq!( - sealed_offset( - Some(policy()), - OffsetRule::Carry(CutOffset::new(2)), - NO_VIEW - ), - CutOffset::ZERO, - "a renewal must not carry an offset the corpus schedule cannot serve", - ); - assert_eq!( - sealed_offset( - Some(policy()), - OffsetRule::Rebind(CutOffset::new(2)), - NO_VIEW - ), - CutOffset::ZERO, - ); - assert_eq!( - policy().resolve(&view()), - CutOffset::new(1), - "the fixture must resolve nonzero, or the assertions above pass for the wrong reason", - ); - } - - /// A renewal of an unchanged view keeps the offset its predecessor sealed. - /// - /// The wanted view's own resolution is 1 and the carried value is 2, so a renewal that - /// re-resolved or clamped would read 1 here. The session serves at the depth it bootstrapped. - #[test] - fn renewed_view_keeps_its_sealed_offset() { - assert_eq!( - sealed_offset( - Some(policy()), - OffsetRule::Carry(CutOffset::new(2)), - Some(&view()) - ), - CutOffset::new(2), - ); - } - - /// A carried offset over a view with no occupancy reaches zero. - /// - /// The empty view resolves to zero through the same argmin, and the clamp is a minimum, so the - /// session's depth collapses however deep it was. - #[test] - fn carried_offset_over_an_empty_view_reaches_zero() { - let empty = ViewOccupancy::of(&mut []); - assert!(empty.is_empty(), "the fixture view is empty"); - - assert_eq!( - sealed_offset( - Some(policy()), - OffsetRule::Rebind(CutOffset::new(2)), - Some(&empty) - ), - CutOffset::ZERO - ); - } - - /// Renders the operation's emitted OpenAPI. - /// - /// The assertions read the serialized document rather than the builder calls, because the - /// emitted contract is what a client receives. - fn emitted() -> serde_json::Value { - let mut operation = Operation::default(); - let _documented = document(TransformOperation::new(&mut operation)); - - serde_json::to_value(&operation).expect("an operation serializes") - } - - /// The operation declares the filter document, and declares it optional. - /// - /// A bodyless request states the unfiltered view (a bootstrap, or a renewal that removes its - /// filter), so a body marked required would document a refusal this operation does not make. - #[test] - fn operation_declares_the_filter_document_as_optional() { - let emitted = emitted(); - let body = &emitted["requestBody"]; - - assert!( - body["content"]["application/json"].is_object(), - "the filter document is not declared as a JSON body: {emitted:#}" - ); - // The emitted form omits `required` when it is false. - assert!( - !body["required"].as_bool().unwrap_or(false), - "the filter document is declared required" - ); - } - - /// The `200` declares both headers it carries. - #[test] - fn response_declares_both_headers_it_carries() { - let headers = &emitted()["responses"]["200"]["headers"]; - - assert!( - headers[super::headers::AUTHORITY_DOCUMENTED].is_object(), - "the issued authority header is not declared" - ); - assert!( - headers["Cache-Control"].is_object(), - "the cache directive is not declared" - ); - } - - /// The operation declares the catch-all response. - /// - /// A manifest fetch resolves the caller's visibility, so it answers `visibility-unavailable` - /// and `internal` beside the four statuses it declares by name. Without the default response - /// the document would omit both. - #[test] - fn operation_declares_the_catch_all_response() { - let responses = &emitted()["responses"]; - - assert!( - responses["default"]["description"].is_string(), - "the operation declares no catch-all response: {responses:#}" - ); - } - - /// The `400` names the body problem the filter document can answer with. - #[test] - fn bad_request_response_names_the_body_problem() { - let description = emitted()["responses"]["400"]["description"] - .as_str() - .expect("the 400 response carries a description") - .to_owned(); - - assert!( - description.contains("invalid-body"), - "the 400 omits the body problem it answers: {description}" - ); - } -} diff --git a/libs/@local/graph/atlas/src/api/mod.rs b/libs/@local/graph/atlas/src/api/mod.rs deleted file mode 100644 index 6eeafda0e33..00000000000 --- a/libs/@local/graph/atlas/src/api/mod.rs +++ /dev/null @@ -1,306 +0,0 @@ -//! The atlas read API, serving Surface v1 routes over one opened generation. -//! -//! The routes are the mutable `current` pointer, the per-generation manifest that also states one -//! caller's resolved delivery schedule, the tile, edges, and locate endpoints answering binary -//! `SALTILE` envelopes, and the JSON translate endpoint. The server pins the generation at startup -//! and serves it until restart. -//! -//! Each route lives in its own module (handler, its own path parameters, and OpenAPI documentation -//! together), so a new endpoint is a new module plus one line in [`router`]'s table. The shared -//! pieces are [`problem`] (the RFC 9457 error surface), [`extract`] (request extraction whose -//! rejections are problem documents, and the generation/variant path shape four routes address a -//! layout by), [`clause`] (the OpenAPI responses more than one route states identically), -//! [`saltile`] (binary envelope responses and their assembly worker), [`headers`] (the -//! Cache-Control postures, sent and documented from one constant), and [`mod@openapi`] (the -//! OpenAPI document and its reference page). -//! -//! Response assembly is synchronous and CPU-bound, so handlers schedule it on a rayon worker behind -//! `catch_unwind` and never inline on the async runtime. Every error a route answers is an RFC 9457 -//! problem document - handler failures directly, extraction failures through [`extract`]'s -//! wrappers; only the router's own rejections (an unmatched route, a wrong method) stay plain. -//! Binary responses send `Cache-Control: private, no-store` because the client's application-layer -//! cache is the cache. - -use alloc::sync::Arc; - -use aide::{ - axum::{ - ApiRouter, - routing::{get_with, post_with}, - }, - openapi::{ - ApiKeyLocation, Components, Info, OpenApi, ReferenceOr, SecurityRequirement, SecurityScheme, - }, -}; -use axum::{Extension, Router, extract::FromRef}; -use hash_graph_postgres_store::store::PostgresStorePool; -use hash_middleware::authentication::{ - request::ACTOR_ID_HEADER, service_secret::SERVICE_AUTH_SCHEME, -}; -use rand::rngs::SysRng; - -use self::{openapi::OpenApiDocument, visibility::ScopeResolver}; -use crate::serve::{ - Atlas, DeltaCell, DeltaEpoch, DensityBand, DensityPolicy, GraphDatabaseClient, ServeLimits, - VisibilityLimits, authorization::TokenAuthority, hydrate::CachedTypeUrlResolver, -}; - -mod authorization; -mod clause; -mod current; -mod edges; -mod extract; -mod headers; -mod locate; -mod manifest; -mod openapi; -pub(crate) mod problem; -mod saltile; -mod tile; -mod translate; -mod visibility; - -/// The OpenAPI document's top-level description. -const API_DESCRIPTION: &str = - "The read API over one published atlas generation: a zoomable map of the HASH graph, served \ - as binary `SALTILE` envelopes. - -## Bootstrap - -1. `GET /v1/atlas/current` - the generation this process serves; the one mutable read. -2. `POST /v1/atlas/generation/{generation}/manifest` - the per-generation bootstrap: the served \ - variants, the published serving limits (`limits`), the bucket schedule the tile grid follows \ - (`bucketSchedule`), and the delivery schedule this caller's own responses follow \ - (`scopeSchedule`, whose `k` a restricted decoder adds to the bucket span to attribute runs \ - to buckets). Every block except `scopeSchedule` is immutable for the generation's lifetime. \ - Every successful response issues the authority token the data routes require, and the body \ - states the view the request wants: a filter document, or nothing for the unfiltered view. -3. `POST` the tile, edges, and locate routes for binary geometry; `POST` translate for JSON \ - identity resolution. - -## Conventions - -- Query-bearing endpoints are `POST` with a JSON body; the body is part of the client's cache key. -- Binary responses ship `Cache-Control: private, no-store`: the client's application-layer cache \ - is the cache, keyed on (authorization context, generation, route, canonical body). Tile \ - labels and icons come from the generation, with a placed arrival's label from its \ - placement's captured display. Edges labels come from the display the server captured at the \ - link's currently served edition, and from the generation's payload for a fitted link it \ - holds no capture for. Edges type references and locate type and property values read \ - request-time store state. Do not retain a detailed response as an immutable generation tile. \ - Cache geometry and refetch edges and locate detail where request-time state matters. -- Every error is an RFC 9457 problem document (`application/problem+json`) whose `type` member is \ - a stable root-relative URI (`/problems/atlas/`): an absent required body answers \ - `missing-body`, a body that is not the operation's JSON shape answers `invalid-body`, an \ - unparsable tile address answers `invalid-coordinate`, and a malformed generation id answers \ - `invalid-generation`. An `unknown-generation` problem always means: re-read `current` and \ - retry. -- Authorization answers three problems. A caller the authentication middleware cannot resolve \ - answers `unauthenticated`, carrying the middleware's own status. An absent, malformed, \ - foreign, or stale `Atlas-Authority` token answers `unauthorized` (401), one uniform refusal \ - whose remedy is a fresh manifest request. A scope this process cannot resolve answers \ - `visibility-unavailable` (503). -- The binary envelope's normative contract is the `Atlas wire format` section below - this \ - document is self-contained; a decoder implements against it."; - -/// The wire-format contract, exported verbatim from `docs/wire.md`. -/// -/// The binary envelope is observable from the outside, so the OpenAPI document includes its -/// normative text rather than pointing at a repository file. -const WIRE_FORMAT: &str = include_str!("../../docs/wire.md"); - -/// The credential schemes the deployment authenticates, by document name. -/// -/// The authentication middleware resolves credentials ahead of this router and cannot write into -/// this document, so this array is the document's statement of what authenticates, and [`router`] -/// carries each scheme into the root security requirements. -// The Kratos and Cloudflare names mirror `hash-graph-authentication`'s provider constants; -// depending on that crate for three strings would pull its Kratos and JWT machinery into this one. -#[expect( - clippy::default_trait_access, - reason = "we do not want to pull in a dependency just to pin its default" -)] -fn credential_schemes() -> [(&'static str, SecurityScheme); 4] { - [ - ( - "sessionToken", - SecurityScheme::ApiKey { - location: ApiKeyLocation::Header, - name: "X-Session-Token".to_owned(), - description: Some("the caller's Kratos session token".to_owned()), - extensions: Default::default(), - }, - ), - ( - "sessionCookie", - SecurityScheme::ApiKey { - location: ApiKeyLocation::Cookie, - name: "ory_kratos_session".to_owned(), - description: Some("the caller's Kratos browser session".to_owned()), - extensions: Default::default(), - }, - ), - ( - "cloudflareAccess", - SecurityScheme::ApiKey { - location: ApiKeyLocation::Header, - name: "Cf-Access-Jwt-Assertion".to_owned(), - description: Some( - "the Cloudflare Access JWT, on deployments behind Cloudflare Access".to_owned(), - ), - extensions: Default::default(), - }, - ), - ( - "serviceDelegation", - SecurityScheme::Http { - scheme: SERVICE_AUTH_SCHEME.to_owned(), - bearer_format: None, - description: Some(format!( - "the shared service secret, with the delegated actor beside it in the \ - `{ACTOR_ID_HEADER}` header" - )), - extensions: Default::default(), - }, - ), - ] -} - -/// The shared route state. -/// -/// The pinned generation, the limits the handlers enforce and the manifest publishes, the store -/// connection detail hydration reads through, the authority every assembly path masks by - read per -/// request through [`visibility::Visibility`] - the delta cell every request captures its -/// withdrawal snapshot from at ingress, and the token authority behind the manifest's tokens, -/// whose sealed scope is the identity every data route resolves its visibility under. -#[derive(Clone)] -struct AppState { - atlas: Arc, - limits: ServeLimits, - visibility: VisibilityLimits, - tokens: Arc>, - /// The published delta snapshot cell, empty until the consumer's first publication. - /// - /// Each request loads it once at ingress and reads that one snapshot at every admission in - /// its answer. A serve without a consumer holds the empty cell for its lifetime, and every - /// load answers [`None`] at no cost. - delta: Arc, - /// The delivery-cut policy a fresh bootstrap resolves `k` under. - /// - /// [`None`] for the schedules no offset deepens, where every scope serves the recorded cut. - density: Option, - scopes: ScopeResolver, - remote: Arc, - /// The type-URL resolution the edges trailer reads through, cached for the process's life. - type_urls: Arc>>, -} - -impl FromRef> for Arc> { - fn from_ref(input: &AppState) -> Self { - Self::clone(&input.tokens) - } -} - -/// Builds the read API router over one opened generation. -/// -/// With the OpenAPI document generated at startup and served beside the API. -/// -/// A visibility proof scopes every corpus-bearing response the router serves, and every request -/// answers under the scope of the actor it names: `pool` is the store every read goes through and -/// `visibility` the window the router reuses a resolved scope for. Every route requires the actor -/// the authentication layer resolves, and serves no actor another's rows. The authority token's key -/// derives from the secret that opened the atlas, and every token seals `epoch`, the serving -/// process's delta epoch. A restarted delta register refuses the tokens issued beside its -/// predecessor. The manifest issues one per fetch, and the data routes refuse without one. -/// -/// # Panics -/// -/// This panics when the OpenAPI document fails to serialize, which the statically declared route -/// table rules out. -pub(crate) fn router( - atlas: Arc, - limits: ServeLimits, - details: Arc, - pool: Arc, - visibility: VisibilityLimits, - epoch: Option, - delta: Arc, -) -> Router { - let state = AppState { - tokens: Arc::new(TokenAuthority::new( - atlas.generation(), - atlas.wire_secret().hex_bytes(), - visibility.hard, - epoch, - SysRng, - )), - density: atlas.density_policy(DensityBand::default()), - atlas, - limits, - visibility, - scopes: ScopeResolver::new(pool, visibility), - type_urls: Arc::new(CachedTypeUrlResolver::new(Arc::clone(&details))), - remote: details, - delta, - }; - - // To increase the accuracy of response inference we manually declare - // responses for each operation. - aide::generate::infer_responses(false); - - let mut components = Components::default(); - let mut security = Vec::new(); - for (name, scheme) in credential_schemes() { - security.push(SecurityRequirement::from_iter([( - name.to_owned(), - Vec::new(), - )])); - components - .security_schemes - .insert(name.to_owned(), ReferenceOr::Item(scheme)); - } - - let mut api = OpenApi { - info: Info { - title: "HASH Atlas API".to_owned(), - description: Some(format!("{API_DESCRIPTION}\n\n---\n\n{WIRE_FORMAT}")), - version: env!("CARGO_PKG_VERSION").to_owned(), - ..Info::default() - }, - security, - components: Some(components), - ..OpenApi::default() - }; - - let router = ApiRouter::new() - .api_route( - "/v1/atlas/current", - get_with(current::handler, current::document), - ) - .api_route( - "/v1/atlas/generation/{generation}/manifest", - post_with(manifest::handler, manifest::document), - ) - .api_route( - "/v1/atlas/tile/{generation}/{variant}/{z}/{x}/{y}", - post_with(tile::handler, tile::document), - ) - .api_route( - "/v1/atlas/edges/{generation}/{variant}", - post_with(edges::handler, edges::document), - ) - .api_route( - "/v1/atlas/locate/{generation}/{variant}", - post_with(locate::handler, locate::document), - ) - .api_route( - "/v1/atlas/translate/{generation}/{variant}", - post_with(translate::handler, translate::document), - ) - .route("/v1/atlas/openapi.json", axum::routing::get(openapi::json)) - .route("/v1/atlas/openapi", axum::routing::get(openapi::html)) - .with_state(state) - .finish_api_with(&mut api, clause::middleware); - - router.layer(Extension(OpenApiDocument::new(&api))) -} diff --git a/libs/@local/graph/atlas/src/api/openapi.rs b/libs/@local/graph/atlas/src/api/openapi.rs deleted file mode 100644 index eda1a1a0b0c..00000000000 --- a/libs/@local/graph/atlas/src/api/openapi.rs +++ /dev/null @@ -1,56 +0,0 @@ -//! The self-serving documentation. -//! -//! The OpenAPI document rendered at startup and the Scalar reference page over it. - -use std::sync::LazyLock; - -use aide::{openapi::OpenApi, scalar::Scalar}; -use axum::{ - Extension, - body::Bytes, - http::header, - response::{Html, IntoResponse}, -}; - -#[derive(Debug, Clone)] -pub(crate) struct OpenApiDocument(Bytes); - -impl OpenApiDocument { - pub(crate) fn new(api: &OpenApi) -> Self { - let document = - Bytes::from(serde_json::to_string(&api).expect("the OpenAPI document serializes")); - - Self(document) - } -} - -/// Serves the OpenAPI document rendered at startup. -pub(super) async fn json( - Extension(OpenApiDocument(document)): Extension, -) -> impl IntoResponse { - ([(header::CONTENT_TYPE, "application/json")], document) -} - -/// Serves the Scalar reference page. -pub(super) async fn html() -> impl IntoResponse { - static BUNDLE: LazyLock = LazyLock::new(|| { - let html = Scalar::new("/v1/atlas/openapi.json") - .with_title("HASH Atlas API") - .html(); - - let patched = html.replace( - "--scalar-font: \"Inter\", var(--system-fonts);", - "--scalar-font: ui-sans-serif, system-ui, -apple-system, \"Segoe UI\", Roboto, \ - \"Helvetica Neue\", Arial, sans-serif;\n --scalar-font-code: ui-monospace, \ - SFMono-Regular, Menlo, Consolas, \"Liberation Mono\", monospace;", - ); - - if patched == html { - tracing::warn!("the scalar theme no longer uses Inter, fonts may render incorrectly"); - } - - Bytes::from(patched) - }); - - Html((*BUNDLE).clone()) -} diff --git a/libs/@local/graph/atlas/src/api/problem.rs b/libs/@local/graph/atlas/src/api/problem.rs deleted file mode 100644 index 0fab717f693..00000000000 --- a/libs/@local/graph/atlas/src/api/problem.rs +++ /dev/null @@ -1,425 +0,0 @@ -//! RFC 9457 problem documents, the error surface of every handler. -//! -//! The `type` member carries Surface v1's stable root-relative URIs, the body goes out as -//! `application/problem+json`, and the shared rejections - foreign generation, foreign variant - -//! live here beside the document they produce. Requests that fail before a handler runs - malformed -//! bodies, wrong content types, unparsable tile addresses - route through [`super::extract`]'s -//! wrappers and answer problem documents too. Only the router's own rejections (an unmatched route, -//! a wrong method) stay plain. - -use alloc::borrow::Cow; -use core::{num::NonZero, task}; - -use aide::{OperationOutput, generate::GenContext, openapi}; -use axum::{ - Json, - http::{self, StatusCode, header}, - response::{IntoResponse, Response}, -}; -use futures::TryFutureExt as _; -use hash_middleware::{ - authentication::{AuthenticationRejection, request::AuthenticationError}, - rate_limit::{RateLimitRejection, TooManyRequests}, -}; - -use super::AppState; -use crate::{file::generation::GenerationId, serve::VARIANTS}; - -/// The `type` member of one problem document: Surface v1's stable root-relative URIs. -#[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, schemars::JsonSchema)] -pub(super) enum ProblemType { - /// A producer bug surfacing as a 500: the assembly panicked or its worker vanished. - #[serde(rename = "/problems/atlas/internal")] - InternalError, - /// The route names a generation this process does not serve. - #[serde(rename = "/problems/atlas/unknown-generation")] - UnknownGeneration, - /// The route names a variant outside the manifest's list. - #[serde(rename = "/problems/atlas/unknown-variant")] - UnknownVariant, - /// A tile coordinate outside the zoom range or off its grid. - #[serde(rename = "/problems/atlas/invalid-coordinate")] - InvalidCoordinate, - /// A generation path segment that is not a sha256 generation id. - #[serde(rename = "/problems/atlas/invalid-generation")] - InvalidGeneration, - /// An edges body listing more tiles than the manifest's cap. - #[serde(rename = "/problems/atlas/too-many-tiles")] - TooManyTiles, - /// A tile body carrying more `coloredTypeIds` than the manifest's cap. - #[serde(rename = "/problems/atlas/too-many-types")] - TooManyTypes, - /// A translate body listing more entity ids than the manifest's cap. - #[serde(rename = "/problems/atlas/too-many-entity-ids")] - TooManyEntityIds, - /// A locate source id that does not name a visible node. - /// - /// Nonexistent, denied, and unparsable answer identically (missing = denied). - #[serde(rename = "/problems/atlas/unknown-entity")] - UnknownEntity, - /// A locate body that does not name exactly one source: `entityId` XOR `row`. - #[serde(rename = "/problems/atlas/invalid-source")] - InvalidSource, - /// A required request body that did not arrive. - #[serde(rename = "/problems/atlas/missing-body")] - MissingBody, - /// A request body that is not the operation's JSON, whether wrong content type, syntax error, - /// shape mismatch, or oversize. - #[serde(rename = "/problems/atlas/invalid-body")] - InvalidBody, - /// A caller the authentication middleware could not resolve. - /// - /// The status and detail are the middleware's own client-safe reading of the failure. - #[serde(rename = "/problems/atlas/unauthenticated")] - Unauthenticated, - /// A request whose authority token is absent, malformed, foreign, or stale. - #[serde(rename = "/problems/atlas/unauthorized")] - Unauthorized, - /// Resolving the caller's scope failed, so the process cannot say what they may see. - #[serde(rename = "/problems/atlas/visibility-unavailable")] - VisibilityUnavailable, - /// The caller is over its request budget, and `Retry-After` states when it admits again. - #[serde(rename = "/problems/atlas/too-many-requests")] - TooManyRequests, -} - -/// Serializes the problem's `status` member as its integer form. -#[expect( - clippy::trivially_copy_pass_by_ref, - reason = "serde's serialize_with contract passes fields by reference" -)] -fn status_as_u16( - status: &StatusCode, - serializer: S, -) -> Result { - serializer.serialize_u16(status.as_u16()) -} - -/// One RFC 9457 problem document. -#[derive(Debug, serde::Serialize, schemars::JsonSchema)] -pub(crate) struct Problem<'content> { - r#type: ProblemType, - title: Cow<'content, str>, - #[serde(serialize_with = "status_as_u16")] - #[schemars(with = "u16")] - status: StatusCode, - detail: Cow<'content, str>, -} - -impl<'content> Problem<'content> { - pub(super) fn new( - status: StatusCode, - r#type: ProblemType, - detail: impl Into>, - ) -> Self { - Self { - r#type, - title: Cow::Borrowed(status.canonical_reason().unwrap_or("error")), - status, - detail: detail.into(), - } - } - - /// A 500 whose source stays in the server log. - /// - /// The document carries only the static `detail`. The log records `source` at error level. - /// Driver errors and panic payloads are log material and never reach a client. - pub(super) fn internal( - source: impl core::fmt::Display, - detail: impl Into>, - ) -> Self { - let detail = detail.into(); - tracing::error!(source = %source, "{detail}"); - Self::new( - StatusCode::INTERNAL_SERVER_ERROR, - ProblemType::InternalError, - if cfg!(debug_assertions) { - detail - } else { - Cow::Borrowed("internal server error") - }, - ) - } -} - -/// Carries an authentication failure as this crate's problem document. -/// -/// The status and detail are [`AuthenticationError`]'s own client-safe readings, so this crate -/// never restates the middleware's status map. -impl From for Problem<'static> { - fn from(error: AuthenticationError) -> Self { - Self::new( - error.status_code(), - ProblemType::Unauthenticated, - error.kind().client_message(), - ) - } -} - -impl From<&AuthenticationError> for Problem<'static> { - fn from(error: &AuthenticationError) -> Self { - Self::new( - error.status_code(), - ProblemType::Unauthenticated, - error.kind().client_message(), - ) - } -} - -impl IntoResponse for Problem<'_> { - fn into_response(self) -> Response { - ( - self.status, - [(header::CONTENT_TYPE, "application/problem+json")], - Json(self), - ) - .into_response() - } -} - -impl OperationOutput for Problem<'_> { - type Inner = Self; - - fn operation_response( - ctx: &mut GenContext, - _operation: &mut openapi::Operation, - ) -> Option { - let json_schema = ctx.schema.subschema_for::>(); - let mut response = openapi::Response { - description: "an RFC 9457 problem document".into(), - ..Default::default() - }; - response.content.insert( - "application/problem+json".into(), - openapi::MediaType { - schema: Some(openapi::SchemaObject { - json_schema, - example: None, - external_docs: None, - }), - ..Default::default() - }, - ); - - Some(response) - } - - fn inferred_responses( - ctx: &mut GenContext, - operation: &mut openapi::Operation, - ) -> Vec<(Option, openapi::Response)> { - // One default response suffices because a problem carries its own status. - let response = Self::operation_response(ctx, operation) - .unwrap_or_else(|| unreachable!("`operation_response` answers every operation")); - - vec![(None, response)] - } -} - -pub(crate) struct ProblemResponse<'content> { - problem: Problem<'content>, - retry_after: Option>, -} - -impl<'content, T> From for ProblemResponse<'content> -where - T: Into>, -{ - fn from(problem: T) -> Self { - Self { - problem: problem.into(), - retry_after: None, - } - } -} - -impl From for ProblemResponse<'static> { - fn from(TooManyRequests { retry_after }: TooManyRequests) -> Self { - Self { - problem: Problem::new( - StatusCode::TOO_MANY_REQUESTS, - ProblemType::TooManyRequests, - "rate limit exceeded", - ), - retry_after: Some(retry_after), - } - } -} - -impl From for ProblemResponse<'static> { - fn from(error: RateLimitRejection) -> Self { - match error { - RateLimitRejection::TooManyRequests(too_many_requests) => too_many_requests.into(), - RateLimitRejection::InternalError => Problem::new( - StatusCode::INTERNAL_SERVER_ERROR, - ProblemType::InternalError, - "internal server error", - ) - .into(), - } - } -} - -impl From for ProblemResponse<'static> { - fn from(error: AuthenticationRejection) -> Self { - match error { - AuthenticationRejection::Authentication { - ref report, - metrics: _, - recorded: _, - } => Problem::from(report.current_context()).into(), - AuthenticationRejection::Misconfigured => Problem::internal( - "`Actor` extracted on a route without the authentication middleware", - "the caller's authentication was never resolved", - ) - .into(), - } - } -} - -impl IntoResponse for ProblemResponse<'_> { - fn into_response(self) -> Response { - let mut response = self.problem.into_response(); - if let Some(retry_after) = self.retry_after { - response - .headers_mut() - .insert(header::RETRY_AFTER, retry_after.get().into()); - } - response - } -} - -#[derive(Debug, Copy, Clone)] -pub(crate) struct IntoProblemLayer; - -impl tower::Layer for IntoProblemLayer { - type Service = IntoProblemService; - - fn layer(&self, inner: S) -> Self::Service { - IntoProblemService { inner } - } -} - -#[derive(Debug, Copy, Clone)] -pub(crate) struct IntoProblemService { - inner: S, -} - -impl tower::Service> for IntoProblemService -where - S: tower::Service, Response = Result>, - U: Into>, -{ - type Error = S::Error; - type Response = Result>; - - type Future = impl Future>; - - fn poll_ready(&mut self, cx: &mut task::Context<'_>) -> task::Poll> { - self.inner.poll_ready(cx) - } - - fn call(&mut self, req: http::Request) -> Self::Future { - self.inner - .call(req) - .map_ok(|result| result.map_err(Into::into)) - } -} - -/// Rejects a route whose generation echo does not name the pinned generation. -/// -/// A well-formed id names a resource, so an id this process does not serve is a 404, and the -/// client's recovery is to re-read `current` and retry. A malformed id never reaches here - the -/// path extractor answers `invalid-generation` (400) first. -pub(super) fn reject_generation( - state: &AppState, - generation: GenerationId, -) -> Result<(), Problem<'static>> { - if generation == state.atlas.generation() { - return Ok(()); - } - - Err(Problem::new( - StatusCode::NOT_FOUND, - ProblemType::UnknownGeneration, - format!("generation {generation} is not served; re-read /v1/atlas/current and retry"), - )) -} - -/// Refuses a request that presents no acceptable authority token. -/// -/// One uniform answer for every cause (an absent header, a malformed encoding, a failed tag, a -/// stale issue time, or an actor mismatch), so a caller learns that its presentation refused and -/// nothing about why. The refused cause reaches the server log alone. -pub(super) fn unauthorized() -> Problem<'static> { - Problem::new( - StatusCode::UNAUTHORIZED, - ProblemType::Unauthorized, - "the request presents no acceptable authority token; re-fetch the manifest presenting the \ - held token to renew, or without one to bootstrap afresh", - ) -} - -/// Refuses a request whose scope resolution failed. -/// -/// The caller's permissions are unknown, so the answer is a 503. This process cannot say what the -/// caller may see, and a later attempt may succeed. The cause stays in the server log, since a -/// resolution failure names store internals. The client reads that the scope is unavailable. -pub(super) fn visibility_unavailable(error: &(impl core::fmt::Debug + ?Sized)) -> Problem<'static> { - tracing::error!(?error, "resolving the caller's visibility failed"); - - Problem::new( - StatusCode::SERVICE_UNAVAILABLE, - ProblemType::VisibilityUnavailable, - "the caller's scope could not be resolved", - ) -} - -/// Rejects a route naming a variant this generation does not serve. -pub(super) fn reject_variant(variant: &str) -> Result<(), Problem<'static>> { - if VARIANTS.contains(&variant) { - return Ok(()); - } - - Err(Problem::new( - StatusCode::NOT_FOUND, - ProblemType::UnknownVariant, - format!("variant {variant} is not served; the manifest lists {VARIANTS:?}"), - )) -} - -#[cfg(test)] -mod tests { - use axum::http::StatusCode; - - use super::{Problem, ProblemType}; - - #[test] - fn type_member_is_a_root_relative_uri() { - let problem = Problem::new( - StatusCode::NOT_FOUND, - ProblemType::UnknownGeneration, - "re-bootstrap via /v1/atlas/current", - ); - let document = serde_json::to_value(&problem).expect("problem documents serialize"); - - assert_eq!(document["type"], "/problems/atlas/unknown-generation"); - assert_eq!(document["status"], 404); - } - - #[test] - fn internal_problems_redact_their_source() { - let problem = Problem::internal( - "connection refused: db=secret host=10.0.0.7", - "the detail hydration failed", - ); - let document = serde_json::to_value(&problem).expect("problem documents serialize"); - - assert_eq!(document["type"], "/problems/atlas/internal"); - assert_eq!(document["detail"], "the detail hydration failed"); - assert!( - !document.to_string().contains("10.0.0.7"), - "the source error must stay out of the document" - ); - } -} diff --git a/libs/@local/graph/atlas/src/api/saltile.rs b/libs/@local/graph/atlas/src/api/saltile.rs deleted file mode 100644 index 663424fbf14..00000000000 --- a/libs/@local/graph/atlas/src/api/saltile.rs +++ /dev/null @@ -1,98 +0,0 @@ -//! Binary envelope responses and the worker that assembles them. -//! -//! [`Saltile`] wraps assembled envelope bytes with the family's media type and the no-store cache -//! posture. [`spawn`] runs the CPU-bound assembly off the async runtime. - -use alloc::borrow::Cow; -use core::panic::UnwindSafe; - -use aide::{OperationOutput, generate::GenContext, openapi}; -use axum::{ - http::header, - response::{IntoResponse, Response}, -}; - -use super::{headers, problem::Problem}; -use crate::offload::{self, OffloadError}; - -/// The tile response media type, the `SALTILE` family at version 1. -const SALTILE: &str = "application/vnd.hash.saltile-v1"; - -/// `SALTILE` envelope bytes as a response. -/// -/// The family's media type, no-store because the client's application-layer cache is the cache. -pub(super) struct Saltile(Vec); - -impl Saltile { - /// Wraps assembled envelope bytes for delivery. - pub(super) const fn new(bytes: Vec) -> Self { - Self(bytes) - } -} - -impl IntoResponse for Saltile { - fn into_response(self) -> Response { - ( - [ - (header::CONTENT_TYPE, SALTILE), - (header::CACHE_CONTROL, headers::NO_STORE), - ], - self.0, - ) - .into_response() - } -} - -impl OperationOutput for Saltile { - type Inner = Vec; - - fn operation_response( - _ctx: &mut GenContext, - _operation: &mut openapi::Operation, - ) -> Option { - let mut response = openapi::Response { - description: "a SALTILE envelope".into(), - ..Default::default() - }; - - response - .content - .insert(SALTILE.into(), openapi::MediaType::default()); - response.headers.insert( - "Cache-Control".to_owned(), - headers::cache_control( - headers::NO_STORE, - "binary envelopes key on the request body, which shared caches cannot see; the \ - client's application-layer cache is the cache", - ), - ); - - Some(response) - } - - fn inferred_responses( - ctx: &mut GenContext, - operation: &mut openapi::Operation, - ) -> Vec<(Option, openapi::Response)> { - let response = Self::operation_response(ctx, operation) - .unwrap_or_else(|| unreachable!("`operation_response` answers every operation")); - - vec![(Some(openapi::StatusCode::Code(200)), response)] - } -} - -/// Runs CPU-bound response assembly on a rayon worker, answering a panic as an internal problem. -pub(super) async fn spawn( - work: impl FnOnce() -> T + Send + UnwindSafe + 'static, -) -> Result> { - offload::run(work).await.map_err(|error| match error { - OffloadError::Panicked(payload) => Problem::internal( - payload.unwrap_or(Cow::Borrowed("non-string panic payload")), - "the response assembly panicked", - ), - OffloadError::Vanished => Problem::internal( - "the assembly worker dropped its channel", - "the assembly worker vanished", - ), - }) -} diff --git a/libs/@local/graph/atlas/src/api/tile.rs b/libs/@local/graph/atlas/src/api/tile.rs deleted file mode 100644 index 9f221a5b24c..00000000000 --- a/libs/@local/graph/atlas/src/api/tile.rs +++ /dev/null @@ -1,144 +0,0 @@ -//! `POST /v1/atlas/tile/{generation}/{variant}/{z}/{x}/{y}`: one tile as `SALTILET` bytes. - -use alloc::sync::Arc; - -use aide::transform::TransformOperation; -use axum::{extract::State, http::StatusCode}; - -use super::{ - AppState, clause, - extract::{Body, Coordinates, Generation, VariantPath}, - problem::{Problem, ProblemType, reject_generation, reject_variant}, - saltile::{Saltile, spawn}, - visibility::{Visibility, view_problem}, -}; -use crate::{ - salt::wire::tile::TileCoordinate, - serve::{TileError, TileQuery, TileRequest}, -}; - -/// The operation's description. -const DESCRIPTION: &str = - "Returns one tile of the map: the points the generation's level-of-detail schedule delivers \ - at `z/x/y`, as a `SALTILET` binary envelope of positions, row ids, the optional type mask, \ - and the children bitmap. - -The JSON body is optional; an absent body reads as the all-defaults query. `mode` selects `delta` \ - (this tile's own additions - the default) or `total` (every ancestor's delivery accumulated, \ - so the tile renders alone). - -`coloredTypeIds` lists versioned type URLs and adds the `TYPE_MASK` column: bit `i` of a point's \ - mask is 1 exactly when the point carries the request's type `i` or one of its descendants. \ - An id that matches no type in this generation is legal and reads 0 in every mask. The \ - manifest's `limits.coloredTypeIds` caps the list. - -`detail: \"auxiliary\"` adds the detail trailer - per-point labels and icons read from the \ - generation, with a placed arrival's label from its placement's captured display. \ - `\"minimal\"`, the default, sends geometry alone. A client must not retain a detailed \ - response as an immutable generation tile. Cache geometry. Section 2 of the Atlas wire format \ - defines the detail that clients refetch where request-time state matters. - -Filtering binds at the manifest. This body has no `filter` field, and an unknown member is \ - rejected as `invalid-body`."; - -/// The `z/x/y` grid cell. -/// -/// Extracted through [`Coordinates`]: an unparsable numeric segment answers the -/// `invalid-coordinate` problem, the same slug an out-of-range address earns from assembly. -#[derive(Debug, serde::Deserialize, schemars::JsonSchema)] -pub(super) struct CellPath { - /// The zoom, a subdivision depth where `0` addresses the root. - z: u8, - /// The cell's `x` index on the `2^z` grid. - x: u32, - /// The cell's `y` index on the `2^z` grid. - y: u32, -} - -/// `POST /v1/atlas/tile/{generation}/{variant}/{z}/{x}/{y}`: one tile as `SALTILET` bytes. -/// -/// An absent body reads as the all-defaults query. -pub(super) async fn handler( - State(state): State>, - visibility: Visibility, - Generation(VariantPath { - generation, - variant, - }): Generation, - Coordinates(CellPath { z, x, y }): Coordinates, - query: Option>, -) -> Result> { - reject_generation(&state, generation)?; - reject_variant(&variant)?; - - let request = TileRequest { - coordinate: TileCoordinate { z, x, y }, - query: query.map_or_else(TileQuery::default, |Body(query)| query), - }; - - // The whole pipeline is one synchronous call on a rayon worker. Binding runs there too, so - // a scope's first request pays its cascade build off the async runtime. - let atlas = Arc::clone(&state.atlas); - let limits = state.limits.tile; - - let result = spawn(move || { - let view = visibility.view(&atlas)?; - - atlas.tile(&request, limits, view) - }) - .await?; - - let bytes = match result { - Ok(bytes) => bytes, - Err(error @ TileError::Types { .. }) => { - return Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::TooManyTypes, - error.to_string(), - )); - } - Err(error @ (TileError::Depth { .. } | TileError::Grid { .. })) => { - return Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidCoordinate, - error.to_string(), - )); - } - // A stale sealed offset answers the uniform refusal. A mismatched pair or width names - // an input this process produced and answers the internal problem. - Err(TileError::View(error)) => return Err(view_problem(error)), - }; - - Ok(Saltile::new(bytes)) -} - -/// Documents the operation. -pub(super) fn document(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation - .id("tile") - .summary("One tile as SALTILET envelope bytes") - .description(DESCRIPTION) - .with(clause::optional_body) - .with(clause::describe_body( - "the query context; absent reads as the all-defaults query", - )) - .response_with::<200, Saltile, _>(|response| { - response.description("a `SALTILET` envelope: the tile's delivered geometry") - }) - .response_with::<400, Problem<'static>, _>(|response| { - response.description( - "`invalid-coordinate` (an unparsable `z`/`x`/`y` segment, a zoom past the deepest \ - cut, or `x`/`y` off the `2^z` grid), `invalid-generation` (a malformed \ - generation id), `too-many-types` (`coloredTypeIds` exceeds the manifest's cap), \ - or `invalid-body` (a body that is not JSON)", - ) - }) - .with(clause::invalid_body_data) - .with(clause::unauthorized) - .response_with::<404, Problem<'static>, _>(|response| { - response.description( - "`unknown-generation` or `unknown-variant`: re-read `current` and retry", - ) - }) - .with(clause::any_problem) -} diff --git a/libs/@local/graph/atlas/src/api/translate.rs b/libs/@local/graph/atlas/src/api/translate.rs deleted file mode 100644 index f463ada6c41..00000000000 --- a/libs/@local/graph/atlas/src/api/translate.rs +++ /dev/null @@ -1,129 +0,0 @@ -//! `POST /v1/atlas/translate/{generation}/{variant}`. -//! -//! Upstream entity ids to atlas row ids, plus wire-frame positions for nodes. - -use alloc::sync::Arc; - -use aide::{axum::IntoApiResponse, transform::TransformOperation}; -use axum::{ - Json, - extract::State, - http::{StatusCode, header}, -}; - -use super::{ - AppState, clause, - extract::{Body, Generation, VariantPath}, - headers, - problem::{Problem, ProblemType, reject_generation, reject_variant}, - saltile::spawn, - visibility::Visibility, -}; -use crate::serve::{TranslateError, TranslateRequest, TranslateResponse}; - -/// The operation's description. -const DESCRIPTION: &str = - "Translates upstream entity ids (`webId~entityUuid`) into the row ids and positions the \ - binary routes speak. - -Use it to place entities fetched elsewhere onto the map. A resolved node answers its row id (the \ - `ROW_IDS` value every binary response uses) plus its position in the map's coordinate frame; \ - a resolved edge answers its two endpoints' row ids. Edges have no row id of their own - \ - binary responses identify an edge by its link entity id, which the requester already holds. - -Row ids are opaque per-generation values, sparse in the full 32-bit range: consistent across every \ - route of one generation, carrying no ordering, adjacency, or count information, and not \ - stable across generations - re-translate after a generation change. An id can also name an \ - entity placed since the generation was fitted. Those ids die with the serving session that \ - minted them, and the uniform token refusal that follows a restart is the signal to \ - re-bootstrap and re-translate. - -The response is two maps - `nodes` and `edges` - keyed by the requested id strings echoed \ - verbatim, so which map answers carries the kind. An id that resolves to nothing is an absent \ - key, never an error and never a null entry: nonexistent ids, draft ids, and entities the \ - caller cannot see are indistinguishable. - -The `edges` map answers a link id when the caller may see the link row and both of its endpoints; \ - otherwise the id is absent, indistinguishable from an id belonging to neither domain. - -The JSON body is required; the manifest's `limits.translateEntityIds` caps the id list. Duplicates \ - are legal and collapse."; - -/// `POST /v1/atlas/translate/{generation}/{variant}`: upstream entity ids to atlas identity. -/// -/// The id list is the request's subject, so this route requires a body. -pub(super) async fn handler( - State(state): State>, - visibility: Visibility, - Generation(VariantPath { - generation, - variant, - }): Generation, - Body(request): Body, -) -> Result> { - reject_generation(&state, generation)?; - reject_variant(&variant)?; - - let atlas = Arc::clone(&state.atlas); - let limits = state.limits.translate; - match spawn(move || { - atlas.translate( - request, - limits, - visibility.proof(), - visibility.delta(), - visibility.cohort(), - ) - }) - .await? - { - Ok(response) => Ok(([(header::CACHE_CONTROL, headers::NO_STORE)], Json(response))), - Err(error @ TranslateError::Ids { .. }) => Err(Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::TooManyEntityIds, - error.to_string(), - )), - } -} - -/// Documents the operation. -pub(super) fn document(operation: TransformOperation<'_>) -> TransformOperation<'_> { - operation - .id("translate") - .summary("Upstream entity ids to atlas row ids and positions") - .description(DESCRIPTION) - .with(clause::describe_body( - "the translate request; the `entityIds` list is the request's subject", - )) - .response_with::<200, Json, _>(|mut response| { - response.inner().headers.insert( - "Cache-Control".to_owned(), - headers::cache_control( - headers::NO_STORE, - "the response keys on the request body, which shared caches cannot see; the \ - client's application-layer cache is the cache", - ), - ); - response.description( - "two maps keyed by the requested id echoed verbatim; unresolvable ids are absent \ - keys", - ) - }) - .response_with::<400, Problem<'static>, _>(|response| { - response.description( - "`too-many-entity-ids`, `invalid-generation`, `missing-body`, or `invalid-body` \ - (a body that is not JSON)", - ) - }) - .with(clause::invalid_body_data) - .with(clause::unauthorized) - .response_with::<404, Problem<'static>, _>(|response| { - response.description( - "`unknown-generation` or `unknown-variant`: re-bootstrap through `current`", - ) - }) - .default_response_with::, _>(|response| { - response - .description("any other problem; `internal` marks a server-side assembly failure") - }) -} diff --git a/libs/@local/graph/atlas/src/api/visibility.rs b/libs/@local/graph/atlas/src/api/visibility.rs deleted file mode 100644 index ae3cda3db96..00000000000 --- a/libs/@local/graph/atlas/src/api/visibility.rs +++ /dev/null @@ -1,500 +0,0 @@ -//! One request's visibility, resolved from its admitted authority token. -//! -//! A [`VisibilityProof`] masks every corpus-bearing response, and [`Visibility`] is that proof for -//! one request. Extraction admits the presented token through [`Scope`]'s own extraction, then -//! resolves the scope the token seals (actor and filter digest, the visibility key's own fields) -//! through the store, held in the visibility cache for its reuse window. A data route that resolves -//! visibility without an admitted token is unrepresentable, because the key's identity comes out of -//! the sealed scope. -//! -//! The caller is the actor the authentication middleware resolved, and the sealed scope binds -//! that identity. A request whose authentication fails answers the middleware's -//! own status, a refused token answers 401, a filter document that does not compile answers 400, -//! and a request the store cannot resolve for any other reason answers 503. [`proof_problem`] is -//! that split. Every refusal happens before any assembly reads the request. - -use alloc::sync::Arc; -use std::time::Instant; - -use aide::{OperationInput, generate::GenContext, openapi}; -use axum::{ - extract::FromRequestParts, - http::{StatusCode, request::Parts}, -}; -use hash_graph_postgres_store::store::{PostgresStorePool, error::StoreError}; -use hash_graph_store::{filter::Filter, pool::StorePool as _}; -use rand::TryCryptoRng; -use type_system::{knowledge::Entity, principal::actor::ActorId}; - -use super::{ - AppState, - problem::{Problem, ProblemType, unauthorized, visibility_unavailable}, -}; -use crate::serve::{ - Atlas, CutOffset, DeltaSnapshot, PlacementCohort, View, ViewError, ViewOccupancy, - VisibilityLimits, VisibilityProof, - authorization::Scope, - cache::{ - CacheEntry, PendingCacheEntry, VisibilityCache, - scope::{CacheKey, FilterDigest}, - }, - hydrate::{ - MaskingActor, - compile::{ProofError, visibility_proof}, - }, -}; - -/// The resolution state behind every [`Visibility`]. -/// -/// A request's visibility comes from the store, through the cache. -/// -/// One cache and one store handle serve the whole router, so concurrent requests for one scope -/// resolve once and the held entry answers a returning caller. -#[derive(Clone)] -pub(super) struct ScopeResolver { - /// The store permission resolution reads through. - pool: Arc, - /// Resolved scopes, held for their reuse window. - cache: Arc, -} - -impl ScopeResolver { - /// Builds the resolution state over `pool`, holding resolved scopes under `limits`. - pub(super) fn new(pool: Arc, limits: VisibilityLimits) -> Self { - Self { - pool, - cache: Arc::new(VisibilityCache::new(limits)), - } - } -} - -impl core::fmt::Debug for ScopeResolver { - /// Names the cache without the store handle, which carries no [`Debug`]. - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - fmt.debug_struct("ScopeResolver") - .field("cache", &self.cache) - .finish_non_exhaustive() - } -} - -/// One resolved scope's delivery inputs. -/// -/// These are the values a scope holds per resolution rather than per request. The manifest consumes -/// this value directly. The data routes receive it through [`Visibility`], which adds the request's -/// token-bound cut offset. -#[derive(Debug, Clone)] -pub(super) struct Resolved { - /// The held entry the scope resolved to. - entry: Arc, -} - -impl Resolved { - /// Returns the scope's occupancy aggregate, absent for a corpus proof. - /// - /// The corpus schedule has one cut per zoom and takes no offset, so an operator scope holds - /// no aggregate and an issuance over it seals zero. A scoped entry aggregates once at - /// resolution, from the store's answer alone, and every issuance under the entry reads that - /// value. - pub(super) fn occupancy(&self) -> Option<&ViewOccupancy> { - self.entry.occupancy() - } - - /// Binds the resolved scope at `k`, the view the manifest's `scopeSchedule` describes. - /// - /// # Errors - /// - /// As [`View::of`]. - pub(super) fn view( - &self, - atlas: &Atlas, - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - k: CutOffset, - ) -> Result, ViewError> { - View::of(atlas, &self.entry, k, None) - } -} - -/// The resolved scope for one request, plus the token-bound delivery-cut offset. -#[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] -#[derive(Debug, Clone)] -pub(super) struct Visibility { - /// The held entry the request's scope resolved to. - /// - /// It hands a request the proof, the census resolved with it, and the view's delivery - /// schedule. One resolution per scope answers every request under that scope, so the root - /// tile's global metadata costs no walk on the request that reads it. - entry: Arc, - /// The delivery-cut offset the presented token seals. - /// - /// Sealed at the manifest by the density policy and read back at admission, so the served cut - /// and the declared cut are the same value by construction. - k: CutOffset, - /// The withdrawal residue of the snapshot captured at ingress, before any proof or cache - /// work. - /// - /// One owned handle pins one publication for the whole request, so every admission in the - /// answer reads the same snapshot however many publications land while it runs. [`None`] - /// before the consumer's first publication, for a serve that starts no consumer, and for a - /// capture the entry's masks already folded, whose subtraction would edit nothing. - delta: Option>, -} - -impl Visibility { - /// Returns the rows the request may see. - pub(super) fn proof(&self) -> &VisibilityProof { - self.entry.proof() - } - - /// Returns the actor this request's hydrations mask properties for. - pub(super) fn masking(&self) -> MaskingActor { - self.entry.masking() - } - - /// Binds the request's delivery view, the held resolution read at the token's cut offset. - /// - /// Every data route calls this once and hands the result to assembly, so the request boundary - /// checks the pairing of proof, census and schedule rather than each endpoint restating it. - /// - /// # Errors - /// - /// As [`View::of`]. A route converts the refusal into its own error union and answers it - /// through [`view_problem`]. - pub(super) fn view(&self, atlas: &Atlas) -> Result, ViewError> { - View::of(atlas, &self.entry, self.k, self.delta.as_deref()) - } - - /// Returns the ingress capture's withdrawal residue, [`None`] when nothing is left to - /// subtract. - /// - /// Translate constructs no [`View`] and takes the same captured value explicitly. - #[expect( - clippy::missing_const_for_fn, - reason = "`Option::as_deref` needs `Arc: [const] Deref`, which the standard library does \ - not provide" - )] - pub(super) fn delta(&self) -> Option<&DeltaSnapshot> { - self.delta.as_deref() - } - - /// Returns the entry's placement cohort, the arrivals snapshot its resolution read. - /// - /// Arrival-sensitive reads take slots, placement payload, and the accepted row universe from - /// this value, and the ingress capture above contributes current withdrawals alone. - pub(super) fn cohort(&self) -> PlacementCohort<'_> { - self.entry.cohort() - } -} - -impl FromRequestParts> for Visibility -where - R: TryCryptoRng + Send, -{ - type Rejection = Problem<'static>; - - async fn from_request_parts( - parts: &mut Parts, - state: &AppState, - ) -> Result { - // One ingress capture before any proof or cache work, so the request's admissions - // cannot straddle a publication. - let delta = state.delta.load_full(); - - let scope = Scope::from_request_parts(parts, state).await?; - let resolved = resolve(state, scope.actor.into(), scope.filter.digest(), None).await?; - - // A capture the entry's masks already folded leaves no residue, so the routes skip - // their admission walks whole instead of subtracting nothing. - let delta = delta.filter(|ingress| !resolved.entry.folded(ingress)); - - Ok(Self { - entry: resolved.entry, - k: scope.k, - delta, - }) - } -} - -impl OperationInput for Visibility { - /// Documents what admission documents: the token, a required request header. - /// - /// Extraction admits through [`Scope`], so the documentation delegates to the same impl and - /// the two extractors cannot drift apart. - fn operation_input(ctx: &mut GenContext, operation: &mut openapi::Operation) { - ::operation_input(ctx, operation); - } -} - -/// Resolves one scope's visibility through the store, held in the cache for its reuse window. -/// -/// The manifest resolves in its handler body, after its generation and continuity judgments. The -/// data routes resolve through [`Visibility`]'s extraction. Both paths call this function, so one -/// scope holds one entry wherever a caller asks for its resolution. -/// -/// A filtered resolution compiles `document`, the filter's bytes as presented, into the proof. A -/// caller holding only the digest - a data route, whose scope travels sealed - reads the held -/// entry's copy, which is what lets the soft window revalidate a filtered scope without a client -/// round trip. -/// -/// # Errors -/// -/// The uniform `401` problem for a filtered scope with no document held anywhere - the client -/// re-presents the document at the manifest - and, for a resolution that fails, whichever problem -/// [`proof_problem`] gives the failing stage: `400` for the caller's own filter, the internal -/// problem for this deployment's, and the `503` visibility problem for every condition of the -/// store's. -pub(super) async fn resolve( - state: &AppState, - actor: ActorId, - filter: Option, - document: Option>, -) -> Result> { - let key = CacheKey { - generation: state.atlas.generation(), - actor, - filter, - }; - - let document = match (filter, document) { - (Some(_), None) => Some( - state - .scopes - .cache - .get(&key) - .await - .and_then(|entry| entry.filter_document()) - .ok_or_else(unauthorized)?, - ), - (_, document) => document, - }; - - let pool = Arc::clone(&state.scopes.pool); - let atlas = Arc::clone(&state.atlas); - let cell = Arc::clone(&state.delta); - - let entry = state - .scopes - .cache - .resolve(key, Instant::now(), async move || { - // The resolution acquires a connection for itself alone, and only a miss reaches here: - // a held scope answers without touching the pool. - let store = pool - .acquire(None) - .await - .map_err(|report| ProofError::Connect(report.change_context(StoreError)))?; - - // Filter protection is the deployment's own setting, read from the store the - // resolution runs against, so a filtered resolution carries the protection that - // store's own reads carry. - let protection = Arc::clone(&store.settings); - - let filter = document - .as_deref() - .map(serde_json::from_slice::>) - .transpose() - .map_err(ProofError::Document)?; - - // One load is the whole resolution's snapshot: the proof resolves against it and the - // entry binds it, so the slots the mask admits and the placements requests read come - // from one publication. - let cohort = cell.load_full(); - - let (proof, masking) = visibility_proof( - actor, - filter.as_ref(), - &protection.filter_protection, - &store, - &atlas, - PlacementCohort::of(cohort.as_deref()), - ) - .await?; - - PendingCacheEntry::of(atlas, proof, masking, document, cohort).await - }) - .await - .map_err(|error| proof_problem(&error))?; - - Ok(Resolved { entry }) -} - -/// Answers one failed resolution with the problem its failing stage earns. -/// -/// Three readings, because a caller cannot repair all three the same way. [`ProofError::Filter`] is -/// the caller's own filter document failing to compile against the entity query surface, which no -/// retry repairs. It answers `400` `invalid-body`, the problem an unparsable document already earns -/// at the manifest. [`ProofError::Convert`] joins that reading, since a filter parameter that -/// does not match its path's type is the caller's document to fix. -/// [`ProofError::PolicyFilter`] is this deployment's policy set failing to compile, so it answers -/// the internal problem with the compiler's message in the log rather than the response. Every -/// remaining stage answers the `503`, because the process cannot say what the caller may see and a -/// later attempt may succeed. -/// -/// [`ProofError::Query`] stays a `503` even though a caller's document can provoke it (a parameter -/// whose type the compiler accepts and Postgres rejects) because that one variant also carries real -/// store faults. Reading every filtered query failure as the caller's would answer an outage with a -/// `400`. The match lists every variant rather than defaulting, so a new failing stage has to -/// choose its status instead of inheriting one. -fn proof_problem(error: &ProofError) -> Problem<'static> { - match error { - ProofError::Filter(report) => Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidBody, - format!( - "the filter document does not compile: {}", - report.current_context() - ), - ), - ProofError::Convert(report) => Problem::new( - StatusCode::BAD_REQUEST, - ProblemType::InvalidBody, - format!( - "a filter parameter does not match its path's type: {}", - report.current_context() - ), - ), - ProofError::PolicyFilter(report) => Problem::internal( - report.current_context(), - "compiling the policy filter failed", - ), - ProofError::Connect(_) - | ProofError::Policies(_) - | ProofError::Document(_) - | ProofError::Query(_) - | ProofError::Rows(_) - | ProofError::ComputeView(_) => visibility_unavailable(error), - } -} - -/// The problem one refused binding answers. -/// -/// A sealed offset the view cannot serve is stale authority rather than a request defect. An -/// operator view takes no offset, so a token carrying one was issued under a contract this -/// process no longer serves. That answers the uniform `401`, whose stated remedy is a fresh -/// manifest request, and the issuance that request runs seals the offset the view does serve. -/// The combination stays a server defect in the log, because no current issuance can produce it. -/// -/// Every other binding failure names an input this process produced and answers the internal -/// problem. -pub(super) fn view_problem(error: ViewError) -> Problem<'static> { - match error { - ViewError::Offset(_) => { - tracing::error!( - ?error, - "a presented token sealed an offset its view cannot serve" - ); - - unauthorized() - } - ViewError::Contract | ViewError::Schedule(_) => { - Problem::internal(error, "delivery refused its schedule") - } - } -} - -#[cfg(test)] -mod tests { - use error_stack::Report; - use hash_graph_postgres_store::store::postgres::query::SelectCompilerError; - - use super::{proof_problem, view_problem}; - use crate::serve::{ - CutOffset, ViewError, hydrate::compile::ProofError, schedule::ScheduleWidthError, - }; - - /// One real compiler error, reused so the mapping is visibly a statement about the failing - /// stage rather than about the error inside it. - /// - /// `UnsupportedEmbeddingPath` is a failure a caller's own filter can produce, which is why it - /// is the one used for both stages here: the same value answers `400` under - /// [`ProofError::Filter`] and `500` under [`ProofError::PolicyFilter`]. - fn compiler_error() -> Report { - Report::new(SelectCompilerError::UnsupportedEmbeddingPath) - } - - /// Renders one problem as the document a client reads. - fn document(error: &ProofError) -> serde_json::Value { - serde_json::to_value(proof_problem(error)).expect("problem documents serialize") - } - - /// The caller's own filter failing to compile is the caller's to repair: `400 invalid-body`. - #[test] - fn caller_filter_that_does_not_compile_answers_invalid_body() { - let document = document(&ProofError::Filter(compiler_error())); - - assert_eq!(document["status"], 400); - assert_eq!(document["type"], "/problems/atlas/invalid-body"); - assert!( - document["detail"] - .as_str() - .expect("the problem carries a detail") - .contains("binary-quantized"), - "the caller is not told what about its document failed: {document:#}" - ); - } - - /// This deployment's policy filter failing to compile answers the internal problem, and the - /// compiler's message stays in the log. - #[test] - fn policy_filter_answers_the_internal_problem_without_its_message() { - let document = document(&ProofError::PolicyFilter(compiler_error())); - - assert_eq!(document["status"], 500); - assert_eq!(document["type"], "/problems/atlas/internal"); - assert!( - !document["detail"] - .as_str() - .expect("the problem carries a detail") - .contains("embedding"), - "the internal problem leaks the compiler's message: {document:#}" - ); - } - - /// Every other stage is a condition of the store's, and answers the `503`. - /// - /// [`ProofError::Document`] stands for the bucket: it is the one store-stage variant a test can - /// construct, since `tokio_postgres::Error` has no public constructor. What keeps the rest of - /// the bucket honest is not this test but the mapping's exhaustive match - a new failing stage - /// does not compile until it has chosen a status. - #[test] - fn store_stage_answers_the_visibility_problem() { - let parse = serde_json::from_str::("not a number").expect_err("the parse fails"); - let document = document(&ProofError::Document(parse)); - - assert_eq!(document["status"], 503); - assert_eq!(document["type"], "/problems/atlas/visibility-unavailable"); - } - - /// A sealed offset the view cannot serve answers the uniform refusal, not the internal problem. - /// - /// The client action for a stale authority is a fresh manifest request, and that issuance - /// reseals - /// the offset the view does serve. Every other binding failure runs beside it here, so the case - /// states which refusals are the caller's to act on and which are this process reporting - /// itself. - #[test] - fn refused_offset_answers_the_uniform_refusal() { - let refused = serde_json::to_value(view_problem(ViewError::Offset(CutOffset::new(1)))) - .expect("problem documents serialize"); - - assert_eq!(refused["status"], 401); - assert_eq!(refused["type"], "/problems/atlas/unauthorized"); - - let width = ScheduleWidthError { - max_tile_depth: 3, - span: 1, - k: CutOffset::new(31), - }; - for defect in [ViewError::Contract, ViewError::Schedule(width)] { - let document = - serde_json::to_value(view_problem(defect)).expect("problem documents serialize"); - - assert_eq!(document["status"], 500, "{defect:?} is a server defect"); - assert_eq!(document["type"], "/problems/atlas/internal"); - } - } -} diff --git a/libs/@local/graph/atlas/src/cli/mod.rs b/libs/@local/graph/atlas/src/cli/mod.rs index 7bf30b4a4a9..09027b0472c 100644 --- a/libs/@local/graph/atlas/src/cli/mod.rs +++ b/libs/@local/graph/atlas/src/cli/mod.rs @@ -1,8 +1,7 @@ -//! The operator commands that fit a generation and serve the atlas. +//! The operator commands that fit a generation. //! //! The `hash-graph atlas` subcommand is one entry point. [`FitArgs`] and [`FitCommand`] run one -//! production generation over the live store. [`ServeArgs`] and [`ServeCommand`] open the root's -//! active generation and build the read-API router ([`crate::api`]) the graph binary hosts. +//! production generation over the live store. //! //! The standalone `hash-graph-atlas` binary is the other entry point, and the `cli` feature gates //! its shell. Its command line carries the fit command over its own store flags ([`PostgresArgs`]) @@ -46,13 +45,11 @@ pub use self::{ embedder::{EmbedderArgs, EmbedderError}, fit::{FitArgs, FitCommand, FitError, FitVerdict}, postgres::{ConnectError, PostgresArgs, connect}, - serve::{ServeArgs, ServeCommand, ServeError, ServeOptions}, }; use crate::{device::PinnedDevice, file::generation::GenerationRoot}; pub use crate::{ integrity::{EmptyPasswordError, PasswordString, SecretString}, salt::runner::operator::{ClassifierSource, Options, Placement, RunError, Summary}, - serve::{EmbeddingWorkflow, LocateLimits, TileLimits, TranslateLimits, VisibilityLimits}, }; mod dump; @@ -60,7 +57,6 @@ mod embedder; mod fit; mod postgres; mod report; -mod serve; mod shell; #[cfg(feature = "cli")] mod tui; diff --git a/libs/@local/graph/atlas/src/cli/serve.rs b/libs/@local/graph/atlas/src/cli/serve.rs deleted file mode 100644 index 205d4c56092..00000000000 --- a/libs/@local/graph/atlas/src/cli/serve.rs +++ /dev/null @@ -1,530 +0,0 @@ -//! Opening the active generation behind the read-API router. - -use alloc::sync::Arc; -use core::{error::Error, fmt, time::Duration}; - -use axum::Router; -use clap::Args; -use hash_graph_postgres_store::store::PostgresStorePool; -use hash_middleware::{ - authentication::{ - AuthenticationLayer, AuthenticationMetrics, provider::AuthenticationProvider, - }, - rate_limit::{IpGateLayer, PrincipalLimitLayer, RateLimitConfig, RateLimiters}, - telemetry::HttpTracingLayer, -}; -use rand::rngs::SysRng; -use tower::ServiceBuilder; -use type_system::principal::actor::ActorId; - -use super::RootArgs; -use crate::{ - api::{self, problem::IntoProblemLayer}, - device::PinnedDevice, - file::generation::{CurrentError, GenerationRoot}, - integrity::{SecretHexBytesValueParser, SecretString}, - serve::{ - Atlas, DeltaCell, DeltaConsumer, DeltaEpoch, DeltaPolling, DeltaRegister, EdgesLimits, - EmbeddingWorkflow, GraphDatabaseClient, LocateLimits, OpenAtlasError, OpenOptions, - PlacementError, ServeLimits, StagingArm, TileLimits, TranslateLimits, VisibilityLimits, - WireSecret, - }, -}; - -/// The per-request serving limits. -/// -/// Every default reads off [`ServeLimits::default`], so the default values live in exactly one -/// place and `--help` renders them. The manifest publishes whatever these resolve to - the handlers -/// enforce the same values by construction. -#[derive(Debug, Args)] -struct LimitsArgs { - /// Most `coloredTypeIds` one tile request may carry. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_COLORED_TYPE_IDS", - default_value_t = ServeLimits::default().tile.colored_type_ids, - )] - colored_type_ids: u32, - - /// Most tiles one edges request may list. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_EDGES_TILES", - default_value_t = ServeLimits::default().edges.tiles, - )] - edges_tiles: u32, - - /// Most edges one response delivers before rank truncation. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_EDGES", - default_value_t = ServeLimits::default().edges.edges, - )] - edges: u32, - - /// Most entity ids one translate request may carry. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_TRANSLATE_ENTITY_IDS", - default_value_t = ServeLimits::default().translate.entity_ids, - )] - translate_entity_ids: u32, - - /// Most ego-graph edges one locate response delivers before the nearest-partner truncation. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_LOCATE_EDGES", - default_value_t = ServeLimits::default().locate.edges, - )] - locate_edges: u32, - - /// Most properties one located source delivers in its trailer map. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_LOCATE_PROPERTIES", - default_value_t = ServeLimits::default().locate.properties, - )] - locate_properties: u32, - - /// Most direct types one locate edge delivers. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_LOCATE_LINK_TYPE_IDS", - default_value_t = ServeLimits::default().locate.link_type_ids, - )] - locate_link_type_ids: u32, - - /// Most properties one locate edge delivers. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_LIMIT_LOCATE_LINK_PROPERTIES", - default_value_t = ServeLimits::default().locate.link_properties, - )] - locate_link_properties: u32, -} - -impl From for ServeLimits { - fn from(args: LimitsArgs) -> Self { - Self { - tile: TileLimits { - colored_type_ids: args.colored_type_ids, - }, - edges: EdgesLimits { - tiles: args.edges_tiles, - edges: args.edges, - }, - locate: LocateLimits { - edges: args.locate_edges, - properties: args.locate_properties, - link_type_ids: args.locate_link_type_ids, - link_properties: args.locate_link_properties, - }, - translate: TranslateLimits { - entity_ids: args.translate_entity_ids, - }, - } - } -} - -/// The delta consumer's polling knobs and its opt-out. -/// -/// Every default reads off [`DeltaPolling`]'s pinned values, so the defaults live in exactly one -/// place and `--help` renders them. -#[derive(Debug, Args)] -struct DeltaArgs { - /// Seconds between entity feed polls. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_DELTA_POLL_INTERVAL", - default_value_t = DeltaPolling::default().interval.as_secs(), - )] - delta_poll_interval: u64, - - /// Seconds a poll reads behind its own watermark, covering the feed's commit-visibility - /// window. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_DELTA_SAFETY_LAG", - default_value_t = DeltaPolling::default().safety_lag.as_secs(), - )] - delta_safety_lag: u64, - - /// Staging cycles an arrival's embedding read runs on each side of its ensure. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_DELTA_RETRY_POLLS", - default_value_t = DeltaPolling::default().retry_polls, - )] - delta_retry_polls: u32, - - /// Number of items that may be queued behind the watermark before backpressure is applied. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_DELTA_PLACEMENT_BACKLOG", - default_value_t = DeltaPolling::default().placement_backlog, - )] - delta_placement_backlog: usize, - - /// Serve the generation's fit-time bytes alone, starting no delta consumer. - #[arg(long, env = "HASH_GRAPH_ATLAS_NO_DELTA")] - no_delta: bool, -} - -impl From<&DeltaArgs> for DeltaPolling { - fn from(args: &DeltaArgs) -> Self { - Self { - interval: Duration::from_secs(args.delta_poll_interval), - safety_lag: Duration::from_secs(args.delta_safety_lag), - retry_polls: args.delta_retry_polls, - placement_backlog: args.delta_placement_backlog, - } - } -} - -/// Root and serving settings of one serve. -/// -/// Listener address, lifecycle, and the store connection belong to the hosting binary; these flags -/// configure what the atlas serves, not where it listens or which store it dials. -#[derive(Debug, Args)] -pub struct ServeArgs { - #[command(flatten)] - limits: LimitsArgs, - - #[command(flatten)] - delta: DeltaArgs, - - /// The server secret behind the wire row-id codec. - /// - /// Required for serving: exactly 64 hexadecimal characters (32 bytes). Generate one with - /// `openssl rand -hex 32`. The secret must not change for a generation that has ever served. - /// Rotate generations to rotate secrets. - #[arg( - long, - env = "HASH_GRAPH_ATLAS_SECRET", - hide_env_values = true, - value_parser = SecretHexBytesValueParser::::new(), - )] - secret: WireSecret, -} - -/// One serve invocation's failure, by step. -#[derive(Debug)] -pub enum ServeError { - /// Reading the current-generation pointer failed. - Current(CurrentError), - /// The root holds no activated generation. - Missing, - /// Neither the flag nor the environment supplies a wire secret. - Secret, - /// The active generation's artifacts could not open. - Open(OpenAtlasError), - /// The generation stages a projector checkpoint whose publish path did not reopen. - Placement(PlacementError), - /// Drawing the delta epoch failed. - /// - /// Entropy failure refuses the serve rather than starting a register lifetime under a - /// predictable name. The source is the system generator's own error, type-erased because its - /// defining crate is not a public dependency. - Epoch(Box), -} - -impl fmt::Display for ServeError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Current(_) => fmt.write_str("the current-generation pointer could not be read"), - Self::Missing => fmt.write_str( - "the root holds no activated generation; run `hash-graph atlas fit` first", - ), - Self::Secret => fmt.write_str( - "no wire secret is configured; set --secret or HASH_GRAPH_ATLAS_SECRET to 64 hex \ - characters (openssl rand -hex 32)", - ), - Self::Open(_) => fmt.write_str("the active generation's artifacts could not open"), - Self::Placement(_) => fmt.write_str( - "the generation promises online placement its artifacts cannot deliver; refit, or \ - disable the delta consumer with --no-delta", - ), - Self::Epoch(_) => fmt.write_str("the delta epoch could not be drawn"), - } - } -} - -impl Error for ServeError { - fn source(&self) -> Option<&(dyn Error + 'static)> { - match self { - Self::Current(error) => Some(error), - Self::Missing | Self::Secret => None, - Self::Open(error) => Some(error), - Self::Placement(error) => Some(error), - Self::Epoch(error) => Some(error.as_ref()), - } - } -} - -/// Path the liveness route answers on. -/// -/// The route sits outside the request budgets. The tracing layer skips it, since spans over a -/// probe answered every few seconds per task only inflate the metrics derived from them. -const STATUS_PATH: &str = "/status"; - -/// Everything the hosting binary supplies for one serve. -/// -/// [`ServeCommand::run`] composes the credential verifier, the service secret and the budgets -/// into the router's request middlewares. Store reads dial through `pool`, and the staging arm -/// takes its ensure client from `ensure`. The listener and the lifecycle stay the hosting -/// binary's own. -pub struct ServeOptions

{ - /// The credential verifier chain resolving each request's headers to an actor. - /// - /// A request without a recognized credential refuses before any handler. - pub provider: Arc

, - - /// The secret internal services present to delegate an actor and to pass the budgets - /// unmetered. - pub service_secret: SecretString, - - /// The request budgets, keyed per client address ahead of authentication and per principal - /// behind it. - pub rate_limit: RateLimitConfig, - - /// The store connection every read the serving process makes goes through, detail trailers - /// and permission resolution alike. - pub pool: Arc, - - /// The window over which the router reuses a resolved scope. - pub visibility: VisibilityLimits, - - /// The client half of the deduplicated embedding ensure. - /// - /// A deployment with no Temporal client configured supplies [`None`]: its arrivals stage and - /// never ensure, which fails closed. - pub workflow: Option, -} - -/// One serve invocation, resolved: the enforced limits and the wire secret over an opened root. -#[derive(Debug)] -pub struct ServeCommand { - root: GenerationRoot, - device: PinnedDevice, - limits: ServeLimits, - secret: WireSecret, - delta: DeltaArgs, -} - -impl ServeCommand { - /// Resolves the parsed flags into one serve invocation over the root. - #[must_use] - pub fn new(root: RootArgs, args: ServeArgs) -> Self { - Self { - root: root.root, - device: root.device, - limits: args.limits.into(), - secret: args.secret, - delta: args.delta, - } - } - - /// Opens the root's active generation and builds the read-API router over it. - /// - /// The router carries its own request middlewares, composed from the supplied - /// [`ServeOptions`]. A request clears the per-address limiter and credential resolution - /// before the per-principal budget meters it, and request tracing wraps the whole router. - /// The `/status` liveness route answers outside the budgets, and tracing opens no span for - /// it. - /// - /// The hosting binary owns the listener and the lifecycle, and supplies everything else - /// through [`ServeOptions`]. The router carries everything the atlas serves. - /// - /// When the generation records temporal axes and the delta opt-out is unset, the invocation - /// also spawns the delta consumer and the staging arm onto the caller's runtime. The - /// consumer polls the entity feed through the same pool for the process's lifetime, and the - /// staging arm walks classified arrivals toward placement, ensuring embeddings when the - /// caller supplies `workflow` and staging without ensures otherwise. Consumer initialization - /// draws a fresh delta epoch, which every authority token seals, so the tokens issued beside - /// an earlier register lifetime refuse uniformly and their sessions bootstrap again. - /// - /// # Errors - /// - /// Returns a [`ServeError`] naming the step that failed: [`ServeError::Current`] for reading - /// the current-generation pointer, [`ServeError::Missing`] for a root with no activated - /// generation, [`ServeError::Secret`] for an invocation with no wire secret, - /// [`ServeError::Open`] for artifacts that do not open, and [`ServeError::Epoch`] when the - /// delta epoch's entropy draw fails. - /// - /// # Panics - /// - /// This panics when called outside a Tokio runtime. The rate limiters' eviction sweep spawns - /// onto the caller's runtime, as does the delta consumer where one runs. - pub fn run

( - self, - ServeOptions { - provider, - service_secret, - rate_limit, - pool, - visibility, - workflow, - }: ServeOptions

, - ) -> Result - where - P: AuthenticationProvider + 'static, - { - // Embedders reach this entry without passing through the shell's main. - crate::math::kernel::verify_cpu_baseline(); - - let generation = self - .root - .current() - .map_err(ServeError::Current)? - .ok_or(ServeError::Missing)?; - - let options = OpenOptions { - wire_secret: self.secret, - }; - - let atlas = Atlas::open(&self.root, generation, options).map_err(ServeError::Open)?; - let atlas = Arc::new(atlas); - - tracing::info!( - root = %self.root.path(), - generation = %atlas.generation(), - "serving the active generation" - ); - - // Every serve reads the store, so detail trailers hydrate live and each caller's scope - // resolves through the same pool. - let details = Arc::new(GraphDatabaseClient::new(Arc::clone(&pool))); - tracing::info!("detail trailers hydrate from the store"); - - // The cell rides the router in every mode. Without a consumer it stays empty for the - // process's lifetime, and every ingress capture answers `None` at no cost. - let cell = Arc::new(DeltaCell::default()); - - let epoch = if self.delta.no_delta { - tracing::info!("configuration turns the delta consumer off"); - None - } else if let Some(fitted) = atlas.fitted_at() { - // The epoch draw is the first act of consumer initialization, so a failed draw - // spawns nothing and refuses the serve whole. - let epoch = DeltaEpoch::fresh(&mut SysRng) - .map_err(|error| ServeError::Epoch(Box::new(error)))?; - let polling = DeltaPolling::from(&self.delta); - - let placer = atlas - .arrival_placer(self.device.resolve()) - .map_err(ServeError::Placement)?; - let (placements_tx, placements_rx) = - tokio::sync::mpsc::channel(polling.placement_backlog); - - let consumer = DeltaConsumer::new( - Arc::clone(&pool), - Arc::clone(&atlas), - Arc::clone(&cell), - fitted, - DeltaRegister::from_atlas(&atlas), - polling, - placements_rx, - ); - let arm = StagingArm::new( - Arc::clone(&pool), - Arc::clone(&cell), - polling, - workflow, - placer, - placements_tx, - ); - - let _consumer_handle = tokio::spawn(consumer.run()); - let _staging_handle = tokio::spawn(arm.run()); - - Some(epoch) - } else { - tracing::info!("the generation records no temporal axes, so no delta consumer runs"); - None - }; - - let meter = opentelemetry::global::meter("hash-graph-atlas"); - let limiters = RateLimiters::start(&rate_limit, &meter); - - let service_secret = service_secret.into_unguarded(); - - let router = api::router(atlas, self.limits, details, pool, visibility, epoch, cell) - .route_layer( - ServiceBuilder::new() - .layer(IntoProblemLayer) - .layer(PrincipalLimitLayer { - limiters: Arc::clone(&limiters), - service_secret: Arc::clone(&service_secret), - }) - .layer(IntoProblemLayer) - .layer(AuthenticationLayer::<_, ActorId> { - provider, - service_secret: Arc::clone(&service_secret), - metrics: Arc::new(AuthenticationMetrics::new(&meter)), - // Atlas requires no bootstrap - bootstrap_route: |_path| false, - caller: core::marker::PhantomData, - }), - ) - .layer( - ServiceBuilder::new() - .layer(IntoProblemLayer) - .layer(IpGateLayer { - limiters, - service_secret, - }), - ) - .route( - STATUS_PATH, - axum::routing::get(async || axum::http::StatusCode::OK), - ) - .layer(HttpTracingLayer::new(|path| path == STATUS_PATH)); - - Ok(router) - } -} - -#[cfg(test)] -mod tests { - use clap::Args as _; - - use super::ServeArgs; - - /// A key in uppercase hex, the form most key exports and `hexdump` emit. - const UPPERCASE: &str = "6AD599A5C17E1FC4D7E2988BD4F3E0367F3C4A35D6DAE135F9A1E0EFC775CE55"; - - /// Parses `ServeArgs` from `arguments` and returns the rendered refusal. - fn refusal(arguments: &[&str]) -> String { - ServeArgs::augment_args(clap::Command::new("serve")) - .try_get_matches_from(arguments) - .expect_err("an invalid secret refuses") - .render() - .to_string() - } - - /// The rendered refusal names a position and none of the secret's characters. - #[track_caller] - fn assert_redacted(rendered: &str, secret: &str) { - assert!( - !rendered.contains(secret), - "the whole secret reached the refusal:\n{rendered}" - ); - for character in secret.chars() { - assert!( - !rendered.contains(&format!("'{character}'")), - "a character of the secret reached the refusal:\n{rendered}" - ); - } - } - - #[test] - fn secret_uppercase() { - assert_redacted(&refusal(&["serve", "--secret", UPPERCASE]), UPPERCASE); - } - - #[test] - fn secret_trailing_newline() { - let secret = format!("{}\n", UPPERCASE.to_ascii_lowercase()); - assert_redacted(&refusal(&["serve", "--secret", &secret]), &secret); - } -} diff --git a/libs/@local/graph/atlas/src/lib.rs b/libs/@local/graph/atlas/src/lib.rs index 9d0e21c9e2a..61884c8639e 100644 --- a/libs/@local/graph/atlas/src/lib.rs +++ b/libs/@local/graph/atlas/src/lib.rs @@ -23,14 +23,14 @@ //! - [`salt`] - the pipeline that runs graph construction, landmark layout, projector training, //! evaluation, and materialization. `salt::runner::operator` holds the entry points the `cli` //! commands drive, over the live store and over a dump directory. -//! - [`serve`] - the serving read surface: opened generations answering tile reads as wire bytes. +//! - `serve` - the serving read surface: opened generations answering tile reads as wire bytes. //! //! # Using the crate //! //! A caller outside this crate reaches a published generation over HTTP. [`cli`] carries the //! operator commands that fit a generation over the live store and serve the active one through the -//! [`api`] router the graph binary hosts. The Rust items behind that router are crate-internal by -//! design. [`serve::Atlas`] carries the worked example for the read path. +//! `api` router the graph binary hosts. The Rust items behind that router are crate-internal by +//! design. `serve::Atlas` carries the worked example for the read path. //! //! # Crate features //! @@ -49,10 +49,10 @@ //! //! Opening a generation validates every artifact once. //! -//! [`serve::Atlas::open`] maps and validates every serving artifact and their cross-artifact +//! `serve::Atlas::open` maps and validates every serving artifact and their cross-artifact //! agreement a single time, so every read after that is an mmap gather and a wire encode, never a //! decode. Every published artifact is a plain file mapped whole by `mmap`, so serving cost after -//! open is page-cache and address-space bound rather than parse bound. An opened [`serve::Atlas`] +//! open is page-cache and address-space bound rather than parse bound. An opened `serve::Atlas` //! is `Send + Sync` and immutable, so a caller can keep one in an `Arc` across requests for the //! process lifetime of the generation. Reads are synchronous and CPU-bound over mapped memory, so //! an async transport schedules them on a compute pool rather than inline on its own runtime @@ -62,8 +62,8 @@ //! //! Serving and fitting never combine implicitly. //! -//! [`cli::ServeCommand`] opens an already-published generation and never fits one. An empty or -//! unfitted root fails the open with a named [`cli::ServeError::Missing`] rather than fitting on +//! `cli::ServeCommand` opens an already-published generation and never fits one. An empty or +//! unfitted root fails the open with a named `cli::ServeError::Missing` rather than fitting on //! demand. //! //! ## Workspace dependencies @@ -133,7 +133,6 @@ extern crate alloc; mod allocator; -pub(crate) mod api; #[cfg(feature = "bench")] pub mod bench; pub(crate) mod bitset; @@ -151,4 +150,3 @@ pub(crate) mod progress; pub(crate) mod random; pub(crate) mod runs; pub(crate) mod salt; -pub(crate) mod serve; diff --git a/libs/@local/graph/atlas/src/salt/mod.rs b/libs/@local/graph/atlas/src/salt/mod.rs index c8543a335dd..95ef799ac5f 100644 --- a/libs/@local/graph/atlas/src/salt/mod.rs +++ b/libs/@local/graph/atlas/src/salt/mod.rs @@ -22,4 +22,3 @@ pub(crate) mod quality; pub(crate) mod relation; pub(crate) mod runner; pub(crate) mod semantic; -pub(crate) mod wire; diff --git a/libs/@local/graph/atlas/src/salt/wire/cbor.rs b/libs/@local/graph/atlas/src/salt/wire/cbor.rs deleted file mode 100644 index bc333d9b8ee..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/cbor.rs +++ /dev/null @@ -1,163 +0,0 @@ -//! The deterministic CBOR emitter. -//! -//! The wire contract restricts RFC 8949 section 4.2.1 deterministic encoding further. It admits -//! definite lengths, integer map keys, and untagged items only. The originating type fixes each -//! float's width - geometry and `HEAD` values are IEEE 754 single, locate trailer property values -//! double (WIRE 6b, store scalars are doubles) - never the deterministic-core value-dependent -//! shortest form. The writer emits exactly this subset and nothing else - tags in particular have -//! no emitter, which is the profile's proof surface staying minimal. -//! -//! The caller upholds two profile laws, which the fixtures check rather than writer state. The -//! caller emits map keys in ascending numeric order (single-byte encodings below 24 make numeric -//! order the required bytewise order), and it follows a declared map or array length with exactly -//! that many items. Every emission site in this module's consumers writes its keys as literals in -//! ascending source order. -//! -//! Heads follow RFC 8949's shortest form: the argument rides inline below 24 and in the narrowest -//! of 1, 2, 4, or 8 big-endian bytes otherwise. CBOR arguments are network byte order - the one -//! big-endian region of a little-endian wire. -#![expect( - clippy::big_endian_bytes, - reason = "CBOR arguments are network byte order (RFC 8949 section 3)" -)] - -/// An emitter appending to a borrowed byte buffer. -/// -/// Every emission writes its bytes directly into the lent buffer, so encoding a value allocates -/// only when the buffer itself grows. -#[derive(Debug)] -pub(crate) struct CborWriter<'buf> { - bytes: &'buf mut Vec, -} - -impl<'buf> CborWriter<'buf> { - /// The head of a single-precision float. - const HEAD_F32: u8 = 0xFA; - /// The head of a double-precision float. - const HEAD_F64: u8 = 0xFB; - /// Major type 4: array. - const MAJOR_ARRAY: u8 = 4; - /// Major type 2: byte string. - const MAJOR_BYTES: u8 = 2; - /// Major type 5: map. - const MAJOR_MAP: u8 = 5; - /// Major type 1: negative integer. - const MAJOR_NEGATIVE: u8 = 1; - /// Major type 3: text string. - const MAJOR_TEXT: u8 = 3; - /// Major type 0: unsigned integer. - const MAJOR_UINT: u8 = 0; - /// The simple value `false`. - const SIMPLE_FALSE: u8 = 0xF4; - /// The simple value `null`. - const SIMPLE_NULL: u8 = 0xF6; - /// The simple value `true`. - const SIMPLE_TRUE: u8 = 0xF5; - - /// Opens a writer appending to `bytes`. - #[must_use] - pub(crate) const fn over(bytes: &'buf mut Vec) -> Self { - Self { bytes } - } - - /// Emits an unsigned integer. - pub(crate) fn uint(&mut self, value: u64) { - self.head(Self::MAJOR_UINT, value); - } - - /// Emits a signed integer. - /// - /// Major type 0 for non-negative values, major type 1 otherwise, shortest form either way. - pub(crate) fn int(&mut self, value: i64) { - match u64::try_from(value) { - Ok(value) => self.head(Self::MAJOR_UINT, value), - // Major type 1 carries -1 - n; the negation cannot - // overflow because value is strictly negative. - Err(_) => self.head(Self::MAJOR_NEGATIVE, !(value.cast_unsigned())), - } - } - - /// Emits a boolean. - pub(crate) fn boolean(&mut self, value: bool) { - self.bytes.push(if value { - Self::SIMPLE_TRUE - } else { - Self::SIMPLE_FALSE - }); - } - - /// Emits `null`. - pub(crate) fn null(&mut self) { - self.bytes.push(Self::SIMPLE_NULL); - } - - /// Emits a single-precision float. - /// - /// Geometry and `HEAD` floats originate as `f32` and stay single on the wire. The emitter never - /// applies the deterministic-core shortest float form, because the originating type fixes the - /// width, not the value. - pub(crate) fn f32(&mut self, value: f32) { - self.bytes.push(Self::HEAD_F32); - self.bytes.extend_from_slice(&value.to_be_bytes()); - } - - /// Emits a double-precision float. - /// - /// Locate trailer property values originate as store doubles and stay double on the wire (WIRE - /// 6b), under the same fixed-width posture as [`CborWriter::f32`]. - pub(crate) fn f64(&mut self, value: f64) { - self.bytes.push(Self::HEAD_F64); - self.bytes.extend_from_slice(&value.to_be_bytes()); - } - - /// Emits a byte string. - pub(crate) fn bytes(&mut self, value: &[u8]) { - self.head(Self::MAJOR_BYTES, value.len() as u64); - self.bytes.extend_from_slice(value); - } - - /// Emits a text string. - pub(crate) fn text(&mut self, value: &str) { - self.head(Self::MAJOR_TEXT, value.len() as u64); - self.bytes.extend_from_slice(value.as_bytes()); - } - - /// Emits an array head that the caller follows with `length` items. - pub(crate) fn array(&mut self, length: u64) { - self.head(Self::MAJOR_ARRAY, length); - } - - /// Emits a map head; the caller emits `length` key-value pairs after it, keys ascending. - pub(crate) fn map(&mut self, length: u64) { - self.head(Self::MAJOR_MAP, length); - } - - /// Emits one head in shortest form. - #[expect( - clippy::cast_possible_truncation, - reason = "each arm's range check proves the narrowing lossless" - )] - fn head(&mut self, major: u8, argument: u64) { - let ty = major << 5; - match argument { - // Additional information 0..24: the argument is inline. - 0..0x18 => self.bytes.push(ty | argument as u8), - // 24 through 27: one, two, four, or eight argument bytes. - 0x18..=0xFF => self.bytes.extend_from_slice(&[ty | 0x18, argument as u8]), - 0x100..=0xFFFF => { - self.bytes.push(ty | 0x19); - self.bytes - .extend_from_slice(&(argument as u16).to_be_bytes()); - } - 0x1_0000..=0xFFFF_FFFF => { - self.bytes.push(ty | 0x1A); - self.bytes - .extend_from_slice(&(argument as u32).to_be_bytes()); - } - _ => { - self.bytes.push(ty | 0x1B); - self.bytes.extend_from_slice(&argument.to_be_bytes()); - } - } - } -} diff --git a/libs/@local/graph/atlas/src/salt/wire/edges.rs b/libs/@local/graph/atlas/src/salt/wire/edges.rs deleted file mode 100644 index ef832f104c0..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/edges.rs +++ /dev/null @@ -1,191 +0,0 @@ -//! The edges response: `HEAD` and three edge columns as one envelope. -//! -//! An edges document carries its delivery-order columns directly. The endpoint assembles them by -//! merging the adjacency artifact over the requested tiles' delivered rows and applying the -//! rank-ordered cap, so nothing here slices base-order arrays the way the tile document does. The -//! `SALTILEE` envelope has four slots: `HEAD`, `EDGE_SOURCES`, `EDGE_TARGETS`, `EDGE_IDS`, every -//! one always present (a tile set without visible edges delivers present-empty columns). -//! -//! An edge's wire identity is its link entity's: `EDGE_IDS` carries raw 32-byte identity records, -//! and delivery order ascends by those bytes - client-verifiable from the column alone. The -//! endpoint columns speak node row ids, and the wire has no edge-scoped id domain. -//! -//! The `HEAD` keeps an explicit `complete` flag, because cap truncation is not derivable -//! client-side. Auth-invisible edges are not truncation at all: a missing edge means authorization -//! denied it. -#![expect( - clippy::little_endian_bytes, - reason = "column integers are pinned little-endian by the wire contract" -)] - -use alloc::borrow::Cow; - -use hashql_core::id::{Id as _, IdSlice}; -use type_system::ontology::id::VersionedUrl; - -use super::{Kind, cbor::CborWriter, envelope::EnvelopeWriter, tile::encode_details}; -use crate::{ - dataset::auxiliary::Label, - integrity::Sha256Digest, - postgres::id::ArchivedEntityId, - serve::{TableIndex, WireRow, hydrate::EdgeSlot, neighbourhood::EdgeColumns}, -}; - -/// One edges response in writable form. -#[derive(Debug)] -pub(crate) struct EdgesResponse<'doc> { - /// `HEAD` key 0: the generation identity, echoing the route. - pub generation: Sha256Digest, - /// `HEAD` key 1: the variant index, echoing the route. - pub variant: u64, - /// `HEAD` key 3: `false` when the rank-ordered cap truncated the set. - pub complete: bool, - /// The delivered edges in column form: `EDGE_SOURCES`, `EDGE_TARGETS` and `EDGE_IDS`. - /// - /// `EDGE_IDS` carries `bstr(32)` records - the web uuid then the entity uuid, sixteen raw - /// bytes each, generation-frozen - and delivery order ascends by those bytes. - pub edges: &'doc EdgeColumns, - /// The hydrated detail trailer, `Some` iff the request set `detail: "auxiliary"`. - pub trailer: Option>, -} - -impl EdgesResponse<'_> { - /// Encodes the response as one `SALTILEE` envelope. - /// - /// # Panics - /// - /// This panics when trailer arrays do not cover the delivered edges. - #[must_use] - pub(crate) fn encode(&self) -> Vec { - /// Bytes per delivered edge across the three columns: source, target, identity. - const ROW_SIZE: usize = size_of::() + size_of::() + size_of::(); - /// Reserve allowance for the `HEAD` payload and the slot padding. - /// - /// `HEAD` is `map(5)` with one-byte uint keys. Its payload is a 34-byte generation echo, - /// two uints of at most nine encoded bytes, and two one-byte booleans, reaching 60 bytes at - /// the ceiling, and the four slots pad to 8-byte boundaries for at most 28 more. The - /// trailer is not counted, because its extent is store-shaped text, unknowable before - /// hydration. - const HEAD_AND_PADDING: usize = 96; - - let count = self.edges.count(); - if let Some(trailer) = &self.trailer { - trailer.debug_assert_invariants(count); - } - - let mut envelope = EnvelopeWriter::new(Kind::Edges, 4); - - envelope.reserve(HEAD_AND_PADDING + count * ROW_SIZE); - envelope.slot(|buf| self.encode_head(buf, count as u64)); - envelope.slot(|buf| write_column(buf, self.edges.sources())); - envelope.slot(|buf| write_column(buf, self.edges.targets())); - envelope.slot(|buf| write_identities(buf, self.edges.ids())); - - match &self.trailer { - Some(trailer) => envelope.finish_with_trailer(|buf| trailer.encode(buf)), - None => envelope.finish(), - } - } - - /// Encodes the `HEAD` map: keys 0 through 4. - fn encode_head(&self, buf: &mut Vec, count: u64) { - let mut cbor = CborWriter::over(buf); - cbor.map(5); - - cbor.uint(0); - cbor.bytes(&self.generation.to_bytes()); - cbor.uint(1); - cbor.uint(self.variant); - cbor.uint(2); - cbor.uint(count); - cbor.uint(3); - cbor.boolean(self.complete); - cbor.uint(4); - cbor.boolean(self.trailer.is_some()); - } -} - -/// The edges detail trailer. -/// -/// The type intern table comes first, then per-edge detail arrays in edge order. A label `null` -/// marks a link whose served display records no label text, and a representative-type `null` -/// marks one the store did not resolve. The bulk surface stays lean, with one label and one -/// representative-type reference per edge, and locate is the detail view. -#[derive(Debug)] -pub(crate) struct EdgesTrailer<'trailer> { - /// Trailer key 0: the type intern table - every referenced versioned type URL once, - /// bytewise-sorted. - pub type_table: &'trailer IdSlice, Cow<'trailer, str>>, - /// Trailer key 1: link labels, edge order. - /// - /// Labels resolve in process, captured display first: fitted and delta links alike read the - /// display the server captured at their currently served edition whenever it holds one, and - /// a fitted link with no capture reads the generation's own payload. `null` marks a link - /// whose served display records no label text. - pub link_labels: &'trailer IdSlice, - /// Trailer key 2. - /// - /// Each edge's representative type as a type-table index, edge order. - /// `null` marks a reference whose type no longer exists in the store, and an empty-shaped - /// trailer after a failed hydration reads null throughout. - pub link_type_ids: &'trailer IdSlice>>, -} - -impl EdgesTrailer<'_> { - /// Asserts that the trailer arrays cover the delivered edges. - /// - /// Coverage is the one trailer law the types do not carry, since the arrays travel as - /// separate slices. Every check is a `debug_assert`, so release builds compile this to - /// nothing. - fn debug_assert_invariants(&self, count: usize) { - debug_assert_eq!( - self.link_labels.len(), - count, - "the trailer link labels must cover exactly the delivered edges", - ); - debug_assert_eq!( - self.link_type_ids.len(), - count, - "the trailer link type ids must cover exactly the delivered edges", - ); - } - - /// Encodes the trailer tail as one self-delimiting CBOR map. - fn encode(&self, buf: &mut Vec) { - let mut cbor = CborWriter::over(buf); - cbor.map(3); - - cbor.uint(0); - cbor.array(self.type_table.len() as u64); - for url in self.type_table { - cbor.text(url); - } - - cbor.uint(1); - encode_details(&mut cbor, self.link_labels.iter()); - - cbor.uint(2); - cbor.array(self.link_type_ids.len() as u64); - for entry in self.link_type_ids { - match entry { - Some(index) => cbor.uint(index.as_u64()), - None => cbor.null(), - } - } - } -} - -/// Writes one u32 column little-endian. -pub(super) fn write_column(bytes: &mut Vec, values: &IdSlice>) { - bytes.reserve(size_of_val(values.as_raw())); - for &value in values { - bytes.extend_from_slice(&value.get().to_le_bytes()); - } -} - -/// Writes one identity column: raw 32-byte records, concatenated. -/// -/// The records are contiguous in memory, so one copy writes the whole column. -pub(super) fn write_identities(bytes: &mut Vec, ids: &IdSlice) { - bytes.extend_from_slice(zerocopy::IntoBytes::as_bytes(ids.as_raw())); -} diff --git a/libs/@local/graph/atlas/src/salt/wire/envelope.rs b/libs/@local/graph/atlas/src/salt/wire/envelope.rs deleted file mode 100644 index f58119e15bd..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/envelope.rs +++ /dev/null @@ -1,210 +0,0 @@ -//! The envelope writer, which lays out a prefix, an offset directory, and padded payloads. -//! -//! One response is a 16-byte prefix (magic with kind byte, wire version, zero flags, slot count, -//! zero reserved), a directory of `slotCount` absolute `(start, end)` byte offsets, the payloads -//! sequential in slot order and zero-padded to 8, and an optional self-delimiting CBOR trailer -//! after the last padded payload. All envelope integers are little-endian; the prefix and every -//! directory entry are `#[repr(C)]` layouts with [`zerocopy`] byteorder-typed fields, so the wire's -//! endianness is part of the type. -//! -//! The directory is the locating mechanism: `(0, 0)` marks an absent slot, `start == end` at a -//! nonzero offset marks a present-but-empty payload, and every present `start` is 8-aligned by -//! construction (the payload region begins at the 8-aligned `16 + 8 * slotCount` and each payload -//! pads to the next 8 boundary). Emitting the right mark carries request semantics, because a -//! zero-point tile's columns are present-empty while an unrequested section is absent. -//! [`slot`](EnvelopeWriter::slot) and [`absent`](EnvelopeWriter::absent) are distinct calls, so -//! every call site keeps that distinction explicit. -//! -//! The writer assembles a response in one buffer. The closure handed to -//! [`slot`](EnvelopeWriter::slot) writes a slot's payload directly into that buffer, and the writer -//! backfills the directory entry as the closure returns, so every payload reaches the response by -//! exactly one write. -//! -//! Offsets are `u32`: directory-addressed payloads end below 4 GiB, the format's representability -//! boundary. The writer enforces it with checked conversions - a payload crossing it is a producer -//! panic (caught and answered as a 500), never a truncated or wrapped offset. The trailer sits -//! outside the directory and shares no such ceiling. - -use zerocopy::{IntoBytes as _, LE, U16, U32}; - -use super::{Kind, WIRE_VERSION}; - -/// Prefix size in bytes, pinned to the layout it measures. -const PREFIX: usize = size_of::(); -/// Directory entry size in bytes, pinned to the layout it measures. -const ENTRY: usize = size_of::(); - -/// Rounds `length` up to the next multiple of 8. -const fn align8(length: usize) -> usize { - length.next_multiple_of(8) -} - -/// The 16-byte envelope prefix. -#[derive(zerocopy::IntoBytes, zerocopy::Immutable)] -#[repr(C)] -struct Prefix { - /// The seven-byte family magic and the kind byte. - kind: Kind, - /// The wire version shared by the whole family. - version: U16, - /// Reserved zero. - flags: U16, - /// The directory's entry count. - slots: U16, - /// Reserved zero. - reserved: U16, -} - -/// One directory entry of absolute byte offsets, with `end` exclusive and unpadded. -#[derive(zerocopy::IntoBytes, zerocopy::Immutable)] -#[repr(C)] -struct Entry { - /// The payload's first byte, 8-aligned when present. - start: U32, - /// One past the payload's last byte. - end: U32, -} - -/// A single-buffer writer assembling one response. -/// -/// The caller records slots in slot order, one call per slot. Each payload goes directly into the -/// envelope's buffer, and the writer backfills the directory entry into the reserved region as each -/// slot closes, so the buffer is complete the moment the last slot closes. [`finish`](Self::finish) -/// checks the slot count and returns the bytes, and -/// [`finish_with_trailer`](Self::finish_with_trailer) writes the trailer tail first. -#[derive(Debug)] -pub(crate) struct EnvelopeWriter { - bytes: Vec, - slots: u16, - recorded: u16, -} - -impl EnvelopeWriter { - /// Opens an envelope of `kind` with `slots` directory entries. - /// - /// `slots` is at least the kind's v1 table size at every call site; appended slots beyond the - /// table are legal by the evolution rule. - #[must_use] - pub(crate) fn new(kind: Kind, slots: u16) -> Self { - let prefix = Prefix { - kind, - version: U16::new(WIRE_VERSION), - flags: U16::ZERO, - slots: U16::new(slots), - reserved: U16::ZERO, - }; - - let mut bytes = Vec::new(); - bytes.extend_from_slice(prefix.as_bytes()); - bytes.resize(PREFIX + ENTRY * slots as usize, 0); - - Self { - bytes, - slots, - recorded: 0, - } - } - - /// Reserves room for `additional` payload bytes beyond the buffer's current length. - /// - /// An allocation hint only, since the buffer grows as needed regardless. - pub(crate) fn reserve(&mut self, additional: usize) { - self.bytes.reserve(additional); - } - - /// Records the next slot as present, writing its payload directly into the buffer. - /// - /// The closure appends the payload bytes. The slot's extent is whatever it appended. The writer - /// zero-pads the payload to the next 8 boundary after the closure returns. - /// - /// # Panics - /// - /// This panics when every declared slot is already recorded, and when the closure shrinks the - /// buffer. - pub(crate) fn slot(&mut self, write: impl FnOnce(&mut Vec)) { - assert!( - self.recorded < self.slots, - "the envelope declares {} slots, all recorded", - self.slots, - ); - - let start = self.bytes.len(); - write(&mut self.bytes); - let end = self.bytes.len(); - assert!(end >= start, "a slot writer must only append"); - - self.bytes.resize(align8(end), 0); - - self.record(start, end); - } - - /// Records the next slot as absent, leaving its directory entry `(0, 0)` and adding zero - /// payload bytes. - /// - /// # Panics - /// - /// This panics when every declared slot is already recorded, and on slot 0, because the `HEAD` - /// is always present. - pub(crate) fn absent(&mut self) { - assert!( - self.recorded < self.slots, - "the envelope declares {} slots, all recorded", - self.slots, - ); - assert!(self.recorded > 0, "slot 0 (HEAD) is always present"); - - self.recorded += 1; - } - - /// Closes the envelope and returns the response bytes. - /// - /// # Panics - /// - /// This panics when the caller recorded fewer slots than the envelope declares. - #[must_use] - pub(crate) fn finish(self) -> Vec { - assert_eq!( - self.recorded, self.slots, - "the envelope declares {} slots", - self.slots, - ); - - self.bytes - } - - /// Closes the envelope, writing the CBOR trailer tail into the buffer first. - /// - /// The trailer starts at the 8-aligned end of the last present payload - where the buffer - /// already stands - and is never padded: its extent is its own CBOR structure. - /// - /// # Panics - /// - /// This panics when the caller recorded fewer slots than the envelope declares. - #[must_use] - pub(crate) fn finish_with_trailer(mut self, write: impl FnOnce(&mut Vec)) -> Vec { - assert_eq!( - self.recorded, self.slots, - "the envelope declares {} slots", - self.slots, - ); - - write(&mut self.bytes); - self.bytes - } - - /// Backfills the directory entry for the most recently written slot. - fn record(&mut self, start: usize, end: usize) { - let entry = Entry { - start: U32::new( - u32::try_from(start).expect("directory offsets fit u32: payloads end below 4 GiB"), - ), - end: U32::new( - u32::try_from(end).expect("directory offsets fit u32: payloads end below 4 GiB"), - ), - }; - - let at = PREFIX + ENTRY * self.recorded as usize; - self.bytes[at..at + ENTRY].copy_from_slice(zerocopy::IntoBytes::as_bytes(&entry)); - self.recorded += 1; - } -} diff --git a/libs/@local/graph/atlas/src/salt/wire/fixtures.rs b/libs/@local/graph/atlas/src/salt/wire/fixtures.rs deleted file mode 100644 index 78cfe6d9bfa..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/fixtures.rs +++ /dev/null @@ -1,1248 +0,0 @@ -//! The cross-language fixture corpus: `fixtures/wire/`. -//! -//! Each fixture is one encoded response checked in as fixture bytes plus a JSON sidecar of the -//! expected decoded values (floats as u32 bit patterns), the envelope prefix, and the directory - -//! so the padding sweep is assertable client-side from the sidecar alone. The Rust side proves the -//! encoder reproduces the pinned bytes; the TypeScript decoder consumes the same files and asserts -//! field-for-field equality - "matches the Rust side" is never asserted by eye. The decoder derives -//! its request echo context from the sidecar `HEAD`; a request field the `HEAD` does not echo joins -//! the sidecar the day a fixture needs one. -//! -//! Pinning a fixture waits on a ratified schema and a serving endpoint, because pinning bytes -//! before the schema exists pins an invention. The end-to-end fixture over a real published -//! generation is separate, and it covers a whole artifact tree instead of one envelope. Every -//! fixture here uses `spanLog2 = 2`, so the cut rule reads `bucket = z + 2` and the root spans -//! buckets `0..=2`. -//! -//! Regenerate with `ATLAS_WIRE_BLESS=1` in the environment; the default run compares response bytes -//! byte-for-byte and sidecars by parsed value - the sidecar contract is structural (decoders parse -//! it, never byte-compare it), so repository JSON formatting passes over the checked-in fixtures -//! are not drift. -#![expect( - clippy::little_endian_bytes, - reason = "the fixtures write the contract's little-endian wire integers" -)] -#![expect( - clippy::single_range_in_vec_init, - reason = "a delta tile's delivered set really is one contiguous range" -)] - -use alloc::borrow::Cow; -use std::fs; - -use camino::Utf8PathBuf; -use hashql_core::id::{Id, IdSlice}; -use serde_json::{Value, json}; -use type_system::ontology::id::VersionedUrl; - -use super::{ - Kind, Mode, - edges::{EdgesResponse, EdgesTrailer}, - envelope::EnvelopeWriter, - locate::{LocateResponse, LocateTrailer, PropertyMap, PropertyValue}, - tests::{directory, section}, - tile::{DeliveredSet, GlobalHead, TileCoordinate, TileHead, TileResponse, TileTrailer}, -}; -use crate::{ - bitset::DenseBitSlice, - dataset::auxiliary::{Icon, Label}, - identity::{BasePosition, NodeRowId}, - integrity::Sha256Digest, - math::{Bounds2, Vec2}, - postgres::id::ArchivedEntityId, - salt::postings::artifact::Membership, - serve::{ - TableIndex, WireRow, hydrate::EdgeSlot, neighbourhood::EdgeColumns, schedule::ViewRow, - }, -}; - -/// One pinned fixture with its name, response bytes, and sidecar. -struct Fixture { - name: &'static str, - bytes: Vec, - sidecar: Value, -} - -#[test] -fn corpus_matches_the_checked_in_fixtures() { - let dir = Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("fixtures/wire"); - let bless = std::env::var_os("ATLAS_WIRE_BLESS").is_some(); - - for fixture in corpus() { - let bytes_path = dir.join(format!("{}.saltile", fixture.name)); - let sidecar_path = dir.join(format!("{}.json", fixture.name)); - let sidecar = format!( - "{}\n", - serde_json::to_string_pretty(&fixture.sidecar).expect("sidecars are plain JSON"), - ); - - if bless { - fs::write(&bytes_path, &fixture.bytes).expect("the fixture directory is writable"); - fs::write(&sidecar_path, &sidecar).expect("the fixture directory is writable"); - continue; - } - - let pinned = fs::read(&bytes_path).unwrap_or_else(|_| { - panic!("{bytes_path} is missing; regenerate with ATLAS_WIRE_BLESS=1") - }); - assert_eq!( - fixture.bytes, pinned, - "{} bytes drifted from the pinned fixture", - fixture.name, - ); - - let pinned = fs::read_to_string(&sidecar_path).unwrap_or_else(|_| { - panic!("{sidecar_path} is missing; regenerate with ATLAS_WIRE_BLESS=1") - }); - let pinned: Value = serde_json::from_str(&pinned).unwrap_or_else(|error| { - panic!( - "{sidecar_path} does not parse as JSON ({error}); regenerate with \ - ATLAS_WIRE_BLESS=1" - ) - }); - assert_eq!( - fixture.sidecar, pinned, - "{} sidecar drifted from the pinned fixture", - fixture.name, - ); - } -} - -/// G9/G10 exist to sweep the padding widths. -/// -/// Every width 1 through 7 must occur across the pair, counting the tail padding of the last -/// present slot. -#[test] -fn padding_sweep_covers_every_width() { - let mut widths = [false; 8]; - for fixture in [g9_padding_low(), g10_padding_high()] { - let slots = u16::from_le_bytes( - fixture.bytes[12..14] - .try_into() - .expect("the prefix carries the slot count"), - ); - for slot in 0..slots as usize { - let (start, end) = directory(&fixture.bytes, slot); - if (start, end) != (0, 0) { - widths[(end as usize).next_multiple_of(8) - end as usize] = true; - } - } - } - - assert_eq!( - widths[1..], - [true; 7], - "G9/G10 must produce every padding width 1..=7", - ); -} - -/// The whole corpus in fixture order. -fn corpus() -> Vec { - vec![ - g1_minimal_tile(), - g2_root_tile(), - g3_total_tile(), - g4_empty_root(), - g5_trailer_tile(), - g6_edges(), - g7_locate(), - g8_appended_slot(), - g9_padding_low(), - g10_padding_high(), - ] -} - -/// Renders a tile fixture's sidecar. -/// -/// Prefix, directory, decoded `HEAD`, gathered columns, and trailer. -/// -/// `colored` is the request's coloredTypeIds count; `type_mask` is the hand-derived expected column -/// (present iff `colored > 0`). `mass` and `appended` describe populated beyond-v1 slots (G8/G10), -/// which a v1 decoder ignores by contract. -fn tile_sidecar( - name: &str, - response: &TileResponse<'_>, - bytes: &[u8], - colored: u64, - type_mask: Option<&[u8]>, - mass: Option<&[u32]>, - appended: &Value, -) -> Value { - assert_eq!( - type_mask.is_some(), - colored > 0, - "TYPE_MASK rides exactly the requests that color types", - ); - let head = &response.head; - - let mut positions_bits = Vec::new(); - let mut row_ids = Vec::new(); - for row in response.delivered { - let ViewRow::Base(position) = row else { - unreachable!("the wire fixtures deliver base rows alone") - }; - let point = response.positions[position]; - positions_bits.push(point.x().to_bits()); - positions_bits.push(point.y().to_bits()); - row_ids.push(response.rows[position]); - } - - let global = head.global.as_ref().map_or(Value::Null, |global| { - let bounds_bits = global.bounds.as_ref().map_or(Value::Null, |bounds| { - json!([ - bounds.min().x().to_bits(), - bounds.min().y().to_bits(), - bounds.max().x().to_bits(), - bounds.max().y().to_bits(), - ]) - }); - json!({ - "visibleAtZoom": global.visible, - "boundsBits": bounds_bits, - "minResolution": global.min_resolution, - }) - }); - - let trailer = response.trailer.as_ref().map_or(Value::Null, |trailer| { - json!({ - "labels": details_sidecar(trailer.labels), - "icons": details_sidecar(trailer.icons), - }) - }); - - let head_json = json!({ - "generation": head.generation.to_string(), - "variant": head.variant, - "coordinate": [head.coordinate.z, head.coordinate.x, head.coordinate.y], - "mode": head.mode.code(), - "delivered": row_ids.len(), - "firstBucket": head.first_bucket, - "runs": head.runs, - "global": global, - "children": head.children, - "trailer": response.trailer.is_some(), - }); - - json!({ - "golden": name, - "layer": "tile", - "prefix": prefix_sidecar(bytes), - "directory": directory_sidecar(bytes), - "head": head_json, - "positions": positions_bits, - "rowIds": row_ids, - "typeMask": type_mask.map_or(Value::Null, |mask| json!(mask)), - "mass": mass.map_or(Value::Null, |values| json!(values)), - "appended": appended, - "trailer": trailer, - }) -} - -/// Renders the envelope prefix fields. -fn prefix_sidecar(bytes: &[u8]) -> Value { - assert!( - bytes.len() >= 16, - "every envelope opens with a 16-byte prefix" - ); - let magic = core::str::from_utf8(&bytes[0..8]).expect("the magic is ASCII"); - json!({ - "magic": magic, - "wireVersion": u16::from_le_bytes(bytes[8..10].try_into().expect("two bytes")), - "flags": u16::from_le_bytes(bytes[10..12].try_into().expect("two bytes")), - "slotCount": u16::from_le_bytes(bytes[12..14].try_into().expect("two bytes")), - "reserved": u16::from_le_bytes(bytes[14..16].try_into().expect("two bytes")), - }) -} - -/// Renders the directory as `[[start, end], ...]`. -fn directory_sidecar(bytes: &[u8]) -> Value { - let slots = u16::from_le_bytes(bytes[12..14].try_into().expect("two bytes")); - Value::Array( - (0..slots as usize) - .map(|slot| { - let (start, end) = directory(bytes, slot); - json!([start, end]) - }) - .collect(), - ) -} - -/// Renders a detail array: strings and nulls. -fn details_sidecar + ?Sized>(entries: &[&T]) -> Value { - Value::Array( - entries - .iter() - .map(|entry| { - Option::Some(entry.as_ref()) - .filter(|entry| !entry.is_empty()) - .map_or(Value::Null, |text| json!(text)) - }) - .collect(), - ) -} - -/// G1. -/// -/// A non-root delta tile - one run, three points, all columns, `TYPE_MASK` over three requested -/// types (stride 1, `n % 8 != 0`), a multi-bit point carrying types 0 and 2, an all-zero point, a -/// single children bit. -fn g1_minimal_tile() -> Fixture { - let positions: Vec = (0_u16..12) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(0.125, -0.5), - f32::from(index).mul_add(-0.125, 0.75), - ) - }) - .collect(); - let rows: Vec> = (0..12) - .map(|index| WireRow::pinned(3 * index + 7)) - .collect(); - let ranges = [8_u32..11] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - // Type 0 delivers base positions 8 and 9, while 3 and 11 lie outside the run. Type 1 uses a - // dense representation whose single bit (bit 2) lies outside the run, so it delivers nothing. - // Type 2 delivers position 9, which makes point 9 the multi-bit point (types 0 and 2). Point 10 - // matches nothing. - let t0 = [3_u32, 8, 9, 11].map(BasePosition::from_u32); - let t1_dense = dense_set(12, &[2]); - let t2 = [9_u32].map(BasePosition::from_u32); - let masks = [ - Membership::List(&t0), - Membership::Dense(&t1_dense), - Membership::List(&t2), - ]; - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x11; 32]), - variant: 0, - coordinate: TileCoordinate { z: 2, x: 3, y: 1 }, - mode: Mode::Delta, - first_bucket: 4, - runs: &[3], - global: None, - children: 0b0100, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: Some(&masks), - trailer: None, - }; - let bytes = response.encode(); - - let expected_mask = [0b001_u8, 0b101, 0b000]; - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - expected_mask, - "the hand-derived G1 masks must match the encoder", - ); - - let sidecar = tile_sidecar( - "g1-minimal-tile", - &response, - &bytes, - 3, - Some(&expected_mask), - None, - &Value::Null, - ); - Fixture { - name: "g1-minimal-tile", - bytes, - sidecar, - } -} - -/// G2. -/// -/// The delta root - buckets `0..=2` with the middle run zero-length, one contiguous multi-segment -/// range, no coloredTypeIds (`TYPE_MASK` absent), the required global map with bounds present, all -/// four children. -fn g2_root_tile() -> Fixture { - let positions: Vec = (0_u16..4) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(0.25, -0.875), - f32::from(index).mul_add(0.125, -0.5), - ) - }) - .collect(); - let rows: Vec> = (0..4) - .map(|index| WireRow::pinned(90 - 10 * index)) - .collect(); - // The root's runs are bucket fencepost differences; its delivered - // set is one contiguous range. - let ranges = [0_u32..3] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x22; 32]), - variant: 0, - coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, - mode: Mode::Delta, - first_bucket: 0, - runs: &[1, 0, 2], - global: Some(GlobalHead { - visible: 3, - bounds: Bounds2::new(Vec2::new(-0.875, -0.5), Vec2::new(0.9375, 0.5)), - min_resolution: 5, - }), - children: 0b1111, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - }; - let bytes = response.encode(); - - let sidecar = tile_sidecar( - "g2-root-tile", - &response, - &bytes, - 0, - None, - None, - &Value::Null, - ); - Fixture { - name: "g2-root-tile", - bytes, - sidecar, - } -} - -/// G3. -/// -/// A total tile - four runs from bucket 0 with a zero-length run interspersed, bucket-major -/// concatenation, nine requested types (two-byte mask stride), zero children. -fn g3_total_tile() -> Fixture { - let positions: Vec = (0_u16..10) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(-0.1875, 0.9), - f32::from(index) * 0.0625, - ) - }) - .collect(); - let rows: Vec> = (0..10) - .map(|index| WireRow::pinned(1000 + 7 * index)) - .collect(); - // Buckets 0..=3: one contiguous base slice each, the second - // empty. Base positions 2, 3, and 7 belong to no run and never - // deliver. - let ranges = [0_u32..1, 1..1, 4..7, 8..10] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - // Stride 2 over nine types, with type 8's bit in the second byte. - let t0 = [0_u32, 5, 9].map(BasePosition::from_u32); - let t1_dense = dense_set(10, &[3, 5]); - let t3 = [4_u32].map(BasePosition::from_u32); - let t7 = [6_u32, 8].map(BasePosition::from_u32); - let t8 = [9_u32].map(BasePosition::from_u32); - let empty: [BasePosition; 0] = []; - let masks = [ - Membership::List(&t0), - Membership::Dense(&t1_dense), - Membership::List(&empty), - Membership::List(&t3), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&t7), - Membership::List(&t8), - ]; - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x33; 32]), - variant: 0, - coordinate: TileCoordinate { z: 1, x: 1, y: 0 }, - mode: Mode::Total, - first_bucket: 0, - runs: &[1, 0, 3, 2], - global: None, - children: 0, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: Some(&masks), - trailer: None, - }; - let bytes = response.encode(); - - // Delivered order 0, 4, 5, 6, 8, 9: - // t0 | t3 | t0+t1 | t7 | t7 | t0+t8. - let expected_mask = [ - 0x01_u8, 0x00, 0x08, 0x00, 0x03, 0x00, 0x80, 0x00, 0x80, 0x00, 0x01, 0x01, - ]; - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - expected_mask, - "the hand-derived G3 masks must match the encoder", - ); - - let sidecar = tile_sidecar( - "g3-total-tile", - &response, - &bytes, - 9, - Some(&expected_mask), - None, - &Value::Null, - ); - Fixture { - name: "g3-total-tile", - bytes, - sidecar, - } -} - -/// G4. -/// -/// The empty root - zero delivered, present-empty columns at one shared offset, `TYPE_MASK` absent, -/// zero children, and the required global map with `visibleAtZoom = 0` and bounds absent - the -/// bounds-absent-iff-empty rule, pinned. -fn g4_empty_root() -> Fixture { - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x44; 32]), - variant: 0, - coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, - mode: Mode::Delta, - first_bucket: 0, - runs: &[0, 0, 0], - global: Some(GlobalHead { - visible: 0, - bounds: None, - min_resolution: 0, - }), - children: 0, - }, - delivered: DeliveredSet::Ranges(&[]), - positions: IdSlice::from_raw(&[]), - rows: IdSlice::from_raw(&[]), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - }; - let bytes = response.encode(); - - let sidecar = tile_sidecar( - "g4-empty-root", - &response, - &bytes, - 0, - None, - None, - &Value::Null, - ); - Fixture { - name: "g4-empty-root", - bytes, - sidecar, - } -} - -/// G5. -/// -/// A delta trailer tile - labels and icons with null entries and non-ASCII UTF-8 (multi-byte -/// sequences and a combining mark), children bits 0 and 2. -fn g5_trailer_tile() -> Fixture { - let positions: Vec = (0_u16..8) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(0.125, -0.25), - f32::from(index).mul_add(-0.25, 0.5), - ) - }) - .collect(); - let rows: Vec> = (0..8) - .map(|index| WireRow::pinned(11 * index + 5)) - .collect(); - let ranges = [2_u32..6] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let labels = [ - Label::new("Z\u{fc}rich"), - Label::EMPTY, - Label::new("e\u{301}"), - Label::new("\u{1f980}"), - ]; - let icons = [ - Icon::empty(), - Icon::new("\u{6c34}\u{6238}"), - Icon::new("\u{2192}"), - Icon::empty(), - ]; - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x55; 32]), - variant: 0, - coordinate: TileCoordinate { z: 1, x: 1, y: 0 }, - mode: Mode::Delta, - first_bucket: 3, - runs: &[4], - global: None, - children: 0b0101, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: Some(TileTrailer { - labels: &labels, - icons: &icons, - }), - }; - let bytes = response.encode(); - - let sidecar = tile_sidecar( - "g5-trailer-tile", - &response, - &bytes, - 0, - None, - None, - &Value::Null, - ); - Fixture { - name: "g5-trailer-tile", - bytes, - sidecar, - } -} - -/// G6: an edges response - three columns. -/// -/// `complete = false` (the cap flag is the point) and `EDGE_IDS` as raw identity records ascending -/// per the delivery-order pin, so the fixture matches the ratified contract rather than structure -/// alone. The interned detail trailer holds the type table plus labels with nulls and -/// representative-type references including a store-absent `null`. -fn g6_edges() -> Fixture { - let link_labels = [ - Label::new("\u{153}uvre"), - Label::new("created by"), - Label::EMPTY, - ]; - let type_table = ["https://t.test/authored/v/1", "https://t.test/cites/v/2"].map(Cow::Borrowed); - let link_type_ids = [Some(TableIndex::new(1)), Some(TableIndex::new(0)), None]; - let edges = EdgeColumns::pinned([ - (4, 11, identity_of(0xA0, 0xA1)), - (4, 7, identity_of(0xB0, 0xB1)), - (9, 2, identity_of(0xC0, 0xC1)), - ]); - - let response = EdgesResponse { - generation: Sha256Digest::from_bytes_unchecked([0x66; 32]), - variant: 0, - complete: false, - edges: &edges, - trailer: Some(EdgesTrailer { - type_table: IdSlice::from_raw(&type_table), - link_labels: IdSlice::from_raw(&link_labels), - link_type_ids: IdSlice::from_raw(&link_type_ids), - }), - }; - let bytes = response.encode(); - - let sidecar = json!({ - "golden": "g6-edges", - "layer": "edges", - "prefix": prefix_sidecar(&bytes), - "directory": directory_sidecar(&bytes), - "head": { - "generation": response.generation.to_string(), - "variant": response.variant, - "count": 3, - "complete": false, - "trailer": true, - }, - "sources": response.edges.sources().as_raw(), - "targets": response.edges.targets().as_raw(), - "edgeIds": identities_sidecar(response.edges.ids().as_raw()), - "trailer": { - "typeTable": type_table, - "linkLabels": details_sidecar(&link_labels), - "linkTypeIds": link_type_ids.map(|entry| entry.map(Id::as_u32)), - }, - }); - - Fixture { - name: "g6-edges", - bytes, - sidecar, - } -} - -/// G7's detail-trailer pins. -struct G7Trailer { - /// The type intern table, bytewise-sorted. - type_table: [Cow<'static, str>; 3], - /// The property intern table, bytewise-sorted. - property_table: [Cow<'static, str>; 4], - /// The source's property map, covering text, a negative integer, a double, and a boolean. - source_map: PropertyMap<'static>, - /// One edge's property map, covering a positive integer, an explicit `null`, and a negative - /// double. - link_map: PropertyMap<'static>, - /// Node labels: non-ASCII, an empty label (a wire `null`), an astral plane character, a - /// combining mark. - labels: [&'static Label; 4], - /// Each node's representative type as a type-table index, one store-absent `null` among them. - type_ids: [Option>; 4], - /// Link labels: non-ASCII, an empty label (a wire `null`), ASCII. - link_labels: [&'static Label; 3], - /// Each edge's direct types as type-table indexes, one list empty. - /// - /// Canonical direct-type order is the store's, never sorted: the first list pins a descending - /// pair on purpose. - link_type_ids: [Vec>; 3], - /// The delivered edges whose type list is complete. - link_type_ids_complete: Box>, - /// The delivered edges whose property map is complete. - link_properties_complete: Box>, -} - -/// Builds G7's detail-trailer pins. -fn g7_trailer() -> G7Trailer { - G7Trailer { - type_table: [ - "https://t.test/authored/v/1", - "https://t.test/person/v/3", - "https://t.test/work/v/2", - ] - .map(Cow::Borrowed), - property_table: [ - "https://x.test/age/", - "https://x.test/name/", - "https://x.test/ok/", - "https://x.test/score/", - ] - .map(Cow::Borrowed), - source_map: PropertyMap::new_unchecked(vec![ - (TableIndex::new(0), PropertyValue::Integer(-3)), - (TableIndex::new(1), PropertyValue::Text("Ada")), - (TableIndex::new(2), PropertyValue::Boolean(true)), - (TableIndex::new(3), PropertyValue::Float(0.5)), - ]), - link_map: PropertyMap::new_unchecked(vec![ - (TableIndex::new(0), PropertyValue::Integer(977)), - (TableIndex::new(1), PropertyValue::Null), - (TableIndex::new(3), PropertyValue::Float(-2.5)), - ]), - labels: [ - Label::new("Caf\u{e9}"), - Label::EMPTY, - Label::new("\u{1d50a}"), - Label::new("e\u{301}"), - ], - type_ids: [ - Some(TableIndex::new(1)), - None, - Some(TableIndex::new(2)), - Some(TableIndex::new(0)), - ], - link_labels: [Label::new("\u{153}uvre"), Label::EMPTY, Label::new("cites")], - link_type_ids: [ - vec![TableIndex::new(1), TableIndex::new(0)], - vec![TableIndex::new(0)], - Vec::new(), - ], - // Slot 0's type list is complete, and slots 0 and 2 carry whole property maps, over the - // three delivered edges. - link_type_ids_complete: dense_set(3, &[0]), - link_properties_complete: dense_set(3, &[0, 2]), - } -} - -/// G7. -/// -/// A locate response - the source first over an arbitrary delivered list (nothing contiguous), -/// `TYPE_MASK` probed per point, `complete = false` (the locate edge cap flag is the point), and -/// both source completeness flags exercised in opposite states. -/// -/// The full detail trailer holds both intern tables, labels with nulls and non-ASCII, node -/// representative-type references including a `null`, the property maps covering every scalar -/// value shape between them (text, positive and negative integers, doubles, booleans, explicit -/// null), link type lists in non-ascending canonical order with an empty entry, link property -/// maps covering map, `null`, and empty shapes, and both completeness bitmasks. The trailer's -/// own pins live in [`G7Trailer`]. -fn g7_locate() -> Fixture { - let positions: Vec = (0_u16..8) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(0.25, -0.875), - f32::from(index).mul_add(-0.125, 0.5), - ) - }) - .collect(); - let rows: Vec> = (0..8) - .map(|index| WireRow::pinned(10 * index + 1)) - .collect(); - // Source at base position 6, then partners ascending by wire row id per the delivery pin. Row - // 61 comes first, then 11 < 21 < 41, so the fixture matches the ratified order, not the - // envelope structure alone. - let delivered = - [6_u32, 1, 2, 4].map(|position| ViewRow::Base(BasePosition::from_u32(position))); - - // type 0: list members at 1 and 6; type 1: dense bit at 4 over - // N = 8. Delivered order 6, 1, 2, 4: - // t0 | t0 | none | t1. - let t0 = [1_u32, 6].map(BasePosition::from_u32); - let t1_dense = dense_set(8, &[4]); - let masks = [Membership::List(&t0), Membership::Dense(&t1_dense)]; - - let trailer = g7_trailer(); - let empty_map = PropertyMap::new_unchecked(Vec::new()); - let link_properties: [Option<&PropertyMap<'_>>; 3] = - [Some(&trailer.link_map), None, Some(&empty_map)]; - let edges = EdgeColumns::pinned([ - (61, 11, identity_of(0xD0, 0xD1)), - (41, 61, identity_of(0xE0, 0xE1)), - (21, 41, identity_of(0xF0, 0xF1)), - ]); - - let response = LocateResponse { - generation: Sha256Digest::from_bytes_unchecked([0x77; 32]), - variant: 0, - cell: TileCoordinate { z: 3, x: 5, y: 2 }, - complete: false, - entity_id: identity_of(0x42, 0x24), - type_ids_complete: true, - properties_complete: false, - delivered: IdSlice::from_raw(&delivered), - arrivals: IdSlice::from_raw(&[]), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - masks: Some(&masks), - edges: &edges, - trailer: LocateTrailer { - type_table: IdSlice::from_raw(&trailer.type_table), - property_table: IdSlice::from_raw(&trailer.property_table), - labels: IdSlice::from_raw(&trailer.labels), - type_ids: IdSlice::from_raw(&trailer.type_ids), - properties: Some(&trailer.source_map), - link_labels: IdSlice::from_raw(&trailer.link_labels), - link_type_ids: IdSlice::from_raw(&trailer.link_type_ids), - link_type_ids_complete: &trailer.link_type_ids_complete, - link_properties: IdSlice::from_raw(&link_properties), - link_properties_complete: &trailer.link_properties_complete, - }, - }; - let bytes = response.encode(); - - let expected_mask = [0b01_u8, 0b01, 0b00, 0b10]; - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - expected_mask, - "the hand-derived G7 masks must match the encoder", - ); - - let sidecar = locate_sidecar( - "g7-locate", - &response, - &bytes, - &expected_mask, - &g7_properties_sidecar(), - &g7_link_properties_sidecar(), - ); - Fixture { - name: "g7-locate", - bytes, - sidecar, - } -} - -/// Renders G7's source property map. -/// -/// Property values ride the sidecar as plain JSON: every pinned double is exactly representable, -/// and none renders integral, so the number forms stay unambiguous (`serde_json` round-trips -/// `f64`). -fn g7_properties_sidecar() -> Value { - json!({ - "https://x.test/age/": -3, - "https://x.test/name/": "Ada", - "https://x.test/ok/": true, - "https://x.test/score/": 0.5, - }) -} - -/// Renders G7's per-edge link property maps: a populated map, `null`, and an empty map. -fn g7_link_properties_sidecar() -> Value { - json!([ - { - "https://x.test/age/": 977, - "https://x.test/name/": Value::Null, - "https://x.test/score/": -2.5, - }, - Value::Null, - {}, - ]) -} - -/// Renders a locate fixture's sidecar. -/// -/// Prefix, directory, decoded `HEAD`, gathered columns, and the detail trailer. The property maps -/// arrive pre-rendered because the fixture itself pins their JSON forms alongside the wire values. -/// The completeness bitmasks render as their logical boolean lists. -fn locate_sidecar( - name: &str, - response: &LocateResponse<'_>, - bytes: &[u8], - type_mask: &[u8], - properties: &Value, - link_properties: &Value, -) -> Value { - let mut positions_bits = Vec::new(); - let mut row_ids = Vec::new(); - for &vessel in response.delivered { - let (point, wire) = match vessel { - ViewRow::Base(position) => (response.positions[position], response.rows[position]), - ViewRow::Arrival(index) => { - let arrival = &response.arrivals[index]; - (arrival.position, arrival.wire) - } - }; - positions_bits.push(point.x().to_bits()); - positions_bits.push(point.y().to_bits()); - row_ids.push(wire); - } - - let trailer = &response.trailer; - json!({ - "golden": name, - "layer": "locate", - "prefix": prefix_sidecar(bytes), - "directory": directory_sidecar(bytes), - "head": { - "generation": response.generation.to_string(), - "variant": response.variant, - "count": response.delivered.len(), - "zoom": response.cell.z, - "cell": [response.cell.z, response.cell.x, response.cell.y], - "edges": response.edges.count(), - "complete": response.complete, - "entityId": identities_sidecar(&[response.entity_id]).remove(0), - "typeIdsComplete": response.type_ids_complete, - "propertiesComplete": response.properties_complete, - }, - "positions": positions_bits, - "rowIds": row_ids, - "typeMask": type_mask, - "sources": response.edges.sources().as_raw(), - "targets": response.edges.targets().as_raw(), - "edgeIds": identities_sidecar(response.edges.ids().as_raw()), - "trailer": { - "typeTable": trailer.type_table.as_raw(), - "propertyTable": trailer.property_table.as_raw(), - "labels": details_sidecar(trailer.labels.as_raw()), - "typeIds": trailer - .type_ids - .iter() - .map(|entry| entry.map(Id::as_u32)) - .collect::>(), - "properties": properties, - "linkLabels": details_sidecar(trailer.link_labels.as_raw()), - "linkTypeIds": trailer - .link_type_ids - .iter() - .map(|list| list.iter().copied().map(Id::as_u32).collect::>()) - .collect::>(), - "linkTypeIdsComplete": flag_row(trailer.link_type_ids_complete), - "linkProperties": link_properties, - "linkPropertiesComplete": flag_row(trailer.link_properties_complete), - }, - }) -} - -/// Renders one completeness set as the sidecar's bool row. -fn flag_row(flags: &DenseBitSlice) -> Vec { - (0..flags.domain_size()) - .map(|edge| flags.contains(EdgeSlot::from_u64(edge))) - .collect() -} - -/// Builds the dense membership set over `domain` rows admitting exactly `members`. -fn dense_set(domain: usize, members: &[u32]) -> Box> { - let mut set = DenseBitSlice::new_empty(domain); - for &member in members { - set.insert(T::from_u32(member)); - } - set -} - -/// Builds one hand-pinned identity: sixteen web bytes then sixteen entity bytes. -fn identity_of(web: u8, entity: u8) -> ArchivedEntityId { - ArchivedEntityId { - web_id: uuid::Uuid::from_bytes([web; 16]).into(), - entity_uuid: uuid::Uuid::from_bytes([entity; 16]).into(), - } -} - -/// Renders one identity column as lowercase hex strings. -fn identities_sidecar(ids: &[ArchivedEntityId]) -> Vec { - use core::fmt::Write as _; - - use zerocopy::IntoBytes as _; - - ids.iter() - .map(|id| { - id.as_bytes() - .iter() - .fold(String::with_capacity(64), |mut hex, byte| { - write!(hex, "{byte:02x}").expect("writing to a string cannot fail"); - hex - }) - }) - .collect() -} - -/// G8: the evolution scenario proven in advance - a slot count one past the v1 tile table. -/// -/// A populated appended slot, and a populated `MASS` slot; a v1 decoder ignores both by contract, -/// so the sidecar's expectations cover only the v1 surface. -fn g8_appended_slot() -> Fixture { - let positions: Vec = (0_u16..6) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(0.25, -0.5), - f32::from(index) * 0.125, - ) - }) - .collect(); - let rows: Vec> = - (0..6).map(|index| WireRow::pinned(2 * index + 1)).collect(); - let ranges = [2_u32..5] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x88; 32]), - variant: 0, - coordinate: TileCoordinate { z: 3, x: 5, y: 2 }, - mode: Mode::Delta, - first_bucket: 5, - runs: &[3], - global: None, - children: 0b0001, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - }; - let encoded = response.encode(); - - // Reassemble the same sections into a six-slot envelope with a - // populated MASS column and one appended opaque slot. - let mass = [7_u32, 1, 9]; - let mut mass_column = Vec::new(); - for value in mass { - mass_column.extend_from_slice(&value.to_le_bytes()); - } - let appended = [0xDE_u8, 0xAD, 0xBE, 0xEF, 0x01]; - - let mut envelope = EnvelopeWriter::new(Kind::Tile, 6); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 0).expect("HEAD is present"))); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 1).expect("POSITIONS is present"))); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 2).expect("ROW_IDS is present"))); - envelope.absent(); - envelope.slot(|buf| buf.extend_from_slice(&mass_column)); - envelope.slot(|buf| buf.extend_from_slice(&appended)); - let bytes = envelope.finish(); - - let sidecar = tile_sidecar( - "g8-appended-slot", - &response, - &bytes, - 0, - None, - Some(&mass), - &json!({ "5": appended }), - ); - Fixture { - name: "g8-appended-slot", - bytes, - sidecar, - } -} - -/// G9. -/// -/// Padding sweep, low widths - `HEAD` sized for pad 1, `TYPE_MASK` for pad 3, `ROW_IDS` for pad 4 -/// (odd delivered). -fn g9_padding_low() -> Fixture { - let positions: Vec = (0_u16..10) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(0.1875, -0.75), - f32::from(index) * 0.09375, - ) - }) - .collect(); - let rows: Vec> = (0..10) - .map(|index| WireRow::pinned(5 * index + 2)) - .collect(); - let ranges = [3_u32..8] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - // Stride 1 over three types, so five mask bytes pad with 3. - let t0 = [3_u32, 6].map(BasePosition::from_u32); - let t1 = [4_u32, 5, 6].map(BasePosition::from_u32); - let empty: [BasePosition; 0] = []; - let masks = [ - Membership::List(&t0), - Membership::List(&t1), - Membership::List(&empty), - ]; - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0x99; 32]), - variant: 24, - // The variant, z, and firstBucket take two-byte arguments - // and x and y three-byte ones, sizing the HEAD to 63 - // bytes: pad 1. - coordinate: TileCoordinate { - z: 24, - x: 1000, - y: 3000, - }, - mode: Mode::Delta, - first_bucket: 26, - runs: &[5], - global: None, - children: 0b0011, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: Some(&masks), - trailer: None, - }; - let bytes = response.encode(); - - // Delivered positions 3..8; t0 at 3, 6; t1 at 4, 5, 6. - let expected_mask = [0b01_u8, 0b10, 0b10, 0b11, 0b00]; - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - expected_mask, - "the hand-derived G9 masks must match the encoder", - ); - - let sidecar = tile_sidecar( - "g9-padding-low", - &response, - &bytes, - 3, - Some(&expected_mask), - None, - &Value::Null, - ); - Fixture { - name: "g9-padding-low", - bytes, - sidecar, - } -} - -/// G10. -/// -/// Padding sweep, high widths - `HEAD` sized for pad 2, `TYPE_MASK` for pad 5, two appended opaque -/// slots for pads 6 and 7. -fn g10_padding_high() -> Fixture { - let positions: Vec = (0_u16..6) - .map(|index| { - Vec2::new( - f32::from(index).mul_add(-0.125, 0.625), - f32::from(index).mul_add(0.25, -0.5), - ) - }) - .collect(); - let rows: Vec> = (0..6).map(|index| WireRow::pinned(13 * index)).collect(); - let ranges = [1_u32..4] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - // One type, stride 1: three mask bytes pad with 5. - let t0 = [2_u32, 3].map(BasePosition::from_u32); - let masks = [Membership::List(&t0)]; - - let response = TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0xAA; 32]), - variant: 25, - // The variant, z, y, and firstBucket take two-byte - // arguments and x a three-byte one, sizing the HEAD to 62 - // bytes: pad 2. - coordinate: TileCoordinate { - z: 25, - x: 300, - y: 170, - }, - mode: Mode::Delta, - first_bucket: 27, - runs: &[3], - global: None, - children: 0b1000, - }, - delivered: DeliveredSet::Ranges(&ranges), - positions: IdSlice::from_raw(&positions), - rows: IdSlice::from_raw(&rows), - arrivals: IdSlice::from_raw(&[]), - masks: Some(&masks), - trailer: None, - }; - let encoded = response.encode(); - - let expected_mask = [0b0_u8, 0b1, 0b1]; - assert_eq!( - section(&encoded, 3).expect("TYPE_MASK is present"), - expected_mask, - "the hand-derived G10 masks must match the encoder", - ); - - // Reassemble with two appended opaque slots: 10 bytes (pad 6) - // and 9 bytes (pad 7). - let six = [0x60_u8; 10]; - let seven = [0x70_u8; 9]; - - let mut envelope = EnvelopeWriter::new(Kind::Tile, 7); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 0).expect("HEAD is present"))); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 1).expect("POSITIONS is present"))); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 2).expect("ROW_IDS is present"))); - envelope.slot(|buf| buf.extend_from_slice(section(&encoded, 3).expect("TYPE_MASK is present"))); - envelope.absent(); - envelope.slot(|buf| buf.extend_from_slice(&six)); - envelope.slot(|buf| buf.extend_from_slice(&seven)); - let bytes = envelope.finish(); - - let sidecar = tile_sidecar( - "g10-padding-high", - &response, - &bytes, - 1, - Some(&expected_mask), - None, - &json!({ "5": six, "6": seven }), - ); - Fixture { - name: "g10-padding-high", - bytes, - sidecar, - } -} diff --git a/libs/@local/graph/atlas/src/salt/wire/locate.rs b/libs/@local/graph/atlas/src/salt/wire/locate.rs deleted file mode 100644 index be62b1e697e..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/locate.rs +++ /dev/null @@ -1,463 +0,0 @@ -//! The locate response: `HEAD`, node and edge columns, and the detail trailer as one envelope. -//! -//! A locate document names its delivered set as an explicit row list - source first, then -//! neighbours in wire order - each row in the domain that publishes it. Fitted rows gather -//! from the generation's base-order columns, and placed arrivals gather from the view's -//! arrival table. Unlike a tile's contiguous ranges the set is arbitrary, so the columns gather -//! point by point. The `SALTILEL` envelope has -//! seven slots: `HEAD`, the tile response's three node column shapes (`POSITIONS`, `ROW_IDS`, -//! `TYPE_MASK`), the edges response's endpoint columns (`EDGE_SOURCES`, `EDGE_TARGETS`), and -//! `EDGE_IDS` - the delivered edges' link-entity identities as raw 32-byte records, the only -//! identity an edge carries on the wire. -//! -//! Every locate document includes the detail trailer, because locate is the detail view. It interns -//! type and property URLs profile-natively. Keys 0 and 1 are string tables that list every -//! referenced URL once in bytewise order, and every type and property reference in the later keys -//! is a uint index into them. The source and the delivered edges get property maps. Neighbour nodes -//! carry a label and a representative type reference, and their own detail is one locate away. The -//! document's consistency laws are producer contracts and panic when violated. -#![expect( - clippy::little_endian_bytes, - reason = "column integers are pinned little-endian by the wire contract" -)] - -use alloc::borrow::Cow; - -use hashql_core::id::{Id as _, IdSlice}; -use type_system::ontology::id::{BaseUrl, VersionedUrl}; -use zerocopy::IntoBytes as _; - -use super::{ - Kind, - cbor::CborWriter, - edges::{write_column, write_identities}, - envelope::EnvelopeWriter, - tile::{TileCoordinate, encode_details}, -}; -use crate::{ - bitset::DenseBitSlice, - dataset::auxiliary::Label, - identity::{BasePosition, NodeRowId}, - integrity::Sha256Digest, - math::Vec2, - postgres::id::ArchivedEntityId, - salt::postings::artifact::Membership, - serve::{ - TableIndex, WireRow, - hydrate::{EdgeSlot, NodeSlot}, - neighbourhood::EdgeColumns, - schedule::{ArrivalIndex, ArrivalRow, ViewRow}, - }, -}; - -/// One locate response in writable form. -#[derive(Debug)] -pub(crate) struct LocateResponse<'doc> { - /// `HEAD` key 0: the generation identity, echoing the route. - pub generation: Sha256Digest, - /// `HEAD` key 1: the variant index, echoing the route. - pub variant: u64, - /// `HEAD` keys 3 and 4: the source's first visible zoom and its tile there. - /// - /// The client's fly-to target. The cell's own `z` is the zoom; both keys ride the wire by the - /// pinned schema. - pub cell: TileCoordinate, - /// `HEAD` key 6: `false` when the locate edge cap truncated the subgraph. - pub complete: bool, - /// `HEAD` key 7: the source's upstream entity id, `bstr(32)`. - /// - /// The web uuid then the entity uuid, sixteen raw bytes each - the generation digest's - /// untagged byte-string shape. A client that named the source by wire row id learns from this - /// key which entity it spotlighted. - pub entity_id: ArchivedEntityId, - /// `HEAD` key 8: whether the request's `coloredTypeIds` cover the source's direct types. - /// - /// `true` when the source records at least one direct type and every one of them names an - /// entry in the requested set - the signal that the client's palette can name everything the - /// source is. Every other case reads `false`. A source the store no longer serves and a source - /// with no recorded types both leave a type list the server cannot attest, and a request set - /// that resolves no entry covers nothing. - pub type_ids_complete: bool, - /// `HEAD` key 9: whether the trailer's source property map is the entity's whole deliverable - /// set. - /// - /// `false` when the scalar-value filter or the property cap dropped anything, and for a source - /// the store no longer serves. - pub properties_complete: bool, - /// The delivered rows in delivered order, source first, each in the domain that publishes - /// it. - pub delivered: &'doc IdSlice, - /// The entry cohort's arrivals, addressed by the delivered arrival rows. - /// - /// Empty when the view holds no admitted arrival. - pub arrivals: &'doc IdSlice, - /// The generation's wire-coordinate column, base order, in full. - pub positions: &'doc IdSlice, - /// The generation's row-id column (row by base position), in full. - pub rows: &'doc IdSlice>, - /// Per-type membership for the request's `coloredTypeIds`, in request order. - /// - /// Bit `i` of every point's mask reads from `masks[i]`. `None` when the request carried no - /// ids: the `TYPE_MASK` slot is then absent rather than empty. - pub masks: Option<&'doc [Membership<'doc>]>, - /// The delivered edges in column form: `EDGE_SOURCES`, `EDGE_TARGETS` and `EDGE_IDS`. - /// - /// `EDGE_IDS` carries `bstr(32)` records. Identity is generation-frozen, so every delivered - /// edge carries one. Edge order itself is ascending by those bytes - the delivery order is - /// client-verifiable from the column alone. - pub edges: &'doc EdgeColumns, - /// The detail trailer; locate is the detail view, so it always rides. - pub trailer: LocateTrailer<'doc>, -} - -impl LocateResponse<'_> { - /// Encodes the response as one `SALTILEL` envelope. - /// - /// # Panics - /// - /// This panics when the trailer arrays do not cover the delivered nodes and edges. - #[must_use] - pub(crate) fn encode(&self) -> Vec { - /// Bytes per delivered point across the node columns: an f32 pair and a row id. - const POINT_SIZE: usize = size_of::<[f32; 2]>() + size_of::(); - /// Bytes per delivered edge across the edge columns: source, target, identity. - const EDGE_SIZE: usize = - size_of::() + size_of::() + size_of::(); - - let edges = self.edges.count(); - self.trailer - .debug_assert_invariants(self.delivered.len(), edges); - - let mut envelope = EnvelopeWriter::new(Kind::Locate, 7); - envelope.reserve(self.delivered.len() * POINT_SIZE + edges * EDGE_SIZE); - envelope.slot(|buf| self.encode_head(buf, edges as u64)); - envelope.slot(|buf| self.write_positions(buf)); - envelope.slot(|buf| self.write_rows(buf)); - match self.masks { - Some(masks) => envelope.slot(|buf| self.write_masks(buf, masks)), - None => envelope.absent(), - } - envelope.slot(|buf| write_column(buf, self.edges.sources())); - envelope.slot(|buf| write_column(buf, self.edges.targets())); - envelope.slot(|buf| write_identities(buf, self.edges.ids())); - - envelope.finish_with_trailer(|buf| self.trailer.encode(buf)) - } - - /// Encodes the `HEAD` map: keys 0 through 9. - fn encode_head(&self, buf: &mut Vec, edges: u64) { - let mut cbor = CborWriter::over(buf); - cbor.map(10); - - cbor.uint(0); - cbor.bytes(&self.generation.to_bytes()); - cbor.uint(1); - cbor.uint(self.variant); - cbor.uint(2); - cbor.uint(self.delivered.len() as u64); - cbor.uint(3); - cbor.uint(u64::from(self.cell.z)); - cbor.uint(4); - cbor.array(3); - cbor.uint(u64::from(self.cell.z)); - cbor.uint(u64::from(self.cell.x)); - cbor.uint(u64::from(self.cell.y)); - cbor.uint(5); - cbor.uint(edges); - cbor.uint(6); - cbor.boolean(self.complete); - cbor.uint(7); - cbor.bytes(zerocopy::IntoBytes::as_bytes(&self.entity_id)); - cbor.uint(8); - cbor.boolean(self.type_ids_complete); - cbor.uint(9); - cbor.boolean(self.properties_complete); - } - - /// Writes the `POSITIONS` column: f32 xy pairs, delivered order. - fn write_positions(&self, column: &mut Vec) { - column.reserve(self.delivered.len() * 8); - for &vessel in self.delivered { - let point = match vessel { - ViewRow::Base(position) => self.positions[position], - ViewRow::Arrival(index) => self.arrivals[index].position, - }; - column.extend_from_slice(&point.x().to_le_bytes()); - column.extend_from_slice(&point.y().to_le_bytes()); - } - } - - /// Writes the `ROW_IDS` column: u32 row ids, delivered order. - fn write_rows(&self, column: &mut Vec) { - column.reserve(self.delivered.len() * 4); - for &vessel in self.delivered { - let wire = match vessel { - ViewRow::Base(position) => self.rows[position], - ViewRow::Arrival(index) => self.arrivals[index].wire, - }; - column.extend_from_slice(&wire.get().to_le_bytes()); - } - } - - /// Assembles the `TYPE_MASK` column. - /// - /// One `ceil(n/8)`-byte mask per delivered point, bit `i` LSB-first when the point carries the - /// request's type `i` - the tile column's shape over an arbitrary delivered list, probed point - /// by point (the set is a spotlight, never a bulk slice). A mask read as its set-bit indexes is - /// the point's colored-type index list. The source's list is the first mask's. - fn write_masks(&self, buf: &mut Vec, masks: &[Membership<'_>]) { - let stride = masks.len().div_ceil(8); - let base = buf.len(); - buf.resize(base + self.delivered.len() * stride, 0); - let column = &mut buf[base..]; - - for (bit, membership) in masks.iter().enumerate() { - let byte = bit >> 3; - let flag = 1_u8 << (bit & 7); - - for (point, &vessel) in self.delivered.iter().enumerate() { - // An arrival's mask reads zero in every bit, the column's answer for an id the - // generation's postings cannot resolve. - let ViewRow::Base(position) = vessel else { - continue; - }; - if membership.contains(position) { - column[point * stride + byte] |= flag; - } - } - } - } -} - -/// The locate detail trailer. -/// -/// The intern tables come first, then node detail in delivered order and link detail in edge order, -/// every type and property reference a uint index into its table. -#[derive(Debug)] -pub(crate) struct LocateTrailer<'trailer> { - /// Trailer key 0: the type intern table - every referenced versioned type URL once, - /// bytewise-sorted. - pub type_table: &'trailer IdSlice, Cow<'trailer, str>>, - /// Trailer key 1: the property intern table - every surviving property base URL once, - /// bytewise-sorted. - pub property_table: &'trailer IdSlice, Cow<'trailer, str>>, - /// Trailer key 2: labels, delivered order. - pub labels: &'trailer IdSlice, - /// Trailer key 3. - /// - /// Each delivered node's representative type as a type-table index, delivered order. `null` - /// marks a node the store no longer serves or whose types the store does not record. - pub type_ids: &'trailer IdSlice>>, - /// Trailer key 4. - /// - /// The source's property map, keyed by uint index into the property table, keys ascending. - /// `null` marks a source the store no longer serves. Neighbour nodes carry no properties, and - /// their detail is one locate away. - pub properties: Option<&'trailer PropertyMap<'trailer>>, - /// Trailer key 5: link labels, edge order. - pub link_labels: &'trailer IdSlice, - /// Trailer key 6. - /// - /// Each delivered edge's direct types as type-table indexes, edge order, canonical type order - /// preserved, capped by the published `locateLinkTypeIds` limit. Empty for a link the store no - /// longer serves. - pub link_type_ids: &'trailer IdSlice>>, - /// Trailer key 7: per-edge type completeness, edge order. - /// - /// Encoded as an LSB-first bitmask in whole 8-byte words, padding bits zero. Bit `e` set - /// means edge `e`'s type list is the link's whole direct set - unset means the cap truncated - /// it or the store no longer serves the link. - pub link_type_ids_complete: &'trailer DenseBitSlice, - /// Trailer key 8. - /// - /// Per-edge property maps, edge order, keyed by uint index into the property table, keys - /// ascending, capped by the published `locateLinkProperties` limit. `null` marks a link the - /// store no longer serves. - pub link_properties: &'trailer IdSlice>>, - /// Trailer key 9: per-edge property completeness, edge order. - /// - /// Encoded as an LSB-first bitmask in whole 8-byte words, padding bits zero. Bit `e` set - /// means edge `e`'s property map is the link entity's whole deliverable set - unset means the - /// scalar-value filter or the cap dropped something, or the store no longer serves the link. - pub link_properties_complete: &'trailer DenseBitSlice, -} - -impl LocateTrailer<'_> { - /// Asserts that the trailer arrays cover the delivered nodes and edges. - /// - /// Coverage is the one trailer law the types do not carry, since the arrays travel as - /// separate slices. Every check is a `debug_assert`, so release builds compile this to - /// nothing. - fn debug_assert_invariants(&self, nodes: usize, edges: usize) { - debug_assert_eq!( - self.labels.len(), - nodes, - "the trailer labels must cover exactly the delivered nodes", - ); - debug_assert_eq!( - self.type_ids.len(), - nodes, - "the trailer type ids must cover exactly the delivered nodes", - ); - debug_assert_eq!( - self.link_labels.len(), - edges, - "the trailer link labels must cover exactly the delivered edges", - ); - debug_assert_eq!( - self.link_type_ids.len(), - edges, - "the trailer link type ids must cover exactly the delivered edges", - ); - debug_assert_eq!( - self.link_properties.len(), - edges, - "the trailer link properties must cover exactly the delivered edges", - ); - debug_assert_eq!( - self.link_type_ids_complete.domain_size(), - edges as u64, - "the trailer link type completeness must cover exactly the delivered edges", - ); - debug_assert_eq!( - self.link_properties_complete.domain_size(), - edges as u64, - "the trailer link property completeness must cover exactly the delivered edges", - ); - } - - /// Encodes the trailer tail as one self-delimiting CBOR map. - fn encode(&self, buf: &mut Vec) { - let mut cbor = CborWriter::over(buf); - cbor.map(10); - - cbor.uint(0); - cbor.array(self.type_table.len() as u64); - for url in self.type_table { - cbor.text(url); - } - - cbor.uint(1); - cbor.array(self.property_table.len() as u64); - for url in self.property_table { - cbor.text(url); - } - - cbor.uint(2); - encode_details(&mut cbor, self.labels.iter()); - - cbor.uint(3); - cbor.array(self.type_ids.len() as u64); - for entry in self.type_ids { - match entry { - Some(index) => cbor.uint(index.as_u64()), - None => cbor.null(), - } - } - cbor.uint(4); - encode_property_map(&mut cbor, self.properties); - - cbor.uint(5); - encode_details(&mut cbor, self.link_labels.iter()); - - cbor.uint(6); - cbor.array(self.link_type_ids.len() as u64); - for indexes in self.link_type_ids { - cbor.array(indexes.len() as u64); - for &index in indexes { - cbor.uint(index.as_u64()); - } - } - - cbor.uint(7); - cbor.bytes(self.link_type_ids_complete.words().as_bytes()); - - cbor.uint(8); - cbor.array(self.link_properties.len() as u64); - for map in self.link_properties { - encode_property_map(&mut cbor, *map); - } - - cbor.uint(9); - cbor.bytes(self.link_properties_complete.words().as_bytes()); - } -} - -/// A property-table key paired with its value, one entry of an encoded property map. -pub(crate) type PropertyEntry<'trailer> = (TableIndex, PropertyValue<'trailer>); - -/// One entity's wire property map, keys ascending into the property table. -/// -/// Ascending keys are the interning derivation's own order. Hydration emits each entity's -/// surviving properties ascending by base URL, a base URL renders as its own string, and the -/// table's wire order is bytewise over renderings, so ascending names map to ascending indexes. -#[derive(Debug, PartialEq)] -pub(crate) struct PropertyMap<'doc> { - /// The encoded entries, keys ascending. - entries: Vec>, -} - -impl<'doc> PropertyMap<'doc> { - /// Builds one map over interned entries. - /// - /// The keys must ascend by table index. - pub(crate) fn new_unchecked(entries: Vec>) -> Self { - // Safe fn: the ascending-keys invariant is correctness rather than memory safety, and - // debug builds check it as a maintainer tripwire. - debug_assert!( - entries.is_sorted_by(|left, right| left.0 < right.0), - "property map keys must ascend", - ); - - Self { entries } - } - - fn encode(&self, cbor: &mut CborWriter<'_>) { - cbor.map(self.entries.len() as u64); - for &(index, ref value) in &self.entries { - cbor.uint(index.as_u64()); - value.encode(cbor); - } - } -} - -/// Encodes one property map, `null` for an entity the store no longer serves. -fn encode_property_map(cbor: &mut CborWriter<'_>, map: Option<&PropertyMap<'_>>) { - match map { - Some(map) => map.encode(cbor), - None => cbor.null(), - } -} - -/// One scalar property value. -/// -/// These variants are the only value shapes the wire encodes. Nested objects and arrays never -/// survive hydration. -#[derive(Debug, Clone, PartialEq)] -pub(crate) enum PropertyValue<'doc> { - /// A text scalar. - Text(&'doc str), - /// An integral scalar. - Integer(i64), - /// A floating scalar. - /// - /// Store scalars are doubles and stay double on the wire. - Float(f64), - /// A boolean scalar. - Boolean(bool), - /// An explicit null the store carries. - Null, -} - -impl PropertyValue<'_> { - /// Emits the value in the profile's encoding. - fn encode(&self, cbor: &mut CborWriter<'_>) { - match *self { - Self::Text(value) => cbor.text(value), - Self::Integer(value) => cbor.int(value), - Self::Float(value) => cbor.f64(value), - Self::Boolean(value) => cbor.boolean(value), - Self::Null => cbor.null(), - } - } -} diff --git a/libs/@local/graph/atlas/src/salt/wire/mod.rs b/libs/@local/graph/atlas/src/salt/wire/mod.rs deleted file mode 100644 index 34df3746c97..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/mod.rs +++ /dev/null @@ -1,83 +0,0 @@ -//! Atlas responses as `SALTILE` envelope bytes. -//! -//! The envelope layout is a pinned public contract; the checked-in fixtures under `fixtures/wire/` -//! are the cross-language proof the TypeScript decoder builds against. One envelope carries every -//! binary response kind - tile, edges and locate - as a 16-byte prefix, a fixed offset directory, -//! 8-aligned payload sections, and an optional CBOR trailer tail. Structured payloads are CBOR -//! under the deterministic profile in [`cbor`]; columns are raw little-endian arrays a decoder -//! views without parsing. -//! -//! The module emits bytes and nothing else. A response document goes in and one `Vec` comes -//! out. Server code assembles every input from validated artifacts and an admitted request, so a -//! disagreement between document fields is a producer bug that panics. No function here returns a -//! data-dependent error. Encoding is deterministic, so equal documents yield byte-identical -//! responses, which is the property the client's application-layer cache keys on. Encoding is also -//! synchronous: the endpoint schedules it on a rayon worker, never on an async runtime thread. -//! -//! [`tile::TileResponse`], [`edges::EdgesResponse`] and [`locate::LocateResponse`] are the v1 -//! documents. -#![expect( - clippy::little_endian_bytes, - reason = "the kind discriminants read the eight magic bytes little-endian so the envelope's \ - IntoBytes write reproduces them verbatim, as the wire contract pins" -)] - -pub(crate) mod cbor; -pub(crate) mod edges; -pub(crate) mod envelope; -pub(crate) mod locate; -pub(crate) mod tile; - -#[cfg(test)] -mod fixtures; -// Crate-visible so the serving tests reuse the section-carving -// helpers against served bytes. -#[cfg(test)] -pub(crate) mod tests; - -/// The envelope wire version, prefix field and media-type suffix. -/// -/// The media type `application/vnd.hash.saltile-v1` must agree with this value; kind discriminates -/// the grammar variant, the version tracks evolution of the whole family. -pub(crate) const WIRE_VERSION: u16 = 1; - -/// A response kind, named by the eighth magic byte as an ASCII initial. -/// -/// The seven-byte family prefix `SALTILE` is constant; the kind byte selects the `HEAD` schema and -/// slot table the decoder applies. -#[derive(Debug, Copy, Clone, PartialEq, Eq, zerocopy::IntoBytes, zerocopy::Immutable)] -#[repr(u64)] -pub(crate) enum Kind { - /// A tile response, magic `SALTILET`. - Tile = u64::from_le_bytes(*b"SALTILET"), - /// An edges response, magic `SALTILEE`. - Edges = u64::from_le_bytes(*b"SALTILEE"), - /// A locate response, magic `SALTILEL`. - Locate = u64::from_le_bytes(*b"SALTILEL"), -} - -/// A tile delivery mode, `HEAD` key 3. -/// -/// Requests carry the mode as the JSON strings `"delta"` and `"total"`; delta is the default when a -/// request names none. -#[derive(Debug, Default, Copy, Clone, PartialEq, Eq, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "lowercase")] -pub(crate) enum Mode { - /// This tile's own additions, on top of the ancestor deliveries the client accumulates. The - /// default. - #[default] - Delta, - /// The whole delivered set for the tile at its zoom, so the tile renders alone. - Total, -} - -impl Mode { - /// Returns the mode's wire code. - #[must_use] - pub(crate) const fn code(self) -> u64 { - match self { - Self::Delta => 0, - Self::Total => 1, - } - } -} diff --git a/libs/@local/graph/atlas/src/salt/wire/tests.rs b/libs/@local/graph/atlas/src/salt/wire/tests.rs deleted file mode 100644 index 2ca166fb2a1..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/tests.rs +++ /dev/null @@ -1,1252 +0,0 @@ -//! Wire encoder tests: hand-derived bytes for every layer. -//! -//! CBOR expectations come from RFC 8949 appendix A where the profile covers them, and otherwise -//! this module derives them by hand at the byte level. Comments above the assertions derive the -//! envelope and response expectations. The checked-in fixtures (`fixtures.rs`) carry the -//! cross-language corpus, and these tests pin one layer each so a failure names its layer. -#![expect( - clippy::little_endian_bytes, - reason = "the tests read and write the contract's little-endian wire integers" -)] -#![expect( - clippy::single_range_in_vec_init, - reason = "a delta tile's delivered set really is one contiguous range" -)] - -use hashql_core::id::Id; -use proptest::{arbitrary::any, prop_assert, prop_assert_eq, property_test}; - -use super::{ - Kind, Mode, - cbor::CborWriter, - edges::{EdgesResponse, EdgesTrailer}, - envelope::EnvelopeWriter, - locate::{LocateResponse, LocateTrailer, PropertyMap, PropertyValue}, - tile::{DeliveredSet, GlobalHead, TileCoordinate, TileHead, TileResponse, TileTrailer}, -}; -use crate::{ - bitset::DenseBitSlice, - dataset::auxiliary::{Icon, Label}, - identity::{BasePosition, NodeRowId}, - integrity::Sha256Digest, - math::{Bounds2, Vec2}, - postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId}, - salt::postings::artifact::Membership, - serve::WireRow, -}; - -/// Builds the dense membership set over `domain` rows admitting exactly `members`. -fn dense_set(domain: usize, members: &[u32]) -> Box> { - let mut set = DenseBitSlice::new_empty(domain); - for &member in members { - set.insert(T::from_u32(member)); - } - set -} - -/// Builds one uniform-byte identity record: `byte` in both uuid halves. -const fn identity_of(byte: u8) -> ArchivedEntityId { - ArchivedEntityId { - web_id: ArchivedWebId::from_bytes([byte; 16]), - entity_uuid: ArchivedEntityUuid::from_bytes([byte; 16]), - } -} - -/// Reads slot `slot`'s directory entry from a finished response. -pub(crate) fn directory(bytes: &[u8], slot: usize) -> (u32, u32) { - let entry = 16 + 8 * slot; - let start = u32::from_le_bytes(bytes[entry..entry + 4].try_into().expect("four bytes")); - let end = u32::from_le_bytes(bytes[entry + 4..entry + 8].try_into().expect("four bytes")); - - (start, end) -} - -/// Slices slot `slot`'s payload, or [`None`] when the slot is absent. -pub(crate) fn section(bytes: &[u8], slot: usize) -> Option<&[u8]> { - let (start, end) = directory(bytes, slot); - if (start, end) == (0, 0) { - return None; - } - - Some(&bytes[start as usize..end as usize]) -} - -mod cbor { - use super::CborWriter; - - /// Encodes one value through a fresh writer. - fn encoded(emit: impl FnOnce(&mut CborWriter<'_>)) -> Vec { - let mut bytes = Vec::new(); - emit(&mut CborWriter::over(&mut bytes)); - bytes - } - - #[test] - fn uints_take_the_shortest_form() { - // RFC 8949 appendix A rows, restricted to unsigned integers. - for (value, expected) in [ - (0_u64, vec![0x00_u8]), - (10, vec![0x0A]), - (23, vec![0x17]), - (24, vec![0x18, 0x18]), - (100, vec![0x18, 0x64]), - (255, vec![0x18, 0xFF]), - (256, vec![0x19, 0x01, 0x00]), - (1000, vec![0x19, 0x03, 0xE8]), - (0xFFFF, vec![0x19, 0xFF, 0xFF]), - (0x1_0000, vec![0x1A, 0x00, 0x01, 0x00, 0x00]), - (1_000_000, vec![0x1A, 0x00, 0x0F, 0x42, 0x40]), - (u64::from(u32::MAX), vec![0x1A, 0xFF, 0xFF, 0xFF, 0xFF]), - ( - 1 << 32, - vec![0x1B, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x00], - ), - ( - 1_000_000_000_000, - vec![0x1B, 0x00, 0x00, 0x00, 0xE8, 0xD4, 0xA5, 0x10, 0x00], - ), - ] { - assert_eq!( - encoded(|cbor| cbor.uint(value)), - expected, - "uint {value} must take the shortest form", - ); - } - } - - #[test] - #[expect( - clippy::redundant_closure_for_method_calls, - reason = "CborWriter::null as a fn item fails higher-ranked inference over the writer's \ - borrowed buffer; the closure is the working spelling" - )] - fn simple_values_and_floats() { - assert_eq!(encoded(|cbor| cbor.boolean(false)), [0xF4]); - assert_eq!(encoded(|cbor| cbor.boolean(true)), [0xF5]); - assert_eq!(encoded(|cbor| cbor.null()), [0xF6]); - - // RFC 8949 appendix A: 100000.0f32 = 0xFA_47C35000; the f32 - // maximum = 0xFA_7F7FFFFF. 1.0 and -0.5 derived by hand from - // the IEEE 754 single layout. - assert_eq!( - encoded(|cbor| cbor.f32(100_000.0)), - [0xFA, 0x47, 0xC3, 0x50, 0x00] - ); - assert_eq!( - encoded(|cbor| cbor.f32(f32::MAX)), - [0xFA, 0x7F, 0x7F, 0xFF, 0xFF] - ); - assert_eq!( - encoded(|cbor| cbor.f32(1.0)), - [0xFA, 0x3F, 0x80, 0x00, 0x00] - ); - assert_eq!( - encoded(|cbor| cbor.f32(-0.5)), - [0xFA, 0xBF, 0x00, 0x00, 0x00] - ); - } - - #[test] - fn signed_integers_take_both_majors() { - // Non-negative values ride major type 0, negatives major - // type 1 with argument -1 - n; -1, -10, -100, -1000 are RFC - // 8949 appendix A rows, the extremes derived by hand. - assert_eq!(encoded(|cbor| cbor.int(5)), [0x05]); - assert_eq!(encoded(|cbor| cbor.int(0)), [0x00]); - assert_eq!(encoded(|cbor| cbor.int(-1)), [0x20]); - assert_eq!(encoded(|cbor| cbor.int(-10)), [0x29]); - assert_eq!(encoded(|cbor| cbor.int(-100)), [0x38, 0x63]); - assert_eq!(encoded(|cbor| cbor.int(-1000)), [0x39, 0x03, 0xE7]); - assert_eq!( - encoded(|cbor| cbor.int(i64::MAX)), - [0x1B, 0x7F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF] - ); - assert_eq!( - encoded(|cbor| cbor.int(i64::MIN)), - [0x3B, 0x7F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF] - ); - } - - #[test] - fn doubles_are_fixed_width() { - // RFC 8949 appendix A gives 1.1 = 0xFB_3FF199999999999A. The 0.5 case follows by hand from - // the IEEE 754 double layout. - assert_eq!( - encoded(|cbor| cbor.f64(1.1)), - [0xFB, 0x3F, 0xF1, 0x99, 0x99, 0x99, 0x99, 0x99, 0x9A] - ); - assert_eq!( - encoded(|cbor| cbor.f64(0.5)), - [0xFB, 0x3F, 0xE0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00] - ); - } - - #[test] - fn strings_carry_byte_lengths() { - assert_eq!(encoded(|cbor| cbor.bytes(&[])), [0x40]); - assert_eq!( - encoded(|cbor| cbor.bytes(&[0x01, 0x02, 0x03, 0x04])), - [0x44, 0x01, 0x02, 0x03, 0x04] - ); - assert_eq!(encoded(|cbor| cbor.text("")), [0x60]); - // RFC 8949 appendix A: "IETF", "\u{fc}" (two UTF-8 bytes), - // "\u{6c34}" (three UTF-8 bytes). - assert_eq!( - encoded(|cbor| cbor.text("IETF")), - [0x64, 0x49, 0x45, 0x54, 0x46] - ); - assert_eq!(encoded(|cbor| cbor.text("\u{fc}")), [0x62, 0xC3, 0xBC]); - assert_eq!( - encoded(|cbor| cbor.text("\u{6c34}")), - [0x63, 0xE6, 0xB0, 0xB4] - ); - - // The length argument follows the integer rules: 24 bytes of - // text need the one-byte argument form. - let long = "abcdefghijklmnopqrstuvwx"; - let encoded_long = encoded(|cbor| cbor.text(long)); - assert_eq!(encoded_long[..2], [0x78, 0x18]); - assert_eq!(&encoded_long[2..], long.as_bytes()); - } - - #[test] - fn container_heads() { - assert_eq!(encoded(|cbor| cbor.array(0)), [0x80]); - assert_eq!(encoded(|cbor| cbor.array(3)), [0x83]); - // RFC 8949 appendix A: a 25-item array head. - assert_eq!(encoded(|cbor| cbor.array(25)), [0x98, 0x19]); - assert_eq!(encoded(|cbor| cbor.map(0)), [0xA0]); - assert_eq!(encoded(|cbor| cbor.map(2)), [0xA2]); - } -} - -mod envelope { - use super::{EnvelopeWriter, Kind, directory}; - - #[test] - fn two_slot_envelope_lays_out_by_hand() { - let mut envelope = EnvelopeWriter::new(Kind::Tile, 2); - envelope.slot(|buf| buf.extend_from_slice(&[0xAA, 0xBB, 0xCC])); - envelope.absent(); - let bytes = envelope.finish(); - - // prefix (16) + directory (16) = payload region at 32; the - // 3-byte payload pads with 5 zeros to 40. - let expected = [ - b'S', b'A', b'L', b'T', b'I', b'L', b'E', b'T', // magic - 0x01, 0x00, // wireVersion 1 - 0x00, 0x00, // flags 0 - 0x02, 0x00, // slotCount 2 - 0x00, 0x00, // reserved 0 - 32, 0, 0, 0, 35, 0, 0, 0, // slot 0: (32, 35) - 0, 0, 0, 0, 0, 0, 0, 0, // slot 1: absent - 0xAA, 0xBB, 0xCC, 0, 0, 0, 0, 0, // payload, padded to 8 - ]; - assert_eq!(bytes, expected); - } - - #[test] - fn present_empty_is_distinct_from_absent() { - let mut envelope = EnvelopeWriter::new(Kind::Edges, 3); - envelope.slot(|buf| buf.extend_from_slice(&[0x01])); - envelope.slot(|buf| buf.extend_from_slice(&[])); - envelope.absent(); - let bytes = envelope.finish(); - - // Payload region starts at 16 + 24 = 40; slot 0 is (40, 41) - // padded to 48; slot 1 is present-empty at (48, 48); slot 2 - // is absent (0, 0). - assert_eq!(directory(&bytes, 0), (40, 41)); - assert_eq!(directory(&bytes, 1), (48, 48)); - assert_eq!(directory(&bytes, 2), (0, 0)); - assert_eq!(bytes.len(), 48); - assert_eq!(bytes[0..8], *b"SALTILEE"); - } - - #[test] - fn trailer_follows_the_aligned_end_unpadded() { - let mut envelope = EnvelopeWriter::new(Kind::Tile, 1); - envelope.slot(|buf| buf.extend_from_slice(&[0x11, 0x22])); - let bytes = envelope.finish_with_trailer(|buf| buf.extend_from_slice(&[0xA0])); - - // Payload region at 24; payload (24, 26) pads to 32; the - // one-byte trailer tail follows unpadded. - assert_eq!(directory(&bytes, 0), (24, 26)); - assert_eq!(bytes.len(), 33); - assert_eq!(bytes[32], 0xA0); - assert_eq!(bytes[26..32], [0; 6]); - } - - #[test] - #[should_panic(expected = "slot 0 (HEAD) is always present")] - fn slot_zero_cannot_be_absent() { - let mut envelope = EnvelopeWriter::new(Kind::Tile, 2); - envelope.absent(); - } - - #[test] - #[should_panic(expected = "declares 1 slots, all recorded")] - fn extra_slots_are_rejected() { - let mut envelope = EnvelopeWriter::new(Kind::Tile, 1); - envelope.slot(|buf| buf.extend_from_slice(&[0x01])); - envelope.slot(|buf| buf.extend_from_slice(&[0x02])); - } - - #[test] - #[should_panic(expected = "declares 2 slots")] - fn missing_slots_are_rejected() { - let mut envelope = EnvelopeWriter::new(Kind::Tile, 2); - envelope.slot(|buf| buf.extend_from_slice(&[0x01])); - let _bytes = envelope.finish(); - } -} - -/// The directory laws hold for every present/absent payload mix. -/// -/// Sequential 8-aligned starts from the fixed payload origin, extents matching payload lengths, -/// zero padding, and a total length of the last present end aligned to 8. -#[property_test] -fn envelope_directory_laws( - #[strategy = proptest::collection::vec( - proptest::option::weighted(0.7, proptest::collection::vec(any::(), 0..64)), - 1..12, - )] - payloads: Vec>>, -) { - // Slot 0 is always present. - let mut payloads = payloads; - if payloads[0].is_none() { - payloads[0] = Some(vec![0xFF]); - } - - let slots = u16::try_from(payloads.len()).expect("the strategy draws at most 12 slots"); - let mut envelope = EnvelopeWriter::new(Kind::Tile, slots); - for payload in &payloads { - match payload { - Some(bytes) => envelope.slot(|buf| buf.extend_from_slice(bytes)), - None => envelope.absent(), - } - } - let bytes = envelope.finish(); - - let mut cursor = 16 + 8 * payloads.len(); - prop_assert_eq!(cursor & 7, 0); - for (slot, payload) in payloads.iter().enumerate() { - let (start, end) = directory(&bytes, slot); - if let Some(expected) = payload { - prop_assert_eq!(start as usize, cursor); - prop_assert_eq!(start & 7, 0); - prop_assert_eq!((end - start) as usize, expected.len()); - prop_assert_eq!(&bytes[start as usize..end as usize], expected.as_slice()); - - let padded = (end as usize).next_multiple_of(8); - prop_assert!(bytes[end as usize..padded].iter().all(|&byte| byte == 0)); - cursor = padded; - } else { - prop_assert_eq!((start, end), (0, 0)); - } - } - prop_assert_eq!(bytes.len(), cursor); -} - -mod tile { - use hashql_core::id::{Id as _, IdSlice}; - - use super::{ - ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId, BasePosition, Bounds2, DeliveredSet, - GlobalHead, Icon, Label, Membership, Mode, NodeRowId, Sha256Digest, TileCoordinate, - TileHead, TileResponse, TileTrailer, Vec2, WireRow, dense_set, section, - }; - use crate::{ - dataset::auxiliary::OwnedLegend, - identity::OntologyRowId, - serve::schedule::{ArrivalIndex, ArrivalRow, Splice, ViewRow}, - }; - - /// Builds a consistent tile of two points from one delta run. - fn minimal<'doc>( - positions: &'doc [Vec2], - rows: &'doc [WireRow], - ranges: &'doc [core::ops::Range], - ) -> TileResponse<'doc> { - TileResponse { - head: TileHead { - generation: Sha256Digest::from_bytes_unchecked([0xAB; 32]), - variant: 0, - coordinate: TileCoordinate { z: 3, x: 2, y: 5 }, - mode: Mode::Delta, - first_bucket: 9, - runs: &[2], - global: None, - children: 0b0010, - }, - delivered: DeliveredSet::Ranges(ranges), - arrivals: IdSlice::from_raw(&[]), - positions: IdSlice::from_raw(positions), - rows: IdSlice::from_raw(rows), - masks: None, - trailer: None, - } - } - - #[test] - fn head_encodes_by_hand() { - let positions = [Vec2::new(0.0, 0.0); 12]; - let rows: Vec> = (0..12).map(WireRow::pinned).collect(); - let ranges = [10_u32..12] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - let bytes = minimal(&positions, &rows, &ranges).encode(); - - // map(9): 0 bstr(32) | 1 uint 0 | 2 [3, 2, 5] | 3 uint 0 | - // 4 uint 2 | 6 uint 9 | 7 [2] | 9 uint 2 | - // 10 false. Key 5 is retired, and key 8 is absent without a global map. - let mut expected = vec![0xA9, 0x00, 0x58, 0x20]; - expected.extend_from_slice(&[0xAB; 32]); - expected.extend_from_slice(&[ - 0x01, 0x00, // variant 0 - 0x02, 0x83, 0x03, 0x02, 0x05, // coordinate [3, 2, 5] - 0x03, 0x00, // mode delta - 0x04, 0x02, // delivered 2 - 0x06, 0x09, // firstBucket 9 - 0x07, 0x81, 0x02, // runs [2] - 0x09, 0x02, // children 0b0010 - 0x0A, 0xF4, // trailer false - ]); - assert_eq!(section(&bytes, 0).expect("HEAD is present"), expected); - } - - #[test] - fn global_map_encodes_by_hand() { - let positions = [Vec2::new(0.0, 0.0); 12]; - let rows: Vec> = (0..12).map(WireRow::pinned).collect(); - let ranges = [10_u32..12] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - let mut response = minimal(&positions, &rows, &ranges); - response.head.global = Some(GlobalHead { - visible: 40, - bounds: Bounds2::new(Vec2::new(-0.5, -1.0), Vec2::new(0.25, 1.0)), - min_resolution: 12, - }); - let bytes = response.encode(); - - let head = section(&bytes, 0).expect("HEAD is present"); - // The head map now holds ten entries, and key 8 sits between 7 and 9: - // map(3): 0 uint 40 | 1 [-0.5, -1.0, 0.25, 1.0] | 2 uint 12. - assert_eq!(head[0], 0xAA); - let global = [ - 0x08, 0xA3, // key 8, map(3) - 0x00, 0x18, 0x28, // visible 40 - 0x01, 0x84, // bounds, array(4) - 0xFA, 0xBF, 0x00, 0x00, 0x00, // -0.5 - 0xFA, 0xBF, 0x80, 0x00, 0x00, // -1.0 - 0xFA, 0x3E, 0x80, 0x00, 0x00, // 0.25 - 0xFA, 0x3F, 0x80, 0x00, 0x00, // 1.0 - 0x02, 0x0C, // minResolution 12 - ]; - let at = head - .windows(global.len()) - .position(|window| window == global) - .expect("the global map is embedded in the HEAD"); - // Key 7's runs array [2] precedes it, key 9 follows. - assert_eq!(head[at - 3..at], [0x07, 0x81, 0x02]); - assert_eq!(head[at + global.len()], 0x09); - } - - #[test] - fn columns_gather_across_ranges() { - let positions: Vec = (0_u16..8) - .map(|index| { - Vec2::new( - f32::from(index) * 0.25, - f32::from(index).mul_add(0.125, -1.0), - ) - }) - .collect(); - let rows: Vec> = - (0..8).map(|index| WireRow::pinned(100 + index)).collect(); - let ranges = [1_u32..3, 6..7] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let mut response = minimal(&positions, &rows, &ranges); - response.head.runs = &[3]; - let bytes = response.encode(); - - // POSITIONS: xy pairs of base positions 1, 2, 6. - let mut expected = Vec::new(); - for position in [1_usize, 2, 6] { - expected.extend_from_slice(&positions[position].x().to_le_bytes()); - expected.extend_from_slice(&positions[position].y().to_le_bytes()); - } - assert_eq!(section(&bytes, 1).expect("POSITIONS is present"), expected); - - // ROW_IDS: rows 101, 102, 106. - let mut expected = Vec::new(); - for row in [101_u32, 102, 106] { - expected.extend_from_slice(&row.to_le_bytes()); - } - assert_eq!(section(&bytes, 2).expect("ROW_IDS is present"), expected); - - // TYPE_MASK is absent without coloredTypeIds, MASS reserved. - assert_eq!(section(&bytes, 3), None); - assert_eq!(section(&bytes, 4), None); - } - - #[test] - fn masks_interleave_list_and_dense_membership() { - let positions = [Vec2::new(0.0, 0.0); 22]; - let rows: Vec> = (0..22).map(WireRow::pinned).collect(); - // Delivered points: base positions 4, 5, 6, 20, 21. - let ranges = [4_u32..7, 20..22] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - // type 0: list members at 5 and 20 (33 lies outside the base - // domain of any range and is never consulted). - let list = [5_u32, 20, 33].map(BasePosition::from_u32); - // type 1: dense bits at 4 and 21 over N = 22. - let dense = dense_set(22, &[4, 21]); - // type 2: no members. - let empty: [BasePosition; 0] = []; - let masks = [ - Membership::List(&list), - Membership::Dense(&dense), - Membership::List(&empty), - ]; - - let mut response = minimal(&positions, &rows, &ranges); - response.head.runs = &[5]; - response.masks = Some(&masks); - let bytes = response.encode(); - - // Stride ceil(3/8) = 1. Point order 4, 5, 6, 20, 21: - // type 1 | type 0 | none | type 0 | type 1. - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - [0b010, 0b001, 0b000, 0b001, 0b010] - ); - } - - #[test] - fn wide_masks_round_the_stride_up() { - let positions = [Vec2::new(0.0, 0.0); 4]; - let rows: Vec> = (0..4).map(WireRow::pinned).collect(); - let ranges = [0_u32..4] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - // With nine types the stride is 2. Type 8's bit is byte 1, bit 0. - let low = [0_u32, 2].map(BasePosition::from_u32); - let high = [2_u32].map(BasePosition::from_u32); - let empty: [BasePosition; 0] = []; - let masks = [ - Membership::List(&low), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&empty), - Membership::List(&high), - ]; - - let mut response = minimal(&positions, &rows, &ranges); - response.head.runs = &[4]; - response.masks = Some(&masks); - let bytes = response.encode(); - - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - [1, 0, 0, 0, 1, 1, 0, 0] - ); - } - - #[test] - fn masks_merge_a_gathered_delivered_set() { - let positions = [Vec2::new(0.0, 0.0); 22]; - let rows: Vec> = (0..22).map(WireRow::pinned).collect(); - // The same five points as the range form, gathered: today's masked walk visits - // ascending corpus buckets, so its list ascends in base position. - let delivered = [4_u32, 5, 6, 20, 21].map(|n| ViewRow::Base(BasePosition::from_u32(n))); - - let list = [5_u32, 20, 33].map(BasePosition::from_u32); - let dense = dense_set(22, &[4, 21]); - let empty: [BasePosition; 0] = []; - let masks = [ - Membership::List(&list), - Membership::Dense(&dense), - Membership::List(&empty), - ]; - - let mut response = minimal(&positions, &rows, &[]); - response.delivered = DeliveredSet::Positions(&delivered); - response.head.first_bucket = 0; - response.head.runs = &[5]; - response.masks = Some(&masks); - let bytes = response.encode(); - - // Point order 4, 5, 6, 20, 21: type 1 | type 0 | none | type 0 | type 1 - the - // range form's column, reached through the other producer shape. - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - [0b010, 0b001, 0b000, 0b001, 0b010] - ); - } - - /// A spliced delivery must encode exactly as the gathered list naming the same merged order, - /// because the shapes are producer conveniences and the wire carries neither. - /// - /// The fixture splices one arrival mid-range and one past the last base row, with masks - /// whose bits differ per base position. A walk that misplaces a splice cannot reproduce the - /// gathered form's bytes, and neither can one that shifts a mask bit or drops the trailing - /// arrival. - #[test] - fn a_spliced_delivery_encodes_as_its_gathered_equivalent() { - let positions: Vec = (0..22_u8) - .map(|index| Vec2::new(f32::from(index), -f32::from(index))) - .collect(); - let rows: Vec> = - (0..22_u32).map(|row| WireRow::pinned(100 + row)).collect(); - let arrivals = [ - arrival_row(0xA0, Vec2::new(0.25, -0.5), 900), - arrival_row(0xA1, Vec2::new(-0.75, 0.75), 901), - ]; - let table = IdSlice::from_raw(&arrivals); - - // type 0: base 5 and 20; type 1: dense bits at 4, 20 and 21. An arrival has no postings - // row, so its mask stays zero in both shapes. - let list = [5_u32, 20].map(BasePosition::from_u32); - let dense = dense_set(22, &[4, 20, 21]); - let masks = [Membership::List(&list), Membership::Dense(&dense)]; - - // Merged order: 4, 5, A0, 6, 20, 21, A1 - arrival 0 at delivery index 2, arrival 1 - // trailing at index 6. The runs partition the merged order as [4, 5, A0] and - // [6, 20, 21, A1]. - let ranges = [ - BasePosition::from_u32(4)..BasePosition::from_u32(7), - BasePosition::from_u32(20)..BasePosition::from_u32(22), - ]; - let splices = [ - Splice { - at: 2, - arrival: ArrivalIndex::from_u32(0), - }, - Splice { - at: 6, - arrival: ArrivalIndex::from_u32(1), - }, - ]; - let mut interleaved = minimal(&positions, &rows, &[]); - interleaved.delivered = DeliveredSet::Spliced { - ranges: &ranges, - splices: &splices, - }; - interleaved.arrivals = table; - interleaved.head.runs = &[3, 4]; - interleaved.masks = Some(&masks); - - let gathered_rows = [ - ViewRow::Base(BasePosition::from_u32(4)), - ViewRow::Base(BasePosition::from_u32(5)), - ViewRow::Arrival(ArrivalIndex::from_u32(0)), - ViewRow::Base(BasePosition::from_u32(6)), - ViewRow::Base(BasePosition::from_u32(20)), - ViewRow::Base(BasePosition::from_u32(21)), - ViewRow::Arrival(ArrivalIndex::from_u32(1)), - ]; - let mut gathered = minimal(&positions, &rows, &[]); - gathered.delivered = DeliveredSet::Positions(&gathered_rows); - gathered.arrivals = table; - gathered.head.runs = &[3, 4]; - gathered.masks = Some(&masks); - - assert_eq!( - interleaved.encode(), - gathered.encode(), - "the two producer shapes of one merged order encode identical bytes", - ); - } - - /// Builds one arrival row from a seed, a frozen coordinate, and a pinned wire id. - fn arrival_row(seed: u8, point: Vec2, wire: u32) -> ArrivalRow { - ArrivalRow { - identity: ArchivedEntityId { - web_id: ArchivedWebId::from_bytes([seed; 16]), - entity_uuid: ArchivedEntityUuid::from_bytes([seed ^ 0xFF; 16]), - }, - position: point, - wire: WireRow::pinned(wire), - legend: OwnedLegend::new(OntologyRowId::new(0), Label::new("arrival")), - } - } - - /// Scope-bucket delivery order does not ascend in corpus base position, so this test proves - /// every column association against delivered lists that break the ascent. - /// - /// The memberships give the five points five distinct masks, so no permutation error can - /// coincide with the right answer, and the first list keeps the lowest position first and the - /// highest last: a merge that assumed monotonicity still spans the correct window and still - /// mis-attributes the interior, which is the silent shape of the defect. - #[test] - fn masks_follow_delivery_order_when_it_inverts_base_order() { - // Coordinates and row ids are both distinct per base position and distinct from each - // other, so a column indexed by the wrong quantity cannot coincide with the right one. - let positions: Vec = (0..22_u8) - .map(|index| Vec2::new(f32::from(index), -f32::from(index))) - .collect(); - let rows: Vec> = - (0..22_u32).map(|row| WireRow::pinned(100 + row)).collect(); - - // type 0: 5 and 20 (33 lies outside the base domain and is never consulted). - let list = [5_u32, 20, 33].map(BasePosition::from_u32); - // type 1: dense bits at 4, 20 and 21 over N = 22. - let dense = dense_set(22, &[4, 20, 21]); - // type 2: 6 and 21. - let third = [6_u32, 21].map(BasePosition::from_u32); - let masks = [ - Membership::List(&list), - Membership::Dense(&dense), - Membership::List(&third), - ]; - // Masks by base position are 0b010 at 4, 0b001 at 5, 0b100 at 6, 0b011 at 20, and 0b110 at - // 21. - - let scrambled = [4_u32, 20, 6, 5, 21].map(|n| ViewRow::Base(BasePosition::from_u32(n))); - let mut response = minimal(&positions, &rows, &[]); - response.delivered = DeliveredSet::Positions(&scrambled); - response.head.first_bucket = 0; - response.head.runs = &[2, 3]; - response.masks = Some(&masks); - let bytes = response.encode(); - - // The column is f32 xy pairs in delivered order, so the flat reading is the contract's. - let (coordinates, _) = section(&bytes, 1) - .expect("POSITIONS is present") - .as_chunks::<4>(); - let delivered_points: Vec = coordinates - .iter() - .copied() - .map(f32::from_le_bytes) - .collect(); - assert_eq!( - delivered_points, - [4.0, -4.0, 20.0, -20.0, 6.0, -6.0, 5.0, -5.0, 21.0, -21.0] - ); - - let (row_ids, _) = section(&bytes, 2) - .expect("ROW_IDS is present") - .as_chunks::<4>(); - let delivered_rows: Vec = row_ids.iter().copied().map(u32::from_le_bytes).collect(); - assert_eq!(delivered_rows, [104, 120, 106, 105, 121]); - - // Point order 4, 20, 6, 5, 21. A monotone cursor answers [2, 3, 0, 0, 6] here. The count - // and the length are right and nothing panics, yet two points come back unmasked. - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - [0b010, 0b011, 0b100, 0b001, 0b110] - ); - - // The literal inversion sets delivery order fully descending in base position. - let inverted = [21_u32, 20, 6, 5, 4].map(|n| ViewRow::Base(BasePosition::from_u32(n))); - response.delivered = DeliveredSet::Positions(&inverted); - let bytes = response.encode(); - - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - [0b110, 0b011, 0b100, 0b001, 0b010] - ); - } - - #[test] - fn trailer_encodes_labels_and_icons() { - let positions = [Vec2::new(0.0, 0.0); 12]; - let rows: Vec> = (0..12).map(WireRow::pinned).collect(); - let ranges = [10_u32..12] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let mut response = minimal(&positions, &rows, &ranges); - response.trailer = Some(TileTrailer { - labels: &const { [Label::new("a"), Label::EMPTY] }, - icons: &const { [Icon::empty(), Icon::new("\u{fc}")] }, - }); - let bytes = response.encode(); - - // The HEAD echoes the trailer. - let head = section(&bytes, 0).expect("HEAD is present"); - assert_eq!(head[head.len() - 2..], [0x0A, 0xF5]); - - // The tail follows the last padded column: - // map(2): 0 ["a", null] | 1 [null, "ü"]. - let last = super::directory(&bytes, 2).1 as usize; - let tail = &bytes[last.next_multiple_of(8)..]; - assert_eq!( - tail, - [ - 0xA2, 0x00, 0x82, 0x61, 0x61, 0xF6, 0x01, 0x82, 0xF6, 0x62, 0xC3, 0xBC - ] - ); - } - - #[test] - fn zero_point_tile_is_present_empty() { - let positions: [Vec2; 0] = []; - let rows: [WireRow; 0] = []; - let ranges: [core::ops::Range; 0] = []; - - let mut response = minimal(&positions, &rows, &ranges); - response.head.runs = &[0]; - response.head.children = 0; - let bytes = response.encode(); - - let (start, end) = super::directory(&bytes, 1); - assert_eq!(start, end); - assert_ne!(start, 0, "zero points are present-empty, not absent"); - assert_eq!(super::directory(&bytes, 2), (start, end)); - } - - #[test] - #[should_panic(expected = "must count exactly the HEAD runs")] - fn disagreeing_runs_are_rejected() { - let positions = [Vec2::new(0.0, 0.0); 12]; - let rows: Vec> = (0..12).map(WireRow::pinned).collect(); - let ranges = [10_u32..12] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let mut response = minimal(&positions, &rows, &ranges); - response.head.runs = &[3]; - let _bytes = response.encode(); - } - - #[test] - #[should_panic(expected = "reserved zero")] - fn reserved_children_bits_are_rejected() { - let positions = [Vec2::new(0.0, 0.0); 12]; - let rows: Vec> = (0..12).map(WireRow::pinned).collect(); - let ranges = [10_u32..12] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let mut response = minimal(&positions, &rows, &ranges); - response.head.children = 16; - let _bytes = response.encode(); - } - - #[test] - #[should_panic(expected = "trailer labels must cover")] - fn short_trailers_are_rejected() { - let positions = [Vec2::new(0.0, 0.0); 12]; - let rows: Vec> = (0..12).map(WireRow::pinned).collect(); - let ranges = [10_u32..12] - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)); - - let mut response = minimal(&positions, &rows, &ranges); - response.trailer = Some(TileTrailer { - labels: &const { [Label::new("a")] }, - icons: &const { [Icon::empty(); 2] }, - }); - let _bytes = response.encode(); - } -} - -mod edges { - use alloc::borrow::Cow; - use std::sync::LazyLock; - - use hashql_core::id::IdSlice; - - use super::{EdgesResponse, EdgesTrailer, Label, Sha256Digest, identity_of, section}; - use crate::serve::{TableIndex, neighbourhood::EdgeColumns}; - - /// The three-edge columns behind the minimal response. - static EDGES: LazyLock = LazyLock::new(|| { - EdgeColumns::pinned([ - (4, 7, identity_of(0x11)), - (9, 2, identity_of(0x22)), - (4, 11, identity_of(0x33)), - ]) - }); - - /// A three-edge response without a trailer. - fn minimal() -> EdgesResponse<'static> { - EdgesResponse { - generation: Sha256Digest::from_bytes_unchecked([0xCD; 32]), - variant: 1, - complete: false, - edges: &EDGES, - trailer: None, - } - } - - #[test] - fn head_encodes_by_hand() { - let bytes = minimal().encode(); - assert_eq!(bytes[0..8], *b"SALTILEE"); - - // map(5): 0 bstr(32) | 1 uint 1 | 2 uint 3 | 3 false | - // 4 false. - let mut expected = vec![0xA5, 0x00, 0x58, 0x20]; - expected.extend_from_slice(&[0xCD; 32]); - expected.extend_from_slice(&[0x01, 0x01, 0x02, 0x03, 0x03, 0xF4, 0x04, 0xF4]); - assert_eq!(section(&bytes, 0).expect("HEAD is present"), expected); - } - - #[test] - fn columns_encode_little_endian() { - let bytes = minimal().encode(); - - let mut expected = Vec::new(); - for source in [4_u32, 9, 4] { - expected.extend_from_slice(&source.to_le_bytes()); - } - assert_eq!( - section(&bytes, 1).expect("EDGE_SOURCES is present"), - expected - ); - - // EDGE_IDS: raw 32-byte identity records, concatenated. - let mut expected = Vec::new(); - for id in [[0x11_u8; 32], [0x22; 32], [0x33; 32]] { - expected.extend_from_slice(&id); - } - assert_eq!(section(&bytes, 3).expect("EDGE_IDS is present"), expected); - } - - #[test] - fn trailer_carries_the_link_columns() { - let mut response = minimal(); - response.trailer = Some(EdgesTrailer { - type_table: IdSlice::from_raw(&const { [Cow::Borrowed("s"), Cow::Borrowed("t")] }), - link_labels: IdSlice::from_raw( - &const { [Label::new("a"), Label::EMPTY, Label::EMPTY] }, - ), - link_type_ids: IdSlice::from_raw( - &const { [Some(TableIndex::new(1)), Some(TableIndex::new(0)), None] }, - ), - }); - let bytes = response.encode(); - - let head = section(&bytes, 0).expect("HEAD is present"); - assert_eq!(head[head.len() - 2..], [0x04, 0xF5]); - - let last = super::directory(&bytes, 3).1 as usize; - let tail = &bytes[last.next_multiple_of(8)..]; - // map(3): 0 ["s", "t"] | 1 ["a", null, null] | - // 2 [1, 0, null]. - let expected = [ - 0xA3, 0x00, 0x82, 0x61, 0x73, 0x61, 0x74, 0x01, 0x83, 0x61, 0x61, 0xF6, 0xF6, 0x02, - 0x83, 0x01, 0x00, 0xF6, - ]; - assert_eq!(tail, expected); - } - - #[test] - fn zero_edge_response_is_present_empty() { - let empty = EdgeColumns::pinned([]); - let mut response = minimal(); - response.edges = ∅ - let bytes = response.encode(); - - for slot in 1..4 { - let (start, end) = super::directory(&bytes, slot); - assert_eq!(start, end); - assert_ne!(start, 0); - } - } - - #[test] - #[should_panic(expected = "link type ids must cover")] - fn short_trailers_are_rejected() { - let mut response = minimal(); - response.trailer = Some(EdgesTrailer { - type_table: IdSlice::from_raw(&[]), - link_labels: IdSlice::from_raw(&const { [Label::EMPTY; 3] }), - link_type_ids: IdSlice::from_raw(&[None]), - }); - let _bytes = response.encode(); - } -} - -mod locate { - use alloc::borrow::Cow; - use std::sync::LazyLock; - - use hashql_core::id::{Id as _, IdSlice}; - use type_system::ontology::id::VersionedUrl; - - use super::{ - BasePosition, DenseBitSlice, Label, LocateResponse, LocateTrailer, Membership, PropertyMap, - PropertyValue, Sha256Digest, TileCoordinate, Vec2, WireRow, dense_set, identity_of, - section, - }; - use crate::serve::{ - TableIndex, hydrate::EdgeSlot, neighbourhood::EdgeColumns, schedule::ViewRow, - }; - - /// The four-point base-order coordinate column behind the tests. - fn points() -> [Vec2; 4] { - [ - Vec2::new(0.0, 0.5), - Vec2::new(1.0, 1.5), - Vec2::new(2.0, 2.5), - Vec2::new(3.0, 3.5), - ] - } - - /// The two-edge link-type lists behind the minimal trailer: both empty. - static NO_TYPES: [Vec>; 2] = [Vec::new(), Vec::new()]; - - /// The two-edge completeness set behind the minimal trailer: nothing complete. - static NO_FLAGS: LazyLock>> = LazyLock::new(|| dense_set(2, &[])); - - /// The two-edge columns behind the minimal response. - static EDGES: LazyLock = LazyLock::new(|| { - EdgeColumns::pinned([(10, 12, identity_of(0x44)), (12, 13, identity_of(0x55))]) - }); - - /// A three-node, two-edge response with an all-null trailer. - /// - /// Delivered base positions 2, 0, 3 (source first) over a four-point base column. Locate is the - /// detail view, so every document includes the trailer, and the minimal document has empty - /// tables and null columns. - fn minimal(positions: &[Vec2]) -> LocateResponse<'_> { - LocateResponse { - generation: Sha256Digest::from_bytes_unchecked([0xAB; 32]), - variant: 0, - cell: TileCoordinate { z: 2, x: 1, y: 3 }, - complete: true, - entity_id: identity_of(0xEE), - type_ids_complete: false, - properties_complete: true, - delivered: IdSlice::from_raw( - &const { - [ - ViewRow::Base(BasePosition::from_u32(2)), - ViewRow::Base(BasePosition::from_u32(0)), - ViewRow::Base(BasePosition::from_u32(3)), - ] - }, - ), - arrivals: IdSlice::from_raw(&[]), - positions: IdSlice::from_raw(positions), - rows: IdSlice::from_raw( - &const { - [ - WireRow::pinned(10), - WireRow::pinned(11), - WireRow::pinned(12), - WireRow::pinned(13), - ] - }, - ), - masks: None, - edges: &EDGES, - trailer: LocateTrailer { - type_table: IdSlice::from_raw(&[]), - property_table: IdSlice::from_raw(&[]), - labels: IdSlice::from_raw(&const { [Label::EMPTY; 3] }), - type_ids: IdSlice::from_raw(&[None, None, None]), - properties: None, - link_labels: IdSlice::from_raw(&const { [Label::EMPTY; 2] }), - link_type_ids: IdSlice::from_raw(&NO_TYPES), - link_type_ids_complete: &NO_FLAGS, - link_properties: IdSlice::from_raw(&[None, None]), - link_properties_complete: &NO_FLAGS, - }, - } - } - - #[test] - fn head_encodes_by_hand() { - let positions = points(); - let bytes = minimal(&positions).encode(); - assert_eq!(bytes[0..8], *b"SALTILEL"); - - // map(10): 0 bstr(32) | 1 uint 0 | 2 uint 3 | 3 uint 2 | - // 4 [2, 1, 3] | 5 uint 2 | 6 true | 7 bstr(32) | 8 false | - // 9 true. - let mut expected = vec![0xAA, 0x00, 0x58, 0x20]; - expected.extend_from_slice(&[0xAB; 32]); - expected.extend_from_slice(&[ - 0x01, 0x00, 0x02, 0x03, 0x03, 0x02, 0x04, 0x83, 0x02, 0x01, 0x03, 0x05, 0x02, 0x06, - 0xF5, 0x07, 0x58, 0x20, - ]); - expected.extend_from_slice(&[0xEE; 32]); - expected.extend_from_slice(&[0x08, 0xF4, 0x09, 0xF5]); - assert_eq!(section(&bytes, 0).expect("HEAD is present"), expected); - } - - #[test] - fn columns_gather_in_delivered_order() { - let positions = points(); - let bytes = minimal(&positions).encode(); - - // POSITIONS: base positions 2, 0, 3 as xy pairs. - let mut expected = Vec::new(); - for point in [positions[2], positions[0], positions[3]] { - expected.extend_from_slice(&point.x().to_le_bytes()); - expected.extend_from_slice(&point.y().to_le_bytes()); - } - assert_eq!(section(&bytes, 1).expect("POSITIONS is present"), expected); - - // ROW_IDS: the row column read at 2, 0, 3. - let mut expected = Vec::new(); - for row in [12_u32, 10, 13] { - expected.extend_from_slice(&row.to_le_bytes()); - } - assert_eq!(section(&bytes, 2).expect("ROW_IDS is present"), expected); - - // No coloredTypeIds: TYPE_MASK is absent, not empty. - assert!(section(&bytes, 3).is_none()); - - // Endpoint columns, edge order. - for (slot, column) in [(4_usize, [10_u32, 12]), (5, [12, 13])] { - let mut expected = Vec::new(); - for value in column { - expected.extend_from_slice(&value.to_le_bytes()); - } - assert_eq!( - section(&bytes, slot).expect("edge columns are present"), - expected, - "slot {slot}", - ); - } - - // EDGE_IDS: raw 32-byte identity records, concatenated. - let mut expected = Vec::new(); - for id in [[0x44_u8; 32], [0x55; 32]] { - expected.extend_from_slice(&id); - } - assert_eq!(section(&bytes, 6).expect("EDGE_IDS is present"), expected); - } - - #[test] - fn masks_probe_the_delivered_list() { - let positions = points(); - // type 0: list members at base positions 0 and 2; type 1: - // dense bit at 3 over N = 4. - let list = [0_u32, 2].map(BasePosition::from_u32); - let dense = dense_set(4, &[3]); - let masks = [Membership::List(&list), Membership::Dense(&dense)]; - - let mut response = minimal(&positions); - response.masks = Some(&masks); - let bytes = response.encode(); - - // Stride ceil(2/8) = 1. Delivered order 2, 0, 3: - // type 0 | type 0 | type 1. - assert_eq!( - section(&bytes, 3).expect("TYPE_MASK is present"), - [0b01, 0b01, 0b10] - ); - } - - #[test] - fn trailer_encodes_by_hand() { - let positions = points(); - let lists = [vec![TableIndex::new(1), TableIndex::new(0)], Vec::new()]; - let source_map = PropertyMap::new_unchecked(vec![ - (TableIndex::new(0), PropertyValue::Text("x")), - (TableIndex::new(1), PropertyValue::Integer(-2)), - ]); - let link_map = PropertyMap::new_unchecked(vec![ - (TableIndex::new(0), PropertyValue::Boolean(true)), - (TableIndex::new(1), PropertyValue::Null), - ]); - let link_properties: [Option<&PropertyMap<'_>>; 2] = [Some(&link_map), None]; - // Slot 0's type list is complete and slot 1's property map is, over the two delivered - // edges. - let type_flags = dense_set::(2, &[0]); - let property_flags = dense_set::(2, &[1]); - let mut response = minimal(&positions); - response.trailer = LocateTrailer { - type_table: IdSlice::from_raw(&const { [Cow::Borrowed("s"), Cow::Borrowed("t")] }), - property_table: IdSlice::from_raw(&const { [Cow::Borrowed("a"), Cow::Borrowed("b")] }), - labels: IdSlice::from_raw(&const { [Label::new("n"), Label::EMPTY, Label::EMPTY] }), - type_ids: IdSlice::from_raw( - &const { [Some(TableIndex::new(1)), None, Some(TableIndex::new(0))] }, - ), - properties: Some(&source_map), - link_labels: IdSlice::from_raw(&const { [Label::new("l"), Label::EMPTY] }), - link_type_ids: IdSlice::from_raw(&lists), - link_type_ids_complete: &type_flags, - link_properties: IdSlice::from_raw(&link_properties), - link_properties_complete: &property_flags, - }; - let bytes = response.encode(); - - let last = super::directory(&bytes, 6).1 as usize; - let tail = &bytes[last.next_multiple_of(8)..]; - // map(10): 0 ["s", "t"] | 1 ["a", "b"] | 2 ["n", null x2] | - // 3 [1, null, 0] | 4 {0: "x", 1: -2} | 5 ["l", null] | - // 6 [[1, 0], []] | 7 bstr 0b01 in one 8-byte word | - // 8 [{0: true, 1: null}, null] | 9 bstr 0b10 in one 8-byte word. - let expected = [ - 0xAA, 0x00, 0x82, 0x61, 0x73, 0x61, 0x74, 0x01, 0x82, 0x61, 0x61, 0x61, 0x62, 0x02, - 0x83, 0x61, 0x6E, 0xF6, 0xF6, 0x03, 0x83, 0x01, 0xF6, 0x00, 0x04, 0xA2, 0x00, 0x61, - 0x78, 0x01, 0x21, 0x05, 0x82, 0x61, 0x6C, 0xF6, 0x06, 0x82, 0x82, 0x01, 0x00, 0x80, - 0x07, 0x48, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x08, 0x82, 0xA2, 0x00, - 0xF5, 0x01, 0xF6, 0xF6, 0x09, 0x48, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, - ]; - assert_eq!(tail, expected); - } - - #[test] - fn source_only_response_is_present_empty_on_edges() { - let positions = points(); - let no_edges = dense_set::(0, &[]); - let empty = EdgeColumns::pinned([]); - let mut response = minimal(&positions); - response.delivered = - IdSlice::from_raw(&const { [ViewRow::Base(BasePosition::from_u32(1))] }); - response.edges = ∅ - response.trailer.labels = IdSlice::from_raw(&const { [Label::EMPTY] }); - response.trailer.type_ids = IdSlice::from_raw(&[None]); - response.trailer.link_labels = IdSlice::from_raw(&[]); - response.trailer.link_type_ids = IdSlice::from_raw(&[]); - response.trailer.link_type_ids_complete = &no_edges; - response.trailer.link_properties = IdSlice::from_raw(&[]); - response.trailer.link_properties_complete = &no_edges; - let bytes = response.encode(); - - for slot in [4_usize, 5, 6] { - let (start, end) = super::directory(&bytes, slot); - assert_eq!(start, end, "slot {slot}"); - assert_ne!(start, 0, "slot {slot}"); - } - } - - #[test] - #[should_panic(expected = "labels must cover exactly the delivered nodes")] - fn short_trailers_are_rejected() { - let positions = points(); - let mut response = minimal(&positions); - response.trailer.labels = IdSlice::from_raw(&const { [Label::EMPTY] }); - let _bytes = response.encode(); - } - - #[test] - #[should_panic(expected = "type ids must cover exactly the delivered nodes")] - fn short_type_id_columns_are_rejected() { - let positions = points(); - let mut response = minimal(&positions); - response.trailer.type_ids = IdSlice::from_raw(&[None]); - let _bytes = response.encode(); - } - - #[test] - #[should_panic(expected = "type completeness must cover exactly the delivered edges")] - fn short_bitmask_columns_are_rejected() { - // One edge slot where the response delivers two. - let positions = points(); - let short = dense_set::(1, &[]); - let mut response = minimal(&positions); - response.trailer.link_type_ids_complete = &short; - let _bytes = response.encode(); - } - - #[test] - #[should_panic(expected = "property map keys must ascend")] - fn descending_property_keys_are_rejected() { - let _map = PropertyMap::new_unchecked(vec![ - (TableIndex::new(1), PropertyValue::Null), - (TableIndex::new(0), PropertyValue::Null), - ]); - } -} diff --git a/libs/@local/graph/atlas/src/salt/wire/tile.rs b/libs/@local/graph/atlas/src/salt/wire/tile.rs deleted file mode 100644 index b9e6430753d..00000000000 --- a/libs/@local/graph/atlas/src/salt/wire/tile.rs +++ /dev/null @@ -1,514 +0,0 @@ -//! The tile response: `HEAD`, columns, and trailer as one envelope. -//! -//! A tile document borrows the generation's base-order columns and names its delivered set in one -//! of [`DeliveredSet`]'s shapes. Contiguous base-position ranges in delivery order are what -//! both modes produce unmasked, where a non-root delta tile is its quad node's one run and the -//! delta root and every total tile are bucket-major run lists. A gathered position list in delivery -//! order is the shape a visibility mask leaves behind, and ranges with spliced arrivals are the -//! unmasked shape when a cohort interleaves. Encoding gathers the column entries and the -//! per-point type masks from the postings membership, then writes the `SALTILET` five-slot envelope -//! of `HEAD`, `POSITIONS`, `ROW_IDS`, `TYPE_MASK`, and the reserved `MASS` slot, which stays absent -//! until the product wants density. -//! -//! The document's consistency laws are producer contracts and panic when violated. The wire -//! reserves children bits beyond the low four, and a producer writes them zero. The range lengths -//! and the `HEAD`'s per-bucket runs must agree on the delivered count, and the trailer arrays must -//! cover exactly the delivered points. -#![expect( - clippy::little_endian_bytes, - reason = "column integers are pinned little-endian by the wire contract" -)] - -use core::ops::Range; - -use hashql_core::id::{Id as _, IdSlice}; - -use super::{Kind, Mode, cbor::CborWriter, envelope::EnvelopeWriter}; -use crate::{ - dataset::auxiliary::{Icon, Label}, - identity::{BasePosition, NodeRowId}, - integrity::Sha256Digest, - math::{Bounds2, Vec2}, - salt::postings::artifact::Membership, - serve::{ - WireRow, - schedule::{ArrivalIndex, ArrivalRow, Splice, ViewRow}, - }, -}; - -/// One tile response in writable form. -#[derive(Debug)] -pub(crate) struct TileResponse<'doc> { - /// The `HEAD` document, slot 0. - pub head: TileHead<'doc>, - /// The delivered set, in either producer shape. - pub delivered: DeliveredSet<'doc>, - /// The generation's wire-coordinate column, base order, in full. - pub positions: &'doc IdSlice, - /// The generation's row-id column (row by base position), in full. - pub rows: &'doc IdSlice>, - /// The entry cohort's arrivals, addressed by the delivered set's arrival rows. - /// - /// Empty when the view holds no admitted arrival. - pub arrivals: &'doc IdSlice, - /// Per-type membership for the request's `coloredTypeIds`, in request order. - /// - /// Bit `i` of every point's mask reads from `masks[i]`. `None` when the request carried no - /// ids: the `TYPE_MASK` slot is then absent rather than empty. - pub masks: Option<&'doc [Membership<'doc>]>, - /// The hydrated detail trailer. - /// - /// `Some` iff the request set `detail: "auxiliary"`. - pub trailer: Option>, -} - -impl TileResponse<'_> { - /// Encodes the response as one `SALTILET` envelope. - /// - /// # Panics - /// - /// This panics when range lengths and `HEAD` runs disagree on the delivered count, when a - /// producer sets a reserved children bit, or when the trailer arrays do not cover the delivered - /// points. - #[must_use] - pub(crate) fn encode(&self) -> Vec { - let delivered = self.delivered.count(); - let counted: u64 = self.head.runs.iter().map(|&count| u64::from(count)).sum(); - assert_eq!( - delivered, counted, - "the delivered set must count exactly the HEAD runs", - ); - - if let Some(trailer) = &self.trailer { - assert_eq!( - trailer.labels.len() as u64, - delivered, - "the trailer labels must cover exactly the delivered points", - ); - assert_eq!( - trailer.icons.len() as u64, - delivered, - "the trailer icons must cover exactly the delivered points", - ); - } - - let mut envelope = EnvelopeWriter::new(Kind::Tile, 5); - let points = usize::try_from(delivered).expect("delivered counts fit usize"); - envelope.reserve(points * 12); - envelope.slot(|buf| self.head.encode(buf, delivered, self.trailer.is_some())); - envelope.slot(|buf| self.write_positions(buf)); - envelope.slot(|buf| self.write_rows(buf)); - match self.masks { - Some(masks) => envelope.slot(|buf| self.write_masks(buf, masks)), - None => envelope.absent(), - } - envelope.absent(); - - match &self.trailer { - Some(trailer) => envelope.finish_with_trailer(|buf| trailer.encode(buf)), - None => envelope.finish(), - } - } - - /// Writes the `POSITIONS` column: f32 xy pairs, delivered order. - fn write_positions(&self, column: &mut Vec) { - for row in self.delivered { - let point = match row { - ViewRow::Base(position) => self.positions[position], - ViewRow::Arrival(index) => self.arrivals[index].position, - }; - column.extend_from_slice(&point.x().to_le_bytes()); - column.extend_from_slice(&point.y().to_le_bytes()); - } - } - - /// Writes the `ROW_IDS` column: u32 row ids, delivered order. - fn write_rows(&self, column: &mut Vec) { - for row in self.delivered { - let wire = match row { - ViewRow::Base(position) => self.rows[position], - ViewRow::Arrival(index) => self.arrivals[index].wire, - }; - column.extend_from_slice(&wire.get().to_le_bytes()); - } - } - - /// Assembles the `TYPE_MASK` column. - /// - /// One `ceil(n/8)`-byte mask per delivered point, bit `i` LSB-first when the point carries the - /// request's type `i`. - /// - /// Each membership contributes by a linear merge of its ascending positions against the - /// delivered set, never a per-point containment probe. A point matching no requested type keeps - /// the zero mask, and no sentinel exists. - /// - /// The merge needs the delivered set in ascending base-position order, which delivery order - /// does not supply. A range list numbers each point from its own range's start. For a gathered - /// list that does not already ascend, the merge walks a permutation of its point indices sorted - /// by position, built once and reused by every membership. A spliced list numbers its base - /// rows as a range list does and shifts each by the splices sitting before it, whose own bits - /// stay zero. Either way the bits for a point occupy its delivery index, so the column stays - /// in delivery order. - fn write_masks(&self, buf: &mut Vec, masks: &[Membership<'_>]) { - let delivered = - usize::try_from(self.delivered.count()).expect("delivered counts fit usize"); - let stride = masks.len().div_ceil(8); - let base = buf.len(); - buf.resize(base + delivered * stride, 0); - let column = &mut buf[base..]; - - match self.delivered { - DeliveredSet::Ranges(ranges) => { - for (bit, membership) in masks.iter().enumerate() { - let byte = bit >> 3; - let flag = 1_u8 << (bit & 7); - - let mut base = 0_usize; - for range in ranges { - for position in membership.positions_in(range.clone()) { - let point = base + (position.as_usize() - range.start.as_usize()); - column[point * stride + byte] |= flag; - } - base += range.end.as_usize() - range.start.as_usize(); - } - } - } - DeliveredSet::Spliced { ranges, splices } => { - for (bit, membership) in masks.iter().enumerate() { - let byte = bit >> 3; - let flag = 1_u8 << (bit & 7); - - // Splice `cursor` counts the arrivals delivered before base ordinal - // `base + offset`: splice `m` precedes it exactly when `at - m` does, and - // both sides of that bound ascend through the walk. - let mut base = 0_usize; - let mut cursor = 0_usize; - for range in ranges { - for position in membership.positions_in(range.clone()) { - let ordinal = base + (position.as_usize() - range.start.as_usize()); - while cursor < splices.len() - && splices[cursor].at as usize - cursor <= ordinal - { - cursor += 1; - } - let point = ordinal + cursor; - column[point * stride + byte] |= flag; - } - base += range.end.as_usize() - range.start.as_usize(); - } - } - } - DeliveredSet::Positions(list) => { - // Only base rows carry postings memberships. An arrival has no postings row, - // so its bits stay zero, the mask's own answer for an id the generation never - // tabulated. - let mut base_points: Vec<(usize, BasePosition)> = list - .iter() - .enumerate() - .filter_map(|(point, row)| match *row { - ViewRow::Base(position) => Some((point, position)), - ViewRow::Arrival(_) => None, - }) - .collect(); - if base_points.is_empty() { - return; - } - base_points.sort_unstable_by_key(|&(_, position)| position); - - let lowest = base_points[0].1; - let highest = base_points[base_points.len() - 1].1; - - for (bit, membership) in masks.iter().enumerate() { - let byte = bit >> 3; - let flag = 1_u8 << (bit & 7); - - let mut rank = 0_usize; - let past_highest = BasePosition::from_u64(highest.as_u64() + 1); - for position in membership.positions_in(lowest..past_highest) { - while rank < base_points.len() && base_points[rank].1 < position { - rank += 1; - } - if rank == base_points.len() { - break; - } - let (point, held) = base_points[rank]; - if held == position { - column[point * stride + byte] |= flag; - } - } - } - } - } - } -} - -/// One delivered point set, in one of the shapes a producer holds. -#[derive(Debug, Copy, Clone)] -pub(crate) enum DeliveredSet<'doc> { - /// Contiguous base-position ranges in delivery order. - /// - /// The unmasked gather. Zero-length ranges are legal and deliver nothing. - Ranges(&'doc [Range]), - /// Gathered view rows in delivery order: the masked gather, visibility already applied. - /// - /// Delivery order is the producer's own, and nothing in the encoder depends on it relating - /// to base order. - Positions(&'doc [ViewRow]), - /// Contiguous base-position ranges with arrivals spliced among them. - /// - /// The unmasked gather when a cohort interleaves. Each splice's delivery index counts rows - /// of both kinds, so the walk emits the named arrival whenever its output index reaches a - /// splice and a base position otherwise. - Spliced { - /// The delivered base-position ranges, in delivery order. - ranges: &'doc [Range], - /// The spliced arrivals, ascending by delivery index. - splices: &'doc [Splice], - }, -} - -impl DeliveredSet<'_> { - /// Counts the delivered points. - pub(crate) fn count(self) -> u64 { - match self { - Self::Ranges(ranges) => ranges - .iter() - .map(|range| range.end.as_u64() - range.start.as_u64()) - .sum(), - Self::Positions(list) => list.len() as u64, - Self::Spliced { ranges, splices } => { - let bases: u64 = ranges - .iter() - .map(|range| range.end.as_u64() - range.start.as_u64()) - .sum(); - - bases + splices.len() as u64 - } - } - } -} - -/// An iterator over a delivered set's rows, in delivery order. -/// -/// A range-shaped set's positions arrive as base rows, and a spliced set's arrivals arrive at -/// their delivery indexes among the base rows. -#[derive(Debug)] -pub(crate) struct DeliveredRows<'doc> { - /// The remaining ranges of a range-shaped set. - ranges: core::slice::Iter<'doc, Range>, - /// The positions remaining in the current range. - current: Range, - /// The splices not yet reached, ascending by delivery index. - splices: core::iter::Peekable>, - /// The delivery index of the next row. - output: u32, - /// The remaining rows of a list-shaped set. - list: core::slice::Iter<'doc, ViewRow>, -} - -impl Iterator for DeliveredRows<'_> { - type Item = ViewRow; - - fn next(&mut self) -> Option { - loop { - let output = self.output; - if let Some(splice) = self.splices.next_if(|splice| splice.at == output) { - self.output += 1; - return Some(ViewRow::Arrival(splice.arrival)); - } - - if let Some(position) = self.current.next() { - self.output += 1; - return Some(ViewRow::Base(position)); - } - - match self.ranges.next() { - Some(range) => self.current = range.clone(), - None => return self.list.next().copied(), - } - } - } -} - -impl<'doc> IntoIterator for DeliveredSet<'doc> { - type IntoIter = DeliveredRows<'doc>; - type Item = ViewRow; - - fn into_iter(self) -> DeliveredRows<'doc> { - let (ranges, splices, list): (&[Range], &[Splice], &[ViewRow]) = match self { - Self::Ranges(ranges) => (ranges, &[], &[]), - Self::Positions(list) => (&[], &[], list), - Self::Spliced { ranges, splices } => (ranges, splices, &[]), - }; - - DeliveredRows { - ranges: ranges.iter(), - current: BasePosition::MIN..BasePosition::MIN, - splices: splices.iter().peekable(), - output: 0, - list: list.iter(), - } - } -} - -/// The tile `HEAD` document, slot 0. -/// -/// Keys 5 and 11 stay reserved: no response emits either. -#[derive(Debug)] -pub(crate) struct TileHead<'doc> { - /// Key 0: the generation identity, echoing the route. - pub generation: Sha256Digest, - /// Key 1: the variant index, echoing the route. - pub variant: u64, - /// Key 2: the tile coordinate, echoing the route. - pub coordinate: TileCoordinate, - /// Key 3: the delivery mode, echoing the request. - pub mode: Mode, - /// Key 6: the first bucket of the runs array. - pub first_bucket: u8, - /// Key 7: per-bucket delivered counts from the first bucket up. - /// - /// Zero-length entries keep their positional slot. - pub runs: &'doc [u32], - /// Key 8: post-intersection set metadata. Required on the root tile, permitted everywhere. - pub global: Option, - /// Key 9: the occupied-child bitmask. - /// - /// Bit `i` = Morton child `i` holds a point below this zoom's cut. The wire reserves the bits - /// beyond the low four, and a producer writes them zero. - pub children: u8, -} - -impl TileHead<'_> { - /// Encodes the `HEAD` map. - /// - /// Key 4 (`delivered`) and key 10 (`trailer`) come from the response rather than from a stored - /// field. - fn encode(&self, buf: &mut Vec, delivered: u64, trailer: bool) { - assert!( - self.children < 16, - "children bits beyond the low four are reserved zero", - ); - - let mut cbor = CborWriter::over(buf); - cbor.map(9 + u64::from(self.global.is_some())); - - cbor.uint(0); - cbor.bytes(&self.generation.to_bytes()); - cbor.uint(1); - cbor.uint(self.variant); - cbor.uint(2); - cbor.array(3); - cbor.uint(u64::from(self.coordinate.z)); - cbor.uint(u64::from(self.coordinate.x)); - cbor.uint(u64::from(self.coordinate.y)); - cbor.uint(3); - cbor.uint(self.mode.code()); - cbor.uint(4); - cbor.uint(delivered); - cbor.uint(6); - cbor.uint(u64::from(self.first_bucket)); - cbor.uint(7); - cbor.array(self.runs.len() as u64); - for &count in self.runs { - cbor.uint(u64::from(count)); - } - if let Some(global) = &self.global { - cbor.uint(8); - global.encode(&mut cbor); - } - cbor.uint(9); - cbor.uint(u64::from(self.children)); - cbor.uint(10); - cbor.boolean(trailer); - } -} - -/// A tile address, the route's `z/x/y` echoed as `HEAD` key 2. -#[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Deserialize, schemars::JsonSchema)] -pub(crate) struct TileCoordinate { - /// The zoom, a subdivision depth. - pub z: u8, - /// The cell's x index on the `2^z` grid. - pub x: u32, - /// The cell's y index on the `2^z` grid. - pub y: u32, -} - -/// `HEAD` key 8: metadata of the entire post-intersection visible set. -#[derive(Debug, Copy, Clone, PartialEq)] -pub(crate) struct GlobalHead { - /// Entry 0: points visible at the current zoom. - pub visible: u64, - /// Entry 1. - /// - /// The tight wire-frame extent of the entire visible set, absent iff that set is empty. - /// Emitted as `[minX, minY, maxX, maxY]`. - pub bounds: Option, - /// Entry 2: the deepest occupied bucket of the visible set. - pub min_resolution: u64, -} - -impl GlobalHead { - /// Encodes the global map as the value of `HEAD` key 8. - fn encode(&self, cbor: &mut CborWriter<'_>) { - cbor.map(2 + u64::from(self.bounds.is_some())); - - cbor.uint(0); - cbor.uint(self.visible); - if let Some(bounds) = &self.bounds { - cbor.uint(1); - cbor.array(4); - cbor.f32(bounds.min().x()); - cbor.f32(bounds.min().y()); - cbor.f32(bounds.max().x()); - cbor.f32(bounds.max().y()); - } - cbor.uint(2); - cbor.uint(self.min_resolution); - } -} - -/// The tile detail trailer. -/// -/// Labels and icons from the generation's own payloads - a placed arrival's label from its -/// placement's captured display - delivered order, `null` marking a row whose display records -/// no text. -#[derive(Debug)] -pub(crate) struct TileTrailer<'trailer> { - /// Trailer key 0. - pub labels: &'trailer [&'trailer Label], - /// Trailer key 1. - pub icons: &'trailer [&'trailer Icon], -} - -impl TileTrailer<'_> { - /// Encodes the trailer tail as one self-delimiting CBOR map. - fn encode(&self, buf: &mut Vec) { - let mut cbor = CborWriter::over(buf); - cbor.map(2); - - cbor.uint(0); - encode_details(&mut cbor, self.labels.iter()); - - cbor.uint(1); - encode_details(&mut cbor, self.icons.iter()); - } -} - -/// Emits one detail array: text entries, `null` for an empty entry. -pub(super) fn encode_details( - cbor: &mut CborWriter<'_>, - entries: impl ExactSizeIterator>, -) { - cbor.array(entries.len() as u64); - for entry in entries { - let entry = entry.as_ref(); - - if entry.is_empty() { - cbor.null(); - } else { - cbor.text(entry); - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/authorization.rs b/libs/@local/graph/atlas/src/serve/authorization.rs deleted file mode 100644 index 30d6a6cc0b8..00000000000 --- a/libs/@local/graph/atlas/src/serve/authorization.rs +++ /dev/null @@ -1,760 +0,0 @@ -//! The authority token, which this server seals and a client carries across requests. -//! -//! A token seals the [`Scope`] that names an authorized view. The scope holds the actor, the filter -//! digest (the visibility proof's identity, resolved over the filter at bootstrap), and the view -//! state derived for that proof, the delivery-cut offset `k`. A per-generation key encrypts the -//! plaintext, and the tag proves this server issued it. The server keeps no token state: a renewal -//! reads the sealed state out of the presented token. A refresh renews authority while the view -//! stays fixed. -//! -//! The filter travels as its digest. The client holds the filter document itself and re-presents it -//! when a server-side entry has expired. The sealed digest is the check that the presented document -//! is this view's filter. An actor with more than one active filter holds one token per filter, and -//! the digests tell them apart. A filter binds at the manifest, where the token renews, and a -//! pinned run accumulates delivered state under one filter for its whole lifetime. -//! -//! # The envelope -//! -//! `header | ciphertext | trailer`, [`TOKEN_BYTES`] wide, with every field at a fixed offset: -//! [`SealedAuthority`] is the envelope as a type, and a blob resolves into one by a zerocopy cast. -//! The header is [`AuthorityHeader`] in the clear, the ciphertext seals [`SealedState`] - the -//! scope and the authority's delta epoch, each its own byte-level form - and the trailer is -//! Poly1305's tag. -//! -//! The associated data is the header's own bytes, the identical form on both sides. The clear -//! header stays as issued, because a rewritten `issued_at` invalidates the tag. -//! -//! # The key -//! -//! The key comes from `HKDF-SHA256` over the server secret, with the generation digest as the salt -//! and the fixed label `atlas.authorization.v1` as the expansion label. RFC 5869 admits a public -//! and predictable salt, and this one separates generations cryptographically: a token opens only -//! under the generation that sealed it. The secret arrives as [`SecretHexBytes`], which fixes -//! its width by type. The derivation runs once, when this module constructs the authority, and the -//! authority keeps only the key, never the secret. -//! -//! # The epoch -//! -//! The sealed state carries the delta epoch the authority held at issuance, absent for a process -//! serving without a delta consumer. Every open compares the sealed value against the held one -//! and refuses any other, the renewal read included, because slot assignment is process-local: a -//! token issued beside one delta register must not authorize reads over another. The refusal is -//! the same uniform answer as every other cause, and the client's remedy is the same fresh -//! manifest, which issues under the held epoch. A process serving with no delta consumer holds the -//! absent form, and its tokens survive restarts. -//! -//! # The nonce -//! -//! Each issuance samples the nonce from an injected [`TryCryptoRng`] behind a lock. Issuing locks -//! for the draw alone and seals outside it, and opening never touches the generator. A nonce is -//! unique per key, because reuse repeats the keystream and the Poly1305 one-time key. `XChaCha20`'s -//! 192-bit width makes sampling a safe way to reach that uniqueness. Collision probability stays -//! below 2⁻³² until about 2⁸⁰ issued tokens, and the same bound covers a fleet of replicas that -//! derive one key per generation from shared configuration. -#![expect( - clippy::empty_enums, - reason = "zerocopy's TryFromBytes derive expands to an empty enum for the discriminant check, \ - which is the validation this type exists for" -)] -use core::{ops::Deref, time::Duration}; -use std::{sync::nonpoison::Mutex, time::SystemTime}; - -use chacha20poly1305::{ - AeadCore, KeyInit as _, KeySizeUser, Tag, XChaCha20Poly1305, XNonce, - aead::{AeadInPlace as _, generic_array::typenum::Unsigned}, -}; -use hkdf::Hkdf; -use rand::{TryCryptoRng, TryRng as _}; -use sha2::Sha256; -use type_system::principal::actor::{ActorEntityUuid, ActorId, AiId, MachineId, UserId}; -use uuid::Uuid; -use zerocopy::{IntoBytes as _, LE, TryFromBytes as _, U64}; - -use super::{CutOffset, cache::scope::FilterDigest, delta::DeltaEpoch}; -use crate::{file::generation::GenerationId, integrity::SecretHexBytes}; - -/// The HKDF expansion label, versioned in place for the one value this module seals. -const LABEL: &[u8] = b"atlas.authorization.v1"; - -/// The nonce width: the cipher's own. -const NONCE_BYTES: usize = <::NonceSize as Unsigned>::USIZE; - -/// The sealing key's width: the cipher's own. -const KEY_BYTES: usize = <::KeySize as Unsigned>::USIZE; - -/// The tag width: the cipher's own. -const TAG_BYTES: usize = <::TagSize as Unsigned>::USIZE; - -/// One refused token, by cause. -/// -/// The causes are server-side diagnostics. Every variant answers the client with the same uniform -/// refusal. A caller learns that it must re-manifest and nothing about why. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum AuthorityError { - /// The blob is not this format. - Envelope, - /// The tag rejected the ciphertext under the header it arrived with. - Authentication, - /// The issue time is outside the acceptance window. - /// - /// Older than the hard window, or dated in the future. - Stale, - /// The token names an actor other than its presenter. - /// - /// The tag proves the server created the token, not that the presenter is its subject. Without - /// this refusal a leaked token would grant any authenticated actor the subject's scope. - Actor, - /// The token seals a delta epoch other than the held one. - /// - /// Slot assignment is process-local, so authority issued beside one delta register must not - /// reach another. - Epoch, -} - -impl core::fmt::Display for AuthorityError { - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - fmt.write_str(match self { - Self::Envelope => "the authority token's envelope is malformed", - Self::Authentication => "the authority token failed authentication", - Self::Stale => "the authority token's issue time is outside the acceptance window", - Self::Actor => "the authority token names an actor other than its presenter", - Self::Epoch => "the authority token seals a delta epoch other than the held one", - }) - } -} - -impl core::error::Error for AuthorityError {} - -/// The envelope's format version. -/// -/// Parsing admits exactly the layout this module writes. Increment on any layout change. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(u8)] -enum MessageVersion { - V1 = 1, -} - -/// The token envelope's clear header. -/// -/// The format version, the issue time, and the nonce. The tag authenticates them verbatim, which -/// fixes their values at issuance. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(C)] -struct AuthorityHeader { - version: MessageVersion, - /// The issue time as whole seconds since the Unix epoch. - /// - /// The wall clock narrows to seconds at this field. Every signature in this module speaks - /// [`SystemTime`]. Clock agreement between the process that issues and the process that opens - /// bounds the field's accuracy, and the acceptance window it feeds spans minutes. Truncation - /// reads earlier than the instant it records, and a token expires marginally early. - issued_at: U64, - nonce: [u8; NONCE_BYTES], -} - -/// The token envelope's trailer, the AEAD's tag over the ciphertext and the clear header. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(C)] -struct AuthorityTrailer { - tag: [u8; TAG_BYTES], -} - -/// The byte-level form of an [`ActorEntityUuid`]. -#[derive( - Debug, - Copy, - Clone, - zerocopy::ByteEq, - zerocopy::ByteHash, - zerocopy::IntoBytes, - zerocopy::FromBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, -)] -#[repr(transparent)] -pub(crate) struct ArchivedActorEntityUuid([u8; 16]); - -impl From for ArchivedActorEntityUuid { - #[inline] - fn from(actor: ActorEntityUuid) -> Self { - Self(Uuid::from(actor).into_bytes()) - } -} - -impl Deref for ArchivedActorEntityUuid { - type Target = ActorEntityUuid; - - #[inline] - fn deref(&self) -> &Self::Target { - const { - assert!(size_of::() == size_of::()); - assert!(align_of::() == align_of::()); - } - - let ptr = &raw const *self; - // SAFETY: `Self` is `repr(transparent)` over `[u8; 16]`, and the target chain - // `ActorEntityUuid(EntityUuid)`, `EntityUuid(Uuid)`, `Uuid([u8; 16])` is - // `repr(transparent)` at every link. - unsafe { &*ptr.cast::() } - } -} - -/// A sealed actor's kind. -/// -/// The discriminant half of [`ArchivedActorId`], one validated byte: parsing refuses every value -/// outside the principal kinds. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::TryFromBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, -)] -#[repr(u8)] -pub(crate) enum ArchivedActorType { - User, - Machine, - Ai, -} - -/// A sealed actor identity holding the kind beside the uuid, the byte-level form of an -/// [`ActorId`]. -/// -/// The `From` conversions are the only writers, so equality over both fields is exact. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::TryFromBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, -)] -#[repr(C)] -pub(crate) struct ArchivedActorId { - pub r#type: ArchivedActorType, - pub id: ArchivedActorEntityUuid, -} - -impl From for ArchivedActorId { - fn from(value: ActorId) -> Self { - match value { - ActorId::User(uuid) => Self { - r#type: ArchivedActorType::User, - id: ArchivedActorEntityUuid::from(ActorEntityUuid::new(uuid)), - }, - ActorId::Machine(uuid) => Self { - r#type: ArchivedActorType::Machine, - id: ArchivedActorEntityUuid::from(ActorEntityUuid::new(uuid)), - }, - ActorId::Ai(uuid) => Self { - r#type: ArchivedActorType::Ai, - id: ArchivedActorEntityUuid::from(ActorEntityUuid::new(uuid)), - }, - } - } -} - -impl From for ActorId { - fn from(value: ArchivedActorId) -> Self { - match value.r#type { - ArchivedActorType::User => Self::User(UserId::new(*value.id)), - ArchivedActorType::Machine => Self::Machine(MachineId::new(*value.id)), - ArchivedActorType::Ai => Self::Ai(AiId::new(*value.id)), - } - } -} - -/// A scope's request filter, by identity. -/// -/// The discriminant is the presence and the payload is the digest, one validated field: parsing -/// admits the two written forms and a tampered discriminant refuses as -/// [`AuthorityError::Envelope`]. The absent form zeroes its payload bytes, and a filter's presence -/// never shows in the envelope's length. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(u8)] -pub(crate) enum ScopeFilter { - Absent([u8; 32]), - Present(FilterDigest), -} - -impl ScopeFilter { - /// Returns the filter's digest, absent when the scope names no filter. - pub(crate) const fn digest(self) -> Option { - match self { - Self::Present(digest) => Some(digest), - Self::Absent(_) => None, - } - } -} - -impl From> for ScopeFilter { - fn from(filter: Option) -> Self { - filter.map_or(Self::Absent([0; 32]), Self::Present) - } -} - -/// A sealed delta epoch, by presence. -/// -/// The discriminant is the presence and the payload is the epoch, one validated field: parsing -/// admits the two written forms and a tampered discriminant refuses as -/// [`AuthorityError::Envelope`]. The absent form zeroes its payload bytes and names a token issued -/// with no delta consumer, so equality between two absent values is what lets those tokens survive -/// a restart. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(u8)] -enum ScopeEpoch { - Absent([u8; 16]), - Present(DeltaEpoch), -} - -impl From> for ScopeEpoch { - fn from(epoch: Option) -> Self { - epoch.map_or(Self::Absent([0; 16]), Self::Present) - } -} - -/// A view's scope as the token carries it, before [`bind`](Self::bind) matches it to a presenter. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(transparent)] -struct UnboundScope(Scope); - -impl UnboundScope { - /// Binds the scope to its presenter, refusing an actor the token does not name. - #[expect( - clippy::missing_const_for_fn, - reason = "the derived `PartialEq` behind `!=` is not const-callable" - )] - fn bind(self, actor: ActorId) -> Result { - if self.0.actor != actor.into() { - return Err(AuthorityError::Actor); - } - - Ok(self.0) - } -} - -/// One view's sealed identity and state. -/// -/// The actor and filter digest name the visibility proof the view answers under. `k` is the -/// delivery depth the session serves at, resolved at its bootstrap over the occupancy then in -/// force. A renewal carries `k` forward: a session keeps one delivery depth rather than -/// re-optimizing it per request. A re-bind of the filter digest keeps `k` unless the new view -/// resolves coarser, and a coarser view clamps it down. The carried value is the caller's own -/// earlier resolution, and delivery depth reflects that caller's own session history and never -/// another actor's rows. -/// -/// The scope is its own byte-level form, with every field a zerocopy type, so issuance seals it -/// verbatim and an open reads it in place. The filter discriminant is the one validated byte. Every -/// other pattern is a valid value, and the tag already vouched for it. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] -#[repr(C)] -pub(crate) struct Scope { - /// The one actor allowed to present this view. - pub actor: ArchivedActorId, - /// The digest of the filter the view's visibility proof resolved over, absent when the view - /// has no filter. - pub filter: ScopeFilter, - /// The view's delivery-cut offset. - pub k: CutOffset, -} - -impl Scope { - /// Binds one view's identity and state. - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - pub(crate) fn new(actor: ActorId, filter: Option, k: CutOffset) -> Self { - Self { - actor: actor.into(), - filter: ScopeFilter::from(filter), - k, - } - } -} - -/// The caller's scope and the authority's delta epoch, sealed as one plaintext. -/// -/// Its own byte-level form exactly as [`Scope`] is. The epoch is the authority's rather than the -/// caller's: issuance stamps the held value and the open refuses any other. No caller can seal a -/// scope under an epoch the process does not hold. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::IntoBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(C)] -struct SealedState { - scope: UnboundScope, - epoch: ScopeEpoch, -} - -/// One sealed token, the envelope as a type, read in place. -/// -/// Every field lies at a fixed offset: a blob of [`TOKEN_BYTES`] resolves into header, ciphertext, -/// and trailer in one zerocopy cast. The cast validates the format version. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, - zerocopy::TryFromBytes, -)] -#[repr(C)] -struct SealedAuthority { - header: AuthorityHeader, - ciphertext: [u8; size_of::()], - trailer: AuthorityTrailer, -} - -impl SealedAuthority { - const SIZE: usize = size_of::(); -} - -/// The token envelope's width, covering the clear header, the sealed state, and the tag. -/// -/// The value derives from the envelope type itself and moves when the layout does. -pub(crate) const TOKEN_BYTES: usize = SealedAuthority::SIZE; - -/// Issues and opens the authority tokens of one generation. -/// -/// One value holds the whole judgment context. The generation's sealing key comes from one -/// derivation at construction, and the acceptance window bounds a token's age. The held delta -/// epoch binds a token to the register lifetime whose process issued it, and the entropy source -/// stays behind its own lock, held for the nonce draw alone. Opening never contends with issuing. -/// A token opens under the authority whose generation sealed it, under the epoch it -/// holds, for the actor it names, and only while its issue time lies inside the window. -#[derive(Debug)] -pub(crate) struct TokenAuthority { - /// This generation's sealing key. - /// - /// Derived once, at construction, with the generation digest as the derivation's salt. - key: SecretHexBytes, - /// The acceptance window. - /// - /// A token older than this at open refuses as stale. - hard: Duration, - /// The delta epoch every issuance stamps and every open requires. - epoch: ScopeEpoch, - rng: Mutex, -} - -impl TokenAuthority { - /// Builds the authority of one generation. - /// - /// The key derives from `secret` with the generation digest as its salt. A token stays - /// acceptable for `hard` after its issue time, nonces come from `rng`, and `epoch` is the - /// serving process's delta epoch, [`None`] when no delta consumer runs. - pub(crate) fn new( - generation: GenerationId, - secret: &SecretHexBytes, - hard: Duration, - epoch: Option, - rng: R, - ) -> Self { - let salt = generation.digest().to_bytes(); - let mut key = SecretHexBytes::zeroed(); - - Hkdf::::new(Some(&salt), secret.as_bytes()) - .expand(LABEL, key.as_mut()) - .expect("the cipher's key size stays within HKDF-SHA256's expansion bound"); - - Self { - key, - hard, - epoch: ScopeEpoch::from(epoch), - rng: Mutex::new(rng), - } - } - - /// Creates the token naming `scope`, issued at `now`, stamped with the held delta epoch. - /// - /// `now` is wall-clock time, because whichever process opens the token judges its age, and - /// [`SystemTime`] is the clock whose value still means the same in another process. - /// - /// # Errors - /// - /// Returns the generator's error when drawing the nonce fails: entropy failure refuses issuance - /// rather than sealing under a predictable nonce. - pub(crate) fn issue( - &self, - scope: Scope, - now: SystemTime, - ) -> Result<[u8; SealedAuthority::SIZE], R::Error> - where - R: TryCryptoRng, - { - let sealed = SealedState { - scope: UnboundScope(scope), - epoch: self.epoch, - }; - - let mut nonce = [0_u8; NONCE_BYTES]; - self.rng.lock().try_fill_bytes(&mut nonce)?; - - let header = AuthorityHeader { - version: MessageVersion::V1, - issued_at: U64::new( - now.saturating_duration_since(SystemTime::UNIX_EPOCH) - .as_secs(), - ), - nonce, - }; - - let mut blob = [0_u8; SealedAuthority::SIZE]; - header - .write_to_prefix(&mut blob) - .expect("the envelope begins with its header"); - sealed - .write_to_prefix(&mut blob[size_of::()..]) - .expect("the envelope seals the state past its header"); - - let sealed_tag = XChaCha20Poly1305::new(self.key.as_bytes().into()) - .encrypt_in_place_detached( - XNonce::from_slice(&nonce), - header.as_bytes(), - &mut blob[size_of::() - ..size_of::() + size_of::()], - ) - .unwrap_or_else(|_error| { - unreachable!("XChaCha20-Poly1305 encryption is infallible for in-memory payloads") - }); - - sealed_tag - .write_to_suffix(&mut blob) - .expect("the envelope ends in its tag"); - - Ok(blob) - } - - /// Opens `blob` as presented by `actor` at `now`, returning the view state it seals. - /// - /// Refuses a token sealed under any delta epoch other than the held one, one whose issue - /// time is older than the acceptance window at `now`, one dated after `now`, and one naming - /// an actor other than `actor`. The open judges the tag first: a rewritten issue time refuses - /// as [`AuthorityError::Authentication`], and only an authentic token reaches the epoch, the - /// window, and the actor comparison. - /// - /// # Errors - /// - /// [`AuthorityError::Envelope`] for a blob that is not this format, - /// [`AuthorityError::Authentication`] when the tag refuses, [`AuthorityError::Epoch`] for a - /// sealed delta epoch other than the held one, [`AuthorityError::Stale`] outside the window, - /// and [`AuthorityError::Actor`] for a presenter the token does not name. - pub(crate) fn open( - &self, - blob: &[u8; TOKEN_BYTES], - actor: ActorId, - now: SystemTime, - ) -> Result { - let (issued_at, sealed) = self.unseal(blob)?; - let scope = self.alive(sealed)?; - - if issued_at > now || now.saturating_duration_since(issued_at) >= self.hard { - return Err(AuthorityError::Stale); - } - - scope.bind(actor) - } - - /// Reads the view state a presented token carries, for a renewal. - /// - /// This read does not judge the acceptance window. An expired token is no longer authority, yet - /// it remains authentic evidence of the view state a past issuance sealed, and re-sealing that - /// state into a fresh token keeps a view stable across a refresh. The tag, the epoch, and the - /// actor still bind, and the leniency reaches no further than the renewal. This read - /// authenticates a scope and carries it forward, while every data request under the fresh - /// token resolves that scope through the visibility cache, whose own hard window bounds how - /// old a resolution may answer. The epoch binds here exactly because this is the renewal: - /// view state accumulated beside a dead register must not carry into a token issued under - /// the live one. - /// - /// # Errors - /// - /// [`AuthorityError::Envelope`] for a blob that is not this format, - /// [`AuthorityError::Authentication`] when the tag refuses, [`AuthorityError::Epoch`] for a - /// sealed delta epoch other than the held one, and [`AuthorityError::Actor`] for a presenter - /// the token does not name. - pub(crate) fn continuity( - &self, - blob: &[u8; TOKEN_BYTES], - actor: ActorId, - ) -> Result { - let (_issued_at, sealed) = self.unseal(blob)?; - let scope = self.alive(sealed)?; - - scope.bind(actor) - } - - /// Parses and authenticates one envelope: the zerocopy cast and the tag, nothing judged. - fn unseal( - &self, - blob: &[u8; TOKEN_BYTES], - ) -> Result<(SystemTime, SealedState), AuthorityError> { - let sealed = - SealedAuthority::try_ref_from_bytes(blob).map_err(|_error| AuthorityError::Envelope)?; - - let mut plaintext = sealed.ciphertext; - XChaCha20Poly1305::new(self.key.as_bytes().into()) - .decrypt_in_place_detached( - &XNonce::from(sealed.header.nonce), - sealed.header.as_bytes(), - &mut plaintext, - &Tag::from(sealed.trailer.tag), - ) - .map_err(|_error| AuthorityError::Authentication)?; - - let state = SealedState::try_read_from_bytes(&plaintext) - .map_err(|_error| AuthorityError::Envelope)?; - - Ok(( - SystemTime::UNIX_EPOCH + Duration::from_secs(sealed.header.issued_at.get()), - state, - )) - } - - /// Resolves the sealed scope, refusing a delta epoch other than the held one. - #[expect( - clippy::missing_const_for_fn, - reason = "the derived `PartialEq` behind `!=` is not const-callable" - )] - fn alive(&self, sealed: SealedState) -> Result { - if sealed.epoch != self.epoch { - return Err(AuthorityError::Epoch); - } - - Ok(sealed.scope) - } -} - -#[cfg(test)] -mod tests { - const ACTOR_UUID_BYTES: [u8; 16] = [ - 0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x10, 0x32, 0x54, 0x76, 0x98, 0xBA, 0xDC, - 0xFE, - ]; - - /// The tests the `miri` nextest profile selects. - /// - /// The test here derefs an archived actor identifier to the identity it wraps. - /// The profile selects by module path: moving a test in or out of this module is the whole - /// edit. - mod miri { - use uuid::Uuid; - use zerocopy::FromBytes as _; - - use super::ACTOR_UUID_BYTES; - use crate::serve::authorization::ArchivedActorEntityUuid; - - /// The archived actor identity derefs to the plain identity the same bytes denote. - #[test] - fn archived_actor_entity_uuid_derefs_to_the_same_identity() { - let archived = ArchivedActorEntityUuid::read_from_bytes(&ACTOR_UUID_BYTES) - .expect("should read an archived actor uuid from any 16 bytes"); - - assert_eq!(Uuid::from(*archived), Uuid::from_bytes(ACTOR_UUID_BYTES),); - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/cache/mod.rs b/libs/@local/graph/atlas/src/serve/cache/mod.rs deleted file mode 100644 index 26ee3545176..00000000000 --- a/libs/@local/graph/atlas/src/serve/cache/mod.rs +++ /dev/null @@ -1,574 +0,0 @@ -//! Resolved visibility, held per scope for a bounded window. -//! -//! Resolving one actor's visible rows costs a store round trip, and a single map view issues one -//! request per tile. [`VisibilityCache`] holds each resolution so those requests share it, and -//! collapses a burst. A request for the same scope arriving during one resolution waits on it and -//! receives its result, so N concurrent tile requests cost one store query. -//! -//! A scope is a [`CacheKey`], naming a generation, an authenticated actor, and the digest of the -//! request filter when the request carries one. Callers presenting the same filter for the same -//! actor and generation share one entry. A filtered request and an unfiltered one are different -//! scopes. -//! -//! [`VisibilityLimits`] bounds reuse. Past the soft window an entry keeps answering while a refresh -//! runs behind it, and at the hard window it stops answering. An authority token presented by a -//! caller carries its own authenticated issue time, and opening a presented token bounds that age -//! against the token's own evidence. -//! -//! # Examples -//! -//! [`VisibilityCache`] is crate-internal, so the sketch below stands in for a compiled example. -//! -//! ```ignore -//! let cache = VisibilityCache::new(VisibilityLimits::default()); -//! let scope = CacheKey { generation, actor, filter: None }; -//! -//! // The first request resolves. A second inside the window reads the held entry. -//! let entry = cache.resolve(scope, Instant::now(), || resolve_from_store(actor)).await?; -//! let again = cache.resolve(scope, Instant::now(), || resolve_from_store(actor)).await?; -//! // One entry, shared: the second request reads what the first published. -//! assert!(Arc::ptr_eq(&entry, &again)); -//! ``` -use alloc::sync::Arc; -use core::{ - sync::atomic::{Atomic, Ordering}, - time::Duration, -}; -use std::time::Instant; - -use moka::ops::compute::Op; - -use self::scope::{CacheKey, Publication, Publications, VisibilityLimits}; -use super::{ - Atlas, ViewCensus, VisibilityProof, - delta::{DeltaSnapshot, PlacementCohort}, - density::ViewOccupancy, - hydrate::{MaskingActor, compile::ProofError}, - schedule::ViewSchedule, - visibility::ProofKind, -}; - -pub(crate) mod scope; -#[cfg(test)] -pub(crate) mod tests; - -/// A proof with the census and schedule of the view it admits, from one resolution. -/// -/// The value exists because one resolution produces all three per scope, and none is a function -/// of a request. Its production constructor censuses and schedules the proof it stores, so a -/// census or schedule paired with a foreign proof is unconstructible rather than forbidden. -#[derive(Debug)] -pub(crate) struct PendingCacheEntry { - /// The rows the actor may see. - proof: VisibilityProof, - /// The actor the scope's hydrations mask properties for, resolved with [`Self::proof`]. - masking: MaskingActor, - /// The corpus-wide census of what [`Self::proof`] admits. - census: ViewCensus, - /// The delivery schedule of [`Self::proof`]'s view. - schedule: ViewSchedule, - /// The filter document the resolution of [`Self::proof`] ran over, as presented, absent when - /// unfiltered. - /// - /// Held so a refresh can recompile the filter without a client round trip: the client is the - /// document's durable holder, and this copy lives exactly as long as the entry it resolved. - filter: Option>, - /// The arrivals snapshot the resolution read, the entry's placement cohort, absent when the - /// resolution read none. - cohort: Option>, - /// The scope's occupancy aggregate, absent for a corpus proof, which takes no cut offset. - /// - /// Aggregated from the resolution before the withdrawal fold, so the cut offset an issuance - /// resolves stays a function of the store's answer alone and no snapshot moves it. - occupancy: Option, - /// The entry's estimated weight, folded into moka's weight domain at resolution. - weight: u32, -} - -/// Folds an entry's retained bytes into moka's `u32` weight domain, saturating. -/// -/// `retained` is what the resolution measured: the proof's masks and a scoped view's own -/// cascade. The fold adds the entry's and its key's inline sizes and the filter document the -/// entry holds. A total past `u32::MAX` saturates, so a single entry retaining more than -/// 4 GiB weighs 4 GiB and the budget under-enforces by the difference - the one gap in the -/// weight domain, stated on [`VisibilityLimits::bytes`] with the other exclusions. -fn weight_of(retained: u64, filter: Option<&[u8]>) -> u32 { - let inline = size_of::() as u64 + size_of::() as u64; - let total = retained - .saturating_add(inline) - .saturating_add(filter.map_or(0, |document| document.len() as u64)); - - total.saturating_cast() -} - -impl PendingCacheEntry { - /// Pairs `proof` with its census and schedule over `atlas` and the `filter` document. - /// - /// The `filter` is the document the resolution ran over, as presented. The census walks the - /// base column once for a masked proof and reads the artifacts for an unmasked one, and the - /// schedule builds a scoped proof's cascade, so the resolution pays those costs rather than - /// the requests that share them. The first request of a scope reads a held schedule instead - /// of building one. - /// - /// `cohort` is the arrivals snapshot the resolution read, bound here for the entry's - /// lifetime. The entry folds that snapshot's withdrawals out of the proof's masks, so the - /// masks, the census, and the schedule all describe what the entry can actually serve, and - /// a request whose ingress capture is this same publication has nothing left to subtract. - /// The occupancy aggregate is taken before the fold, so the cut offset an issuance resolves - /// stays a function of the store's answer alone. The root's aggregates follow the folded - /// view while the issuance's input does not. - /// - /// Caller requirement: `proof` resolved against that same snapshot, so the slots its node - /// mask admits are the cohort's own. - pub(crate) async fn of( - atlas: Arc, - proof: VisibilityProof, - masking: MaskingActor, - filter: Option>, - cohort: Option>, - ) -> Result { - let (schedule, census, proof, occupancy, retained, cohort) = - crate::offload::run(move || { - let occupancy = match proof.kind() { - ProofKind::Scope => Some(atlas.visible_occupancy(&proof)), - ProofKind::Corpus => None, - }; - - let mut proof = proof; - if let Some(snapshot) = cohort.as_deref() { - proof.fold_withdrawn(snapshot); - } - - let schedule = - ViewSchedule::of(&atlas, &proof, PlacementCohort::of(cohort.as_deref())); - let census = atlas.census(&proof); - - // The saturated cascade is the generation's own memo, alive for the atlas's - // lifetime, so an entry sharing it retains none of it. A sharer - // took its `Arc` from the memo itself, so an unbuilt memo already - // proves this schedule is not shared - recognition peeks and never - // forces the full-corpus build. The arrival overlay is the entry's own - // either way and prices in full. - let schedule_bytes = match &schedule { - ViewSchedule::Corpus(overlay) => overlay.heap_bytes(), - ViewSchedule::Scope(scope, overlay) => { - let cascade = match atlas.saturated_scope_schedule_if_built() { - Some(memo) if Arc::ptr_eq(scope, memo) => 0, - _ => scope.heap_bytes(), - }; - - cascade + overlay.heap_bytes() - } - }; - let retained = proof.heap_bytes() + schedule_bytes; - - (schedule, census, proof, occupancy, retained, cohort) - }) - .await?; - - Ok(Self { - proof, - masking, - census, - schedule, - weight: weight_of(retained, filter.as_deref()), - filter, - cohort, - occupancy, - }) - } -} - -/// One resolved scope, as the cache holds it. -/// -/// Everything one resolution produced, together with the bookkeeping that decides when it stops -/// answering and which write may replace it. -/// -/// The cache hands out one [`Arc`] per scope, so every reader shares the entry rather than a copy -/// of its parts. No caller can detach a proof from the census resolved with it. -/// -/// A refresh publishes a new entry instead of mutating this one. A caller holding an entry across a -/// refresh keeps reading the resolution it was handed. -#[derive(Debug)] -pub(crate) struct CacheEntry { - /// The rows the actor may see. - proof: VisibilityProof, - /// The actor the scope's hydrations mask properties for, resolved with [`Self::proof`]. - masking: MaskingActor, - /// The corpus-wide census of what [`Self::proof`] admits, resolved with it. - census: ViewCensus, - /// The filter document the resolution of [`Self::proof`] ran over, as presented, absent when - /// unfiltered. - filter: Option>, - /// The arrivals snapshot the resolution read, the entry's placement cohort, absent when the - /// resolution read none. - /// - /// The entry keeps a separate [`Arc`], so the cohort outlives snapshot republication while - /// any request still answers from this entry, and a refresh rebinds the then-current - /// snapshot beside the proof it resolves. - cohort: Option>, - /// The scope's occupancy aggregate, absent for a corpus proof, which takes no cut offset. - /// - /// Aggregated from the resolution before the withdrawal fold, so the cut offset an issuance - /// resolves stays a function of the store's answer alone and no snapshot moves it. Holding - /// the aggregate here is also what lets an issuance answer without a pass over the code - /// column. - occupancy: Option, - /// The delivery schedule of [`Self::proof`]'s view, resolved with it. - /// - /// One scope builds its cascade once, at resolution, and every request under it reads that - /// one. Replacing the entry retires the schedule along with the proof it describes. - schedule: ViewSchedule, - /// When the resolution behind [`Self::proof`] ran. - resolved_at: Instant, - /// Which write into the slot published this entry. - publication: Publication, - /// Held while a refresh of this entry is in flight. - /// - /// This has one job. A burst of requests crossing the refresh horizon together runs one - /// resolution rather than one each. Which entry that resolution may publish over is - /// [`Publication`]'s question, and this answers nothing about it. - refreshing: Atomic, - /// The entry's estimated weight, priced once at resolution, read by the cache's weigher. - weight: u32, -} - -impl CacheEntry { - /// Builds an entry around a freshly resolved scope. - fn new( - PendingCacheEntry { - proof, - masking, - census, - schedule, - filter, - cohort, - occupancy, - weight, - }: PendingCacheEntry, - resolved_at: Instant, - publication: Publication, - ) -> Self { - Self { - proof, - masking, - census, - filter, - cohort, - occupancy, - schedule, - resolved_at, - publication, - refreshing: Atomic::::new(false), - weight, - } - } - - /// Returns the rows the scope may see. - pub(crate) const fn proof(&self) -> &VisibilityProof { - &self.proof - } - - /// Returns the actor the scope's hydrations mask properties for. - pub(crate) const fn masking(&self) -> MaskingActor { - self.masking - } - - /// Returns the corpus-wide census of what [`Self::proof`] admits. - pub(crate) const fn census(&self) -> &ViewCensus { - &self.census - } - - /// Returns the delivery schedule of this entry's view, resolved with its proof. - pub(crate) const fn view_schedule(&self) -> &ViewSchedule { - &self.schedule - } - - /// Returns the filter document the resolution ran over, as presented. - pub(crate) fn filter_document(&self) -> Option> { - self.filter.clone() - } - - /// Returns the scope's occupancy aggregate, absent for a corpus proof. - /// - /// The delivery-cut policy's input, aggregated once at resolution from the store's answer - /// alone, before the entry folded its snapshot's withdrawals. - pub(crate) const fn occupancy(&self) -> Option<&ViewOccupancy> { - self.occupancy.as_ref() - } - - /// Returns whether this entry's masks folded `ingress`'s withdrawals. - /// - /// True exactly when the proof declares a scope and `ingress` is the publication the entry's - /// resolution bound, by pointer identity. A capture this answers true for has no residue to - /// subtract: every row and identity it withdraws is already out of the masks, so a request - /// may skip its admission walks whole. The corpus proof never answers true, since it - /// declines the fold and admission subtraction stays its whole withdrawal authority. - /// - /// Pointer identity is the answer's width. An equal-content republication proves nothing - /// about what the masks folded, and the skip it would buy is the vacuous subtraction, so - /// false costs a walk that edits nothing and never a wrong byte. - pub(crate) fn folded(&self, ingress: &Arc) -> bool { - matches!(self.proof.kind(), ProofKind::Scope) - && self - .cohort - .as_ref() - .is_some_and(|cohort| Arc::ptr_eq(cohort, ingress)) - } - - /// Returns the entry's placement cohort, the arrivals snapshot its resolution read. - #[expect( - clippy::missing_const_for_fn, - reason = "`Option::as_deref` needs `Arc: [const] Deref`, which the standard library does \ - not provide" - )] - pub(crate) fn cohort(&self) -> PlacementCohort<'_> { - PlacementCohort::of(self.cohort.as_deref()) - } - - /// Returns whether this entry has reached its refresh horizon by `now`. - fn is_stale(&self, now: Instant, soft: Duration) -> bool { - now.saturating_duration_since(self.resolved_at) >= soft - } - - /// Returns whether this entry has outlived its reuse window by `now`. - /// - /// The age runs from the resolution, so the window bounds how long permissions may lag however - /// long the resolution itself took. - fn is_expired(&self, now: Instant, hard: Duration) -> bool { - now.saturating_duration_since(self.resolved_at) >= hard - } - - /// Takes the refresh latch, returning whether this caller now owns the refresh. - fn claim_refresh(&self) -> bool { - self.refreshing - .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) - .is_ok() - } -} - -/// Resolved scopes, held for their reuse window. -/// -/// One entry per scope, weighed by the bytes it retains, so the capacity bound follows what the -/// held resolutions actually keep allocated rather than their count. A scope with no held entry -/// resolves again on its next request and admits the same rows when its permissions have not -/// moved, so eviction costs one store round trip and nothing else. -/// -/// Admission favours the scopes that come back. Since a scope's key includes its filter, a client -/// exploring filters produces a stream of scopes asked for once each, and those cost one resolution -/// apiece and leave returning callers' entries in place. A newly active scope pays an extra -/// resolution or two before admission while the cache is full. -#[derive(Debug)] -pub(crate) struct VisibilityCache { - entries: moka::future::Cache>, - publications: Arc, - soft: Duration, - hard: Duration, -} - -impl VisibilityCache { - /// Builds a cache holding scopes under the given [`VisibilityLimits`]. - pub(crate) fn new(VisibilityLimits { bytes, soft, hard }: VisibilityLimits) -> Self { - Self { - // The eviction policy is moka's default, named here so an upstream change of default - // cannot swap it. The time-to-live runs from insertion, which is the resolution's END, - // so the window it enforces is `hard` plus however long the resolution took: `resolve` - // refuses on the entry's own `resolved_at`, and this bounds the memory behind it. - // Weights are the entries' estimated bytes, priced once at resolution, so the - // capacity is an estimated-weight budget rather than an allocator-faithful ceiling. - entries: moka::future::Cache::builder() - .max_capacity(bytes) - .weigher(|_key, entry: &Arc| entry.weight) - .eviction_policy(moka::policy::EvictionPolicy::tiny_lfu()) - .time_to_live(hard) - .build(), - publications: Arc::new(Publications::new()), - soft, - hard, - } - } - - /// Returns the entry held for `key`. - pub(crate) async fn get(&self, key: &CacheKey) -> Option> { - self.entries.get(key).await - } - - /// Resolves `key` again in place of the expired entry held for it. - /// - /// The replacement takes the slot under moka's key lock, so a burst of requests past the hard - /// window costs one store round trip: the first publishes and the rest read what it published. - /// Dropping the entry and inserting its replacement as two operations would leave the slot - /// empty in between, where a concurrent refresh's publication or an inline resolution can land - /// and then be overwritten by this one. - async fn replaced_expired( - &self, - key: CacheKey, - now: Instant, - resolver: R, - ) -> Result, Arc> - where - R: AsyncFnOnce() -> Result, - E: Send + Sync + 'static, - { - let published = self - .entries - .entry(key) - .and_try_compute_with(async |held| { - // Another caller past the same window may have published while this one queued, and - // its resolution is no older than the one this call would run. - if held.is_some_and(|held| !held.value().is_expired(now, self.hard)) { - return Ok::<_, E>(Op::Nop); - } - - Ok(Op::Put(Arc::new(CacheEntry::new( - resolver().await?, - now, - self.publications.draw(), - )))) - }) - .await - .map_err(Arc::new)?; - - Ok(published - .into_entry() - .expect( - "a compute that publishes over an absent slot and removes nothing holds an entry", - ) - .into_value()) - } - - /// Answers from the entry held for `key`, resolving inline when the cache holds none. - /// - /// Concurrent callers of one key share one inline resolution: the initializer runs once and - /// every waiter receives its result, which is what keeps a burst of tile requests for one scope - /// to a single store round trip. - async fn resolved_inline( - &self, - key: CacheKey, - now: Instant, - resolver: R, - ) -> Result, Arc> - where - R: AsyncFnOnce() -> Result, - E: Send + Sync + 'static, - { - self.entries - .try_get_with(key, async { - // Called during the generation, not before, as the function indicates a write, not - // an intention to write. - resolver().await.map(|resolution| { - Arc::new(CacheEntry::new(resolution, now, self.publications.draw())) - }) - }) - .await - } - - /// Returns the scope `key` names at `now`, resolving it through `resolver` when needed. - /// - /// A held entry inside the soft window answers immediately, and `resolver` goes uncalled. - /// - /// An entry past the soft window answers too, and one caller takes the refresh. That resolution - /// runs behind the answer already returned and replaces the entry once it completes, so a - /// request that crosses the horizon costs what a hit costs and receives the proof it already - /// had. - /// - /// An entry past the hard window answers nothing. The cache drops it and resolves again, so the - /// answer is a resolution no older than `now`. The cache judges both windows on the `now` this - /// call supplies, the monotonic clock `resolved_at` came from, and measures them from the - /// resolution the entry carries, so a slow resolution shortens the entry's reuse rather than - /// extending its window. - /// - /// A refresh publishes only over the entry it refreshed, and only while that entry is the - /// newest held. A refresh completing after a newer resolution, or after the entry it refreshed - /// has left the cache, publishes nothing, so the newest resolution of a scope is the one that - /// answers and no proof re-enters the cache with a window it did not earn. - /// - /// With nothing held, the resolution runs inline and every request arriving during it receives - /// its result, so a burst of tile requests for one scope costs one store round trip. - /// - /// An unchanged permission set resolves to the same rows. When a refresh resolves a *narrower* - /// proof, it replaces the entry and the requests after it answer from the narrower view. The - /// cache takes that mid-run change while the graph has no permission epochs for invalidating an - /// entry, so no caller learns of a refresh. - /// - /// # Errors - /// - /// Returns `resolver`'s error when an inline resolution fails, holding no entry. A failed - /// resolution publishes nothing, and the requests that shared it share its error. Failing a - /// refresh instead leaves the held entry serving and releases the refresh, so the next request - /// past the soft window tries again. - pub(crate) async fn resolve( - &self, - key: CacheKey, - now: Instant, - resolver: R, - ) -> Result, Arc> - where - R: AsyncFnOnce() -> Result + Send + 'static, - R::CallOnceFuture: Send, - E: Send + Sync + 'static, - { - // Exactly one of the three paths below resolves, which is what lets the signature ask for - // an `AsyncFnOnce`. Any call that resolves returns an entry younger than both windows, so - // no later path in the same call can resolve again. Reading the slot first is what makes - // that visible to the compiler as well as true. - // - // The windows are judged on the `now` this call is given, the one `resolved_at` came from, - // rather than on moka's, whose expiry starts when the resolution finishes. - let Some(entry) = self.entries.get(&key).await else { - return self.resolved_inline(key, now, resolver).await; - }; - - if entry.is_expired(now, self.hard) { - return self.replaced_expired(key, now, resolver).await; - } - - if entry.is_stale(now, self.soft) && entry.claim_refresh() { - let entries = self.entries.clone(); - let publications = Arc::clone(&self.publications); - let refreshed = Arc::clone(&entry); - - let _handle = tokio::spawn(async move { - let Ok(resolution) = resolver().await else { - // A failed refresh releases the latch and leaves the entry it read serving, so - // the next request past the soft window tries again. Nothing here answers a - // caller, because the request that triggered this refresh was answered before - // it began. - refreshed.refreshing.store(false, Ordering::Release); - return; - }; - - // The resolution ran outside moka's key lock. The comparison and the write happen - // inside it with nothing awaited in between. The write goes to whatever holds the - // slot when this closure reads it. - // - // Publication names the entry this refresh read, and age cannot. A request stamps - // `now` on arrival, while its insert lands a pool acquire and a store round trip - // later, so an entry resolved before this refresh began can reach the slot after - // it. A slot emptied by the hard window or by an eviction gets refilled by a - // resolution carrying its own publication. That stranger keeps the slot however its - // timestamp reads, and no proof re-enters a slot with a window it did not earn. - let _result = entries - .entry(key) - .and_compute_with(async |held| { - if held.is_none_or(|held| held.value().publication != refreshed.publication) - { - return Op::Nop; - } - - // The refreshed entry carries `now`, the time of the request that triggered - // it, so no entry claims a time later than the permissions it - // reflects. - Op::Put(Arc::new(CacheEntry::new( - resolution, - now, - publications.draw(), - ))) - }) - .await; - }); - } - - Ok(entry) - } -} diff --git a/libs/@local/graph/atlas/src/serve/cache/scope.rs b/libs/@local/graph/atlas/src/serve/cache/scope.rs deleted file mode 100644 index 4e2570ff6ba..00000000000 --- a/libs/@local/graph/atlas/src/serve/cache/scope.rs +++ /dev/null @@ -1,194 +0,0 @@ -//! The vocabulary the cache holds scopes under. -//! -//! A scope is what the cache keeps one resolution for, so its name and the policy bounding its -//! reuse stand apart from the holding machinery. [`CacheKey`] names the scope and -//! [`FilterDigest`] folds a request filter into that name. [`VisibilityLimits`] bounds how long -//! and how much the cache holds, and [`Publication`] orders the writes into one slot. -#![expect( - clippy::empty_enums, - reason = "zerocopy's FromBytes derive expands to an empty enum for its validation machinery" -)] -use core::{ - sync::atomic::{Atomic, Ordering}, - time::Duration, -}; - -use type_system::principal::actor::ActorId; - -use crate::{ - file::generation::GenerationId, - integrity::{Sha256, Sha256Digest, Update as _}, -}; - -/// The digest of a request filter, over the bytes exactly as presented. -/// -/// Byte-equal filter documents share a digest, so one held entry answers both. The digest is taken -/// over the presented bytes alone, so documents differing only in whitespace or key order are -/// different digests and resolve as different scopes, and a client re-presenting a filter sends the -/// bytes it sent before. Every digest has the same width, and equal bytes always produce equal -/// digests, on any host and in any process, which is what lets it bind a scope inside a sealed -/// token as well as name one in a key. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - Hash, - zerocopy::IntoBytes, - zerocopy::FromBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, -)] -#[repr(transparent)] -pub(crate) struct FilterDigest(Sha256Digest); - -impl FilterDigest { - /// Digests one filter document's bytes. - /// - /// `presented` is the filter exactly as the request carried it. Requests carrying byte-equal - /// filters share a scope. A request whose filter differs by so much as a space describes a - /// different scope and resolves on its own. - pub(crate) fn of(presented: &[u8]) -> Self { - let mut hasher = Sha256::new(); - hasher.update(b"filter"); - hasher.update(presented); - - Self(hasher.finalize()) - } -} - -/// Names one write into one slot, ordered against every other write this cache makes. -/// -/// Drawn when a write publishes, so every publication is unique and names exactly the entry that -/// write produced. A refresh reads an entry before it publishes over that entry, and the slot -/// stands open to a stranger in between. The refresh recognises such a stranger by its publication -/// and leaves it in place. -#[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord)] -pub(super) struct Publication(u64); - -/// One cache's source of publications. -#[derive(Debug)] -pub(super) struct Publications(Atomic); - -impl Publications { - /// Returns a source that has published nothing. - pub(super) const fn new() -> Self { - Self(Atomic::::new(0)) - } - - /// Returns a publication greater than every one this source has returned. - /// - /// [`Ordering::Relaxed`] carries it, because the only ordering the comparison needs is moka's - /// key lock, which every read and write of a published value already happens under. This - /// counter owes uniqueness and monotonicity, which `fetch_add` gives at any ordering. - /// Exhausting `u64` at a billion publications a second takes over five centuries, so the count - /// does not wrap. - pub(super) fn draw(&self) -> Publication { - Publication(self.0.fetch_add(1, Ordering::Relaxed)) - } -} - -/// One scope, naming an actor's view of one generation under one filter. -/// -/// Equal keys name the same view, so they share one held entry. -#[derive(Debug, Copy, Clone, PartialEq, Eq, Hash)] -pub(crate) struct CacheKey { - /// The generation whose row ids the proof indexes. - pub generation: GenerationId, - /// The actor whose policies the proof resolves. - pub actor: ActorId, - /// The request filter's identity, when the request carries one. - pub filter: Option, -} - -/// How long the cache reuses a resolved scope, and how much heap the held scopes retain. -/// -/// Choose [`Self::hard`] first. It is the ceiling on how long the cache goes on answering a request -/// under permissions that have since changed, which makes it the deployment's tolerated revocation -/// lag. A ten-minute hard window means a revoked permission takes effect within ten minutes for a -/// scope in continuous use, and on the next request for one that was idle. -/// -/// [`Self::soft`] is where an entry begins refreshing behind the answer it still serves. The gap -/// between the two windows is what the refresh runs in, so no request waits on store latency while -/// a scope stays in use. -/// -/// [`Self::bytes`] budgets the held scopes by estimated weight rather than by an -/// allocator-faithful ceiling. Each entry is weighed once at resolution, and the configured -/// value bounds the sum of those insertion-time estimates. The weight covers the proof's mask -/// containers with a fixed per-container allowance and the scoped view's own cascade, measured -/// exactly. The weight also adds the filter document and the entry's and key's inline sizes. The -/// saturated cascade is the generation's own memo, alive for the atlas's lifetime outside this -/// budget, and an entry sharing it weighs none of it. A deployment whose active scopes outweigh the -/// budget still answers every request, and the scopes that fall out resolve again, one store round -/// trip each. -/// -/// What the estimate leaves out is stated rather than implied. Eviction is asynchronous and -/// trails admission, so a burst briefly holds more than -/// the budget. A value a caller still holds after eviction lives for that caller's request -/// and is off the ledger. A refresh builds its replacement unpriced until publication. A -/// single entry weighing past `u32::MAX` bytes weighs exactly `u32::MAX`. The cache's own -/// per-entry bookkeeping, `Arc` control blocks and allocator slack go uncounted. An entry's -/// placement cohort is the delta register's own publication, shared across the entries whose -/// resolutions read it and priced by the register's resident telemetry, so it too goes -/// uncounted here. -/// -/// [`Self::soft`] is shorter than [`Self::hard`]. Where it is not, the expiry test runs first and -/// no entry ever refreshes behind its answer, so every request past the hard window waits on a -/// resolution of its own. -/// -/// An authority token names a cached scope, so this pair also bounds a held token's age. The -/// authority's `open` judges the issue time a token carries against the same `hard` window, and the -/// manifest publishes both values as the client's refresh and expiry horizons. -/// -/// # Examples -/// -/// Revocation taking effect within a minute, refreshing fifteen seconds before expiry: -/// -/// ``` -/// use core::time::Duration; -/// -/// use hash_graph_atlas::cli::VisibilityLimits; -/// -/// let limits = VisibilityLimits { -/// hard: Duration::from_secs(60), -/// soft: Duration::from_secs(45), -/// ..VisibilityLimits::default() -/// }; -/// -/// assert_eq!(limits.hard, Duration::from_secs(60)); -/// assert_eq!(limits.hard - limits.soft, Duration::from_secs(15)); -/// ``` -/// -/// The cost of a short window is store traffic: each active scope re-resolves once per window, so -/// halving it doubles the resolutions a steady population of actors produces. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub struct VisibilityLimits { - /// The estimated-weight budget for held scopes, in bytes. - pub bytes: u64, - /// When a held entry starts refreshing behind the answer it serves. - pub soft: Duration, - /// When a held entry stops answering. - pub hard: Duration, -} - -impl Default for VisibilityLimits { - /// A ten-minute revocation lag, refreshing at eight, budgeting one gibibyte of estimated - /// weight. - /// - /// The refresh runs in the two-minute gap between the windows. A scope in continuous use - /// re-resolves at eight minutes and keeps answering from the held proof while that runs, so - /// only a scope whose refresh failed or that no longer receives requests reaches the hard - /// window. The byte budget is an unvalidated starting point - a scoped view retains tens of - /// bytes per visible row, so the default holds on the order of a hundred concurrently hot - /// heavyweight scopes, and thousands of small ones - and the measurement that revises it is - /// the deployment's own heap reading beside the cache's hit rate. - fn default() -> Self { - Self { - bytes: 1 << 30, - soft: Duration::from_mins(8), - hard: Duration::from_mins(10), - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/cache/tests.rs b/libs/@local/graph/atlas/src/serve/cache/tests.rs deleted file mode 100644 index 1f08f441bb5..00000000000 --- a/libs/@local/graph/atlas/src/serve/cache/tests.rs +++ /dev/null @@ -1,978 +0,0 @@ -use alloc::sync::Arc; -use core::{ - sync::atomic::{AtomicUsize, Ordering}, - time::Duration, -}; -use std::time::Instant; - -use hashql_core::{collections::fast_hash_set, id::Id as _}; -use type_system::principal::actor::{ActorId, UserId}; -use uuid::Uuid; - -use super::{ - CacheEntry, PendingCacheEntry, VisibilityCache, - scope::{CacheKey, FilterDigest, VisibilityLimits}, - weight_of, -}; -use crate::{ - bitset::CompressedBitSet, - identity::{BasePosition, EdgeRowId, NodeRowId}, - salt::wire::{Mode, tests::section}, - serve::{ - CutOffset, TileLimits, View, ViewCensus, VisibilityProof, - delta::{DeltaSnapshot, PlacementCohort}, - hydrate::MaskingActor, - schedule::{ArrivalOverlay, ScopeSchedule, ViewSchedule}, - tests::{ROW_IDS, mask_hiding, narrow_usize, publish, request, test_codec, withdrawing}, - }, -}; - -fn actor(id: u128) -> ActorId { - ActorId::User(UserId::new(Uuid::from_u128(id))) -} - -/// Pairs `proof` with the empty view's census and schedule. -/// -/// The cache neither reads a census or schedule nor derives one, so the tests here - over -/// holding, refreshing and expiring entries - need no walked ones. Production keeps exactly -/// one constructor, [`PendingCacheEntry::of`], which censuses and schedules the proof it -/// stores. -fn with_empty_view(proof: VisibilityProof) -> PendingCacheEntry { - PendingCacheEntry { - proof, - masking: MaskingActor { - id: actor(1), - instance_admin: false, - }, - census: ViewCensus::EMPTY, - schedule: ViewSchedule::Scope(Arc::new(ScopeSchedule::empty()), ArrivalOverlay::empty()), - filter: None, - cohort: None, - occupancy: None, - weight: weight_of(0, None), - } -} - -/// Binds `snapshot` as the pending entry's cohort. -/// -/// The cache neither reads a cohort nor resolves one, so the tests here bind snapshots directly -/// where production receives them through [`PendingCacheEntry::of`]. -fn with_cohort(mut entry: PendingCacheEntry, snapshot: Arc) -> PendingCacheEntry { - entry.cohort = Some(snapshot); - entry -} - -const SOFT: Duration = Duration::from_mins(8); -const HARD: Duration = Duration::from_mins(10); - -/// A budget no fixture here evicts under. -const LIMITS: VisibilityLimits = VisibilityLimits { - bytes: 1 << 20, - soft: SOFT, - hard: HARD, -}; - -/// The scope of the actor numbered `actor`. -fn key_of(actor: u128) -> CacheKey { - CacheKey { - generation: "07" - .repeat(32) - .parse() - .expect("64 hexadecimal digits name a generation"), - actor: ActorId::User(UserId::new(Uuid::from_u128(actor))), - filter: None, - } -} - -fn key() -> CacheKey { - key_of(11) -} - -/// Builds a masked proof admitting `nodes` as node rows and no link rows. -fn proof_of(nodes: &[u32]) -> VisibilityProof { - VisibilityProof::from_masks( - CompressedBitSet::from_rows(nodes.iter().copied().map(NodeRowId::from_u32)), - CompressedBitSet::::new(), - fast_hash_set(), - ) -} - -/// Whether `entry`'s proof admits the node row numbered `node`. -/// -/// The fixtures give each resolution a row of its own, so one membership test names which -/// resolution an entry carries. -fn admits(entry: &CacheEntry, node: u32) -> bool { - entry.proof().contains(NodeRowId::from_u32(node)) -} - -/// A resolver answering `rows` and counting its calls. -/// -/// A macro rather than a function, because the cache asks its resolver for a `Send` future at -/// every lifetime, and an opaque return type publishes only the bounds it names. Expanding at -/// the call site hands the compiler the closure itself, which carries the property. -macro_rules! answering { - ($rows:expr, $calls:expr) => {{ - let calls = Arc::clone($calls); - - async move || { - calls.fetch_add(1, Ordering::Relaxed); - Ok::<_, ()>(with_empty_view(proof_of($rows))) - } - }}; -} - -/// An entry inside the soft window answers without resolving again. -#[tokio::test] -async fn held_entry_hit() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - for elapsed in [Duration::ZERO, SOFT / 2] { - cache - .resolve(key(), now + elapsed, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - } - - assert_eq!( - calls.load(Ordering::Relaxed), - 1, - "the second read was served" - ); -} - -/// Requests arriving during one resolution share it. -/// -/// Without sharing, a burst of tile requests for one scope would each issue the store query the -/// cache exists to avoid. -#[tokio::test] -async fn concurrent_misses_resolve_once() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - let (first, second) = tokio::join!( - cache.resolve(key(), now, answering!(&[1, 2, 3], &calls)), - cache.resolve(key(), now, answering!(&[1, 2, 3], &calls)), - ); - - let first = first.expect("the resolution answers"); - let second = second.expect("the resolution answers"); - - assert_eq!(calls.load(Ordering::Relaxed), 1, "one resolution ran"); - assert!(Arc::ptr_eq(&first, &second), "both callers hold one entry"); -} - -/// An entry past the soft window answers from the held proof while a refresh replaces it. -/// -/// The request crossing the horizon pays nothing: it receives the proof it already had. The -/// refresh runs as a task, so the loop below drives the runtime until the refresh publishes -/// rather than waiting on a clock. -#[tokio::test] -async fn stale_serves_while_refreshing() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - let held = cache - .resolve(key(), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - - let stale = cache - .resolve(key(), now + SOFT, answering!(&[9], &calls)) - .await - .expect("the held entry answers"); - assert!( - Arc::ptr_eq(&stale, &held), - "the stale read answered from the held proof" - ); - assert_eq!( - calls.load(Ordering::Relaxed), - 1, - "the stale answer ran no resolution of its own" - ); - - let mut refreshed = None; - for _ in 0..16_u8 { - tokio::task::yield_now().await; - let entry = cache.get(&key()).await.expect("the entry stays held"); - if !Arc::ptr_eq(&entry, &held) { - refreshed = Some(entry); - break; - } - } - - let refreshed = refreshed.expect("the refresh replaced the held entry"); - assert!( - admits(&refreshed, 9), - "the published entry carries the refresh's own resolution" - ); - assert_eq!( - calls.load(Ordering::Relaxed), - 2, - "the refresh resolved behind the answer" - ); -} - -/// A filter's identity separates entries for one actor. -#[tokio::test] -async fn filter_identity_separates_entries() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - let mut filtered = key(); - filtered.filter = Some(FilterDigest::of(b"web = 7")); - - for scope in [key(), filtered] { - cache - .resolve(scope, now, answering!(&[1], &calls)) - .await - .expect("the resolution answers"); - } - - assert_eq!( - calls.load(Ordering::Relaxed), - 2, - "the filtered scope resolved on its own" - ); -} - -/// The capacity bound prices retained bytes, so the budget holds what actually fits. -/// -/// The budget is exactly two fixture entries' weight, computed through the same fold the cache -/// weighs with. Both scopes fit it whole, and a third cannot raise the held bytes past the -/// budget: the bite lands by eviction or by refused admission, since the byte price is what -/// this fixture pins. -#[tokio::test] -async fn capacity_bound_prices_retained_bytes() { - let each = weight_of(proof_of(&[1, 2, 3]).heap_bytes(), None); - let cache = VisibilityCache::new(VisibilityLimits { - bytes: u64::from(each) * 2, - ..LIMITS - }); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - for actor in [1_u128, 2] { - cache - .resolve(key_of(actor), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - } - cache.entries.run_pending_tasks().await; - - for actor in [1_u128, 2] { - assert!( - cache.get(&key_of(actor)).await.is_some(), - "scope {actor} is held: two entries weigh exactly the byte budget" - ); - } - - cache - .resolve(key_of(3), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - cache.entries.run_pending_tasks().await; - - assert_eq!( - cache.entries.entry_count(), - 2, - "the third scope did not raise the held bytes past the budget" - ); -} - -/// Weighing a small scope never builds the saturated memo it is compared against. -/// -/// The weigher recognizes a memo sharer by pointer identity, and a sharer took its `Arc` from -/// the memo itself, so an unbuilt memo already answers no. This pins the regression where the -/// weigher reached the memo through its building accessor and billed the first small scope's -/// resolution for the whole corpus's cascade construction. -#[tokio::test] -async fn small_scope_memo_unbuilt() { - let (_generation, atlas) = publish("cache-weigher-memo-unbuilt").await; - let atlas = Arc::new(atlas); - let proof = mask_hiding(&atlas, &[0]); - - let entry = PendingCacheEntry::of( - Arc::clone(&atlas), - proof, - MaskingActor { - id: actor(2), - instance_admin: false, - }, - None, - None, - ) - .await - .expect("a masked proof resolves"); - - assert!(entry.weight > 0, "the scope weighed its own cascade"); - assert!( - atlas.saturated_scope_schedule_if_built().is_none(), - "recognition peeked: pricing a small scope must not construct the full-corpus memo", - ); -} - -/// The entry censuses the folded view, and aggregates occupancy from the unfolded one. -/// -/// [`PendingCacheEntry::of`] folds the cohort's withdrawals out of the proof before censusing, -/// so the aggregates the root publishes describe what the entry can actually serve. The -/// occupancy aggregate reads the proof before the fold, which keeps the cut offset an issuance -/// resolves a function of the store's answer alone. This pins the census/occupancy pair's ordering -/// at the constructor: nothing but statement order inside `of` holds it. Reordering the census -/// against the fold reddens exactly here, and reordering the occupancy reddens this witness -/// and the issuance witness beside it. The schedule's place in that order has its own witness in -/// the delivery test below, which takes the entry to bytes. -#[tokio::test] -async fn entry_census_folded_occupancy_unfolded() { - let (_generation, atlas) = publish("cache-census-folded").await; - let atlas = Arc::new(atlas); - let proof = mask_hiding(&atlas, &[]); - let masking = MaskingActor { - id: actor(3), - instance_admin: false, - }; - - // Withdraw every fixture row but one, which moves any census the fold reaches: the folded - // view holds one point where the resolution's holds the corpus. - let survivor = 7_u8; - let count = u8::try_from(atlas.row_ids().len()).expect("the fixture universe fits u8"); - let seeds: Vec = (0..count).filter(|&seed| seed != survivor).collect(); - let snapshot = Arc::new(withdrawing(&atlas, &seeds)); - - let entry = PendingCacheEntry::of( - Arc::clone(&atlas), - proof.clone(), - masking, - None, - Some(Arc::clone(&snapshot)), - ) - .await - .expect("the folded entry builds"); - - let mut folded = proof.clone(); - folded.fold_withdrawn(&snapshot); - assert_ne!( - atlas.census(&folded), - atlas.census(&proof), - "the fold moves this view's census, so the equalities below have teeth" - ); - - assert_eq!( - entry.census, - atlas.census(&folded), - "the entry's census is the folded view's own" - ); - assert_eq!( - entry.occupancy, - Some(atlas.visible_occupancy(&proof)), - "the entry's occupancy is the unfolded proof's own" - ); -} - -/// The occupancy aggregate an issuance reads ignores the entry's folded snapshot. -/// -/// The entry aggregates occupancy from the store's answer before folding withdrawals, so the cut -/// offset an issuance resolves is a function of the resolution alone and no snapshot moves it. The -/// witness withdraws every row of one occupied cell, because occupancy counts cells over the -/// fixture's co-located points and a lesser withdrawal cannot move it - which is exactly what -/// makes the equal aggregates a statement rather than a tautology, and the folded proof's own -/// occupancy pins that the harness holds the condition. -#[tokio::test] -async fn mint_occupancy_ignores_fold() { - let (_generation, atlas) = publish("cache-mint-occupancy").await; - let atlas = Arc::new(atlas); - - // Every row of the deepest grid's first occupied cell, by its position's Morton key. Fixture - // node row `r` owns seed `r`, so the cell's rows withdraw as their own seeds. - let row_ids = atlas.rows.view(); - let first_key = atlas.morton.code(BasePosition::MIN); - let cell_seeds: Vec = (0..row_ids.len()) - .map(narrow_usize) - .map(BasePosition::from_u32) - .filter(|&position| atlas.morton.code(position) == first_key) - .map(|position| u8::try_from(row_ids[position].as_u32()).expect("fixture rows fit u8")) - .collect(); - - let proof = mask_hiding(&atlas, &[]); - let masking = MaskingActor { - id: actor(3), - instance_admin: false, - }; - let snapshot = Arc::new(withdrawing(&atlas, &cell_seeds)); - - let folded = PendingCacheEntry::of( - Arc::clone(&atlas), - proof.clone(), - masking, - None, - Some(Arc::clone(&snapshot)), - ) - .await - .expect("the folded entry builds"); - let bare = PendingCacheEntry::of(Arc::clone(&atlas), proof.clone(), masking, None, None) - .await - .expect("the bare entry builds"); - - assert!( - folded.occupancy.is_some(), - "a scoped entry holds its aggregate", - ); - assert_eq!( - folded.occupancy, bare.occupancy, - "no snapshot moves an issuance's input", - ); - assert_eq!( - folded.occupancy, - Some(atlas.visible_occupancy(&proof)), - "the aggregate is the unfolded proof's own", - ); - - let mut hand_folded = proof.clone(); - hand_folded.fold_withdrawn(&snapshot); - assert_ne!( - atlas.visible_occupancy(&hand_folded), - atlas.visible_occupancy(&proof), - "the folded proof's occupancy differs, so the equalities above have teeth", - ); -} - -/// The entry's own delivery hides the rows its cohort withdrew. -/// -/// For a scoped view the cascade is the delivery authority: tile assembly reads the cut and -/// re-consults no mask per delivered row, so a schedule built before the fold keeps delivering -/// the withdrawn row while every aggregate-reading witness stays green. The witness therefore -/// takes the entry to bytes, bound exactly as a request whose ingress capture is the entry's -/// own publication binds it - with no capture left to subtract, so the entry's delivery is the -/// whole authority. -#[tokio::test] -async fn entry_withholds_cohort_withdrawn() { - let (_generation, atlas) = publish("cache-entry-delivery").await; - let atlas = Arc::new(atlas); - let proof = mask_hiding(&atlas, &[]); - let masking = MaskingActor { - id: actor(4), - instance_admin: false, - }; - - // A row the root delivers. Fixture node row `r` owns seed `r`. - let row = atlas.row_ids()[BasePosition::from_u32(1)].as_u32(); - let seed = u8::try_from(row).expect("fixture rows fit u8"); - let snapshot = Arc::new(withdrawing(&atlas, &[seed])); - - let entry = PendingCacheEntry::of( - Arc::clone(&atlas), - proof, - masking, - None, - Some(Arc::clone(&snapshot)), - ) - .await - .expect("the folded entry builds"); - - // The nulled capture models the request whose ingress capture is this entry's own bound - // publication: the extractor skips its admission walk, so the entry's own delivery is the - // whole authority. - let view = View::bind( - atlas.grid, - &entry.proof, - entry.census, - &entry.schedule, - CutOffset::ZERO, - PlacementCohort::of(entry.cohort.as_deref()), - None, - ) - .expect("the entry pairs its own proof and schedule"); - - let bytes = atlas - .tile(&request(0, 0, 0, Mode::Delta), TileLimits::default(), view) - .expect("the fixture tile serves"); - let (chunks, remainder) = section(&bytes, ROW_IDS) - .expect("ROW_IDS is present") - .as_chunks::<4>(); - assert!(remainder.is_empty(), "row sections are whole u32 columns"); - - let wire = test_codec(&atlas) - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get(); - assert!( - !chunks - .iter() - .copied() - .map(u32::from_le_bytes) - .any(|delivered| delivered == wire), - "the entry delivered a row its own cohort withdrew" - ); - - // Positive control: the same binding without a cohort delivers the row, so the absence - // above is the fold's doing rather than the probe's. The count matches on both sides, - // because the cascade substitutes the next survivor instead of shrinking the delivery. - let unfolded = PendingCacheEntry::of( - Arc::clone(&atlas), - mask_hiding(&atlas, &[]), - masking, - None, - None, - ) - .await - .expect("the unfolded entry builds"); - let control_view = View::bind( - atlas.grid, - &unfolded.proof, - unfolded.census, - &unfolded.schedule, - CutOffset::ZERO, - PlacementCohort::of(unfolded.cohort.as_deref()), - None, - ) - .expect("the entry pairs its own proof and schedule"); - let control = atlas - .tile( - &request(0, 0, 0, Mode::Delta), - TileLimits::default(), - control_view, - ) - .expect("the fixture tile serves"); - let (control_chunks, control_remainder) = section(&control, ROW_IDS) - .expect("ROW_IDS is present") - .as_chunks::<4>(); - assert!( - control_remainder.is_empty(), - "row sections are whole u32 columns" - ); - assert!( - control_chunks - .iter() - .copied() - .map(u32::from_le_bytes) - .any(|delivered| delivered == wire), - "the control must deliver the row the folded entry hides" - ); - assert_eq!( - control_chunks.len(), - chunks.len(), - "the fold substitutes the next survivor rather than shrinking the delivery" - ); -} - -/// The weight fold saturates at the weight domain's top instead of failing or wrapping. -#[test] -fn weight_of_saturates() { - assert_eq!( - weight_of(u64::MAX, None), - u32::MAX, - "a retained figure past the domain weighs the domain's top", - ); - assert_eq!( - weight_of(u64::from(u32::MAX), Some(&[0_u8; 16])), - u32::MAX, - "the fold's own additions saturate with it", - ); -} - -/// The cache answers nothing from an entry past the hard window. -/// -/// The window runs from the resolution, so the clock this advances is the injected `now` and -/// not the cache's own. An entry whose age reaches `hard` resolves again inline, and the answer -/// carries the new proof rather than the held one. -#[tokio::test] -async fn expired_entry_resolves_again() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - let held = cache - .resolve(key(), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - - let answered = cache - .resolve(key(), now + HARD, answering!(&[9], &calls)) - .await - .expect("the expired entry resolves again"); - - assert_eq!( - calls.load(Ordering::Relaxed), - 2, - "the expired read resolved rather than answering from the held proof" - ); - assert!( - !Arc::ptr_eq(&answered, &held), - "the answer carries the new resolution" - ); - assert!(admits(&answered, 9), "the answer is the inline resolution"); -} - -/// A refresh landing after a newer resolution publishes nothing. -/// -/// The refresh task holds the proof it resolved. An unconditional `insert` would let an older -/// proof replace a newer one and restart the window it lives in. The cache would then go on -/// serving a permission revoked between the two for a fresh window. The fixture forces that -/// order. A paused resolver holds the refresh in flight until a newer inline resolution has -/// published, and only then does the refresh complete. -#[tokio::test] -async fn stale_refresh_publishes_nothing() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - let held = cache - .resolve(key(), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - - // The refresh resolver blocks on a permit that the `Notify` stores rather than signals, so the - // order below holds however the runtime interleaves the spawned task. - let gate = Arc::new(tokio::sync::Notify::new()); - let refresh_gate = Arc::clone(&gate); - let refresh_calls = Arc::clone(&calls); - let stale = cache - .resolve(key(), now + SOFT, move || { - let gate = Arc::clone(&refresh_gate); - let calls = Arc::clone(&refresh_calls); - async move { - // `notify_one` stores a permit, so this completes whether the release ran - // before this task was first polled or after. - gate.notified().await; - calls.fetch_add(1, Ordering::Relaxed); - Ok::<_, ()>(with_empty_view(proof_of(&[1, 2, 3]))) - } - }) - .await - .expect("the held entry answers"); - assert!( - Arc::ptr_eq(&stale, &held), - "the stale read answered as held" - ); - - // A newer resolution replaces the entry while the refresh is still in flight. - cache.entries.invalidate(&key()).await; - let newer = cache - .resolve(key(), now + SOFT + SOFT, answering!(&[7], &calls)) - .await - .expect("the newer resolution answers"); - assert_eq!( - calls.load(Ordering::Relaxed), - 2, - "the refresh is still in flight: one initial resolution, one newer" - ); - - gate.notify_one(); - - // The refresh's own resolution must COMPLETE for this fixture to witness anything: without - // that count the assertion below would hold of a task that never published at all. - let mut resolved = false; - for _ in 0..64_u8 { - tokio::task::yield_now().await; - if calls.load(Ordering::Relaxed) == 3 { - resolved = true; - } - } - assert!(resolved, "the refresh resolved behind the newer entry"); - - let after = cache.get(&key()).await.expect("an entry stays held"); - assert!( - Arc::ptr_eq(&after, &newer), - "the newer resolution still answers: the refresh published nothing over it" - ); -} - -/// A refresh publishes only over the entry it refreshed. -/// -/// Identity names that entry. Age cannot name it: a request stamps `now` when it arrives, while -/// its insert completes only after a pool acquire and a store round trip. An entry resolved -/// before the refresh was triggered can be written into the slot after that refresh started. An -/// age comparison overwrites it, and its proof is a different resolution that no request asked -/// to have replaced. -/// -/// The fixture holds a refresh in flight while it empties the slot and refills it with a -/// stranger stamped below the refresh trigger. Only then does the refresh complete. -#[tokio::test] -async fn refresh_scoped_to_own_entry() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - let held = cache - .resolve(key(), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - - let gate = Arc::new(tokio::sync::Notify::new()); - let refresh_gate = Arc::clone(&gate); - let refresh_calls = Arc::clone(&calls); - let stale = cache - .resolve(key(), now + SOFT, move || { - let gate = Arc::clone(&refresh_gate); - let calls = Arc::clone(&refresh_calls); - async move { - gate.notified().await; - calls.fetch_add(1, Ordering::Relaxed); - Ok::<_, ()>(with_empty_view(proof_of(&[1, 2, 3]))) - } - }) - .await - .expect("the held entry answers"); - assert!( - Arc::ptr_eq(&stale, &held), - "the stale read answered as held" - ); - - // A different entry takes the slot while the refresh is in flight, stamped below the - // refresh trigger: the request that arrived before the refreshing one and whose insert - // landed after it. - cache.entries.invalidate(&key()).await; - let stranger = cache - .resolve(key(), now + SOFT / 2, answering!(&[7], &calls)) - .await - .expect("the stranger resolution answers"); - assert!( - stranger.resolved_at < now + SOFT, - "the stranger is older than the refresh trigger, so an age test replaces it" - ); - - gate.notify_one(); - - // The refresh's own resolution must complete for this fixture to witness anything. - let mut resolved = false; - for _ in 0..64_u8 { - tokio::task::yield_now().await; - if calls.load(Ordering::Relaxed) == 3 { - resolved = true; - } - } - assert!(resolved, "the refresh resolved behind the stranger"); - - let after = cache.get(&key()).await.expect("an entry stays held"); - assert!( - Arc::ptr_eq(&after, &stranger), - "the stranger still answers: a refresh publishes only over the entry it refreshed" - ); -} - -/// A refresh whose entry has left the cache publishes nothing. -/// -/// The slot empties for exactly the reasons the windows exist, either because the hard window -/// dropped the entry or because capacity evicted it, so a refresh that filled the slot again -/// would give a proof resolved before that removal a fresh window to live in, with no request -/// having asked for it. A refresh replaces the entry it refreshed and creates none. -#[tokio::test] -async fn removed_entry_refresh_noop() { - let cache = VisibilityCache::new(LIMITS); - let calls = Arc::new(AtomicUsize::new(0)); - let now = Instant::now(); - - cache - .resolve(key(), now, answering!(&[1, 2, 3], &calls)) - .await - .expect("the resolution answers"); - - let gate = Arc::new(tokio::sync::Notify::new()); - let refresh_gate = Arc::clone(&gate); - let refresh_calls = Arc::clone(&calls); - cache - .resolve(key(), now + SOFT, move || { - let gate = Arc::clone(&refresh_gate); - let calls = Arc::clone(&refresh_calls); - async move { - gate.notified().await; - calls.fetch_add(1, Ordering::Relaxed); - Ok::<_, ()>(with_empty_view(proof_of(&[1, 2, 3]))) - } - }) - .await - .expect("the held entry answers"); - - // The entry leaves while its refresh is in flight, and nothing resolves after it. - cache.entries.invalidate(&key()).await; - gate.notify_one(); - - let mut resolved = false; - for _ in 0..64_u8 { - tokio::task::yield_now().await; - if calls.load(Ordering::Relaxed) == 2 { - resolved = true; - } - } - assert!(resolved, "the refresh resolved after the entry was removed"); - - assert!( - cache.get(&key()).await.is_none(), - "the refresh published nothing into the empty slot" - ); -} - -/// A failed resolution holds no entry. -#[tokio::test] -async fn failed_resolution_holds_no_entry() { - let cache = VisibilityCache::new(LIMITS); - let now = Instant::now(); - - let failed = cache - .resolve(key(), now, async || { - Err::("the store refused") - }) - .await; - - let _refusal = failed.expect_err("a failed resolution answers its error"); - assert!( - cache.get(&key()).await.is_none(), - "a failed resolution leaves no entry" - ); -} - -/// Identity tables of a generation that fitted nothing, so a snapshot resolves no rows. -struct Unfitted; - -impl crate::serve::delta::IdentityTables for Unfitted { - fn node_row_of(&self, _id: crate::postgres::id::ArchivedEntityId) -> Option { - None - } - - fn edge_row_of(&self, _id: crate::postgres::id::ArchivedEntityId) -> Option { - None - } - - fn ontology_row_of( - &self, - _id: crate::postgres::id::ArchivedOntologyTypeUuid, - ) -> Option { - None - } -} - -/// An entry hands requests the cohort its resolution bound. -/// -/// The universe check reads the bound snapshot's own bound rather than the base the caller -/// offers, so the entry's arrival-sensitive reads follow the resolution's publication. An entry -/// resolved with no publication answers the base, which is the empty cohort's contract. -#[tokio::test] -async fn entry_exposes_bound_cohort() { - use hash_graph_temporal_versioning::Timestamp; - - use crate::serve::{ - codec::Universe, - delta::{DeltaRegister, DeltaRevision}, - }; - - let register = DeltaRegister::new( - Universe::new(NodeRowId::new(10)), - Universe::new(crate::identity::EdgeRowId::new(6)), - Universe::new(crate::identity::OntologyRowId::new(4)), - ); - let snapshot = Arc::new(register.snapshot( - &Unfitted, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(1), - )); - - let cache = VisibilityCache::new(LIMITS); - let now = Instant::now(); - - let bound = cache - .resolve(key(), now, { - let snapshot = Arc::clone(&snapshot); - async move || Ok::<_, ()>(with_cohort(with_empty_view(proof_of(&[1])), snapshot)) - }) - .await - .expect("the resolution answers"); - assert_eq!( - bound.cohort().universe(Universe::new(NodeRowId::new(7))), - Universe::new(NodeRowId::new(10)), - "the bound snapshot's universe answers" - ); - - let unbound = cache - .resolve(key_of(12), now, async || { - Ok::<_, ()>(with_empty_view(proof_of(&[2]))) - }) - .await - .expect("the resolution answers"); - assert_eq!( - unbound.cohort().universe(Universe::new(NodeRowId::new(7))), - Universe::new(NodeRowId::new(7)), - "an empty cohort answers the base" - ); -} - -/// An entry answers `folded` for exactly the publication its resolution bound. -/// -/// Pointer identity is the release: an equal-content republication proves nothing about what -/// the masks folded and answers false, which costs the vacuous subtraction rather than a wrong -/// byte. A corpus entry declines the fold and never answers true, whatever snapshot it bound, -/// and an entry that bound none folded nothing. -#[tokio::test] -async fn folded_matches_bound_publication() { - use hash_graph_temporal_versioning::Timestamp; - - use crate::serve::{ - codec::Universe, - delta::{DeltaRegister, DeltaRevision}, - }; - - let register = DeltaRegister::new( - Universe::new(NodeRowId::new(10)), - Universe::new(crate::identity::EdgeRowId::new(6)), - Universe::new(crate::identity::OntologyRowId::new(4)), - ); - let publish = || { - Arc::new(register.snapshot( - &Unfitted, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(1), - )) - }; - let snapshot = publish(); - let republished = publish(); - assert_eq!( - *snapshot, *republished, - "the two publications carry one content" - ); - - let cache = VisibilityCache::new(LIMITS); - let now = Instant::now(); - - let scoped = cache - .resolve(key(), now, { - let snapshot = Arc::clone(&snapshot); - async move || Ok::<_, ()>(with_cohort(with_empty_view(proof_of(&[1])), snapshot)) - }) - .await - .expect("the resolution answers"); - assert!(scoped.folded(&snapshot), "the bound publication is folded"); - assert!( - !scoped.folded(&republished), - "an equal-content republication is not the bound one", - ); - - let corpus = cache - .resolve(key_of(12), now, { - let snapshot = Arc::clone(&snapshot); - async move || { - Ok::<_, ()>(with_cohort( - with_empty_view(VisibilityProof::full_visibility()), - snapshot, - )) - } - }) - .await - .expect("the resolution answers"); - assert!( - !corpus.folded(&snapshot), - "a corpus proof declines the fold", - ); - - let unbound = cache - .resolve(key_of(13), now, async || { - Ok::<_, ()>(with_empty_view(proof_of(&[2]))) - }) - .await - .expect("the resolution answers"); - assert!( - !unbound.folded(&snapshot), - "an entry that bound no publication folded nothing", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/codec.rs b/libs/@local/graph/atlas/src/serve/codec.rs deleted file mode 100644 index bb3ec4e8cab..00000000000 --- a/libs/@local/graph/atlas/src/serve/codec.rs +++ /dev/null @@ -1,277 +0,0 @@ -//! The wire row-id boundary. -//! -//! A keyed permutation of the full `u32` range, applied where ids cross the wire. -//! -//! Internal row ids are dense and assignment-ordered, so sending them verbatim lets a principal -//! bound hidden row counts between two visible ids (gap analysis) and estimate the universe size -//! from any received sample. Ids therefore cross the wire through [`RowCodec`], a keyed bijection -//! of the full `u32` range. Wire ids are opaque and sparse - a valid id is any `u32` value, and the -//! mapping is independent of the universe size. To the extent the keyed permutation is -//! indistinguishable from a random permutation of `[0, 2^32)` at the volume of ids an observer -//! collects, the wire ids one scope receives follow the distribution of a uniform subset of the -//! range, and order, adjacency, creation time, and the universe size stay hidden. That -//! indistinguishability is the construction's design target, not a proved bound: the codec is an -//! obfuscation layer with exact decode guarantees, not a demonstrated security boundary. -//! -//! # Model -//! -//! An eight-round balanced Feistel network permutes the `u32` range. Round `i` splits the state -//! into two 16-bit halves and maps `(L, R)` to `(R, L xor F_i(R))` under the keyed round function -//! `F_i` (SipHash-2-4 truncated to 16 bits). The permutation stays fixed as the universe grows, so -//! appending rows leaves every existing mapping unchanged and wire ids are stable within a -//! generation under row addition. Encoding applies the network to a row id. Decoding applies the -//! inverse network and bounds-checks the result against the accepted [`Universe`], so exactly the -//! `N` wire values in the image of `[0, N)` decode and every other value answers [`None`]. -//! -//! The universe arrives per call rather than living in the codec, because the accepted row set -//! grows while a generation serves: delta slot allocation extends it past the fitted rows. A -//! caller takes one [`Universe`] value and reads it at every encode and decode in one answer, so -//! the accepted set cannot shift inside a response. -//! -//! # Keys -//! -//! Round keys derive from `HKDF-SHA256` over the server secret, salted by the generation identity -//! and expanded under a per-universe label, when a generation opens for serving. Equal `(secret, -//! generation, label)` give equal mappings, so responses stay byte-deterministic across restarts; a -//! different generation changes every wire id, and the label names the mapping's universe - Surface -//! v1 exposes one universe, the node rows. Edges carry their link entity's identity instead of a -//! wire id of their own. Wire ids never reach the fit pipeline, and no artifact stores one. - -use core::{fmt, hash::Hasher as _, marker::PhantomData}; - -use hashql_core::id::Id; -use hkdf::Hkdf; -use sha2::Sha256; -use siphasher::sip::SipHasher24; -use zeroize::Zeroizing; - -use super::WireSecret; -use crate::file::generation::GenerationId; - -/// The Feistel round count one codec applies. -// -// The per-id cost of eight rounds stays under a microsecond. The classical Luby-Rackoff strong-PRP -// threshold is four rounds, an asymptotic result, and this codec claims no concrete -// indistinguishability bound at this domain size and round function. -const ROUNDS: usize = 8; - -/// The Feistel half width. -/// -/// The network splits the `u32` state into two 16-bit halves. -const HALF_BITS: u32 = 16; - -/// The low-half mask. -const HALF_MASK: u32 = 0xFFFF; - -/// The HKDF expansion label of the node-row universe. -pub(crate) const NODE_LABEL: &[u8] = b"atlas.wire.node.v1"; - -/// The accepted row universe, the exclusive bound on the rows a codec maps. -/// -/// Rows live in `[0, N)` and the value is `N`. The generation's open derives the base bound from -/// the validated row column, and a delta snapshot carries the wider bound its slot allocation has -/// reached, so the accepted set is a fact about one snapshot rather than about the generation. A -/// caller resolves one value and reads it at every encode and decode in one answer. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct Universe(N); - -impl Universe -where - N: Id, -{ - /// Bounds the universe at `rows`. - #[must_use] - pub(crate) const fn new(rows: N) -> Self { - Self(rows) - } - - /// Returns the exclusive row bound. - #[must_use] - pub(crate) const fn size(self) -> usize - where - N: [const] Id, - { - self.0.as_usize() - } - - pub(crate) const fn grow(self) -> Option<(Self, N)> - where - N: [const] Id, - { - let next = self.0.next()?; - Some((Self(next), self.0)) - } - - /// Returns whether `row` lies inside the universe. - pub(crate) const fn contains(self, id: N) -> bool - where - N: [const] PartialOrd, - { - id < self.0 - } -} - -/// A row id as it crosses the wire. -/// -/// The value relates to an internal row id only through the owning generation's [`RowCodec`]: -/// [`RowCodec::encode`] produces egress values, and deserialization admits client-echoed values -/// whose meaning only [`RowCodec::decode`] assigns - an arbitrary `u32` is a well-formed -/// [`WireRow`] that decodes to [`None`] outside the encoded image. Comparisons order wire values, -/// so a tie broken on [`WireRow`] is client-observable without exposing internal order. -#[derive(Debug, PartialEq, Eq, PartialOrd, Ord, Hash, schemars::JsonSchema)] -#[repr(transparent)] -#[schemars(transparent)] -pub(crate) struct WireRow(u32, #[schemars(skip)] PhantomData); - -impl WireRow { - /// Returns the wire value. - #[inline] - #[must_use] - pub(crate) const fn get(self) -> u32 { - self.0 - } -} -#[cfg(test)] // The serve tests pin wire values without an encoding pass. -impl WireRow { - /// Pins a wire value at its literal wire-domain representation. - /// - /// The value already carries its wire form, and this constructor encodes nothing. - pub(crate) const fn pinned(value: u32) -> Self { - Self(value, PhantomData) - } -} - -impl Copy for WireRow {} - -impl Clone for WireRow { - fn clone(&self) -> Self { - *self - } -} - -// Manual impls: only the wire value crosses the wire, and the phantom -// parameter stays out of the serde bounds. -impl serde::Serialize for WireRow { - fn serialize(&self, serializer: S) -> Result - where - S: serde::Serializer, - { - self.0.serialize(serializer) - } -} - -impl<'de, I> serde::Deserialize<'de> for WireRow { - fn deserialize(deserializer: D) -> Result - where - D: serde::Deserializer<'de>, - { - u32::deserialize(deserializer).map(|value| Self(value, PhantomData)) - } -} - -/// The keyed mapping between one dense row domain and its wire ids. -/// -/// One codec serves one row domain of one generation. [`Self::derive`] is the constructor. The -/// underlying permutation bijects the `u32` range for every key. Encoding restricts it to the -/// caller's [`Universe`] and decoding inverts exactly the image of that universe, answering -/// [`None`] elsewhere. Both are pure: the mapping never changes while the generation serves, and -/// only the accepted bound moves as slots allocate. -pub(crate) struct RowCodec { - /// The per-round SipHash-2-4 keys. - keys: Zeroizing<[[u8; 16]; ROUNDS]>, - _marker: PhantomData, -} - -impl RowCodec -where - I: Id, -{ - /// Derives the codec of one row domain from the server secret. - /// - /// The generation identity salts the extraction and `label` separates row domains under one - /// generation. Equal arguments derive equal codecs. - pub(crate) fn derive(secret: &WireSecret, generation: GenerationId, label: &[u8]) -> Self { - let salt = generation.digest().to_bytes(); - - let mut keys = Zeroizing::new([[0_u8; 16]; ROUNDS]); - Hkdf::::new(Some(&salt), secret.as_bytes()) - .expand(label, (*keys).as_flattened_mut()) - .expect("128 octets stay within HKDF-SHA256's expansion bound"); - - Self { - keys, - _marker: PhantomData, - } - } - - /// Encodes an internal row id of `universe` as its wire id. - /// - /// # Panics - /// - /// This panics when `row` lies outside `universe`. Encoding is a producer contract, so an - /// out-of-universe row upstream is a defect in the caller rather than input to reject. - pub(crate) fn encode(&self, row: I, universe: Universe) -> WireRow { - assert!( - universe.contains(row), - "the codec encodes rows of the caller's universe", - ); - - WireRow(self.permute(row.as_u32()), PhantomData) - } - - /// Decodes a wire value back to its internal row id, [`None`] outside the image of `universe`. - /// - /// [`None`] is the single out-of-image answer. Ingress resolution collapses it with every other - /// lookup failure before a response can observe the cause. - pub(crate) fn decode(&self, wire: WireRow, universe: Universe) -> Option { - let row = I::from_u32(self.unpermute(wire.get())); - universe.contains(row).then_some(row) - } - - /// Applies the Feistel network once over the `u32` range. - fn permute(&self, mut state: u32) -> u32 { - for key in &*self.keys { - let left = state >> HALF_BITS; - let right = state & HALF_MASK; - state = (right << HALF_BITS) | (left ^ (round(key, right) & HALF_MASK)); - } - - state - } - - /// Applies the inverse network once over the `u32` range. - fn unpermute(&self, mut state: u32) -> u32 { - for key in self.keys.iter().rev() { - let right = state >> HALF_BITS; - let left = (state & HALF_MASK) ^ (round(key, right) & HALF_MASK); - state = (left << HALF_BITS) | right; - } - - state - } -} - -impl fmt::Debug for RowCodec { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.debug_struct("RowCodec") - .field("rounds", &ROUNDS) - .finish_non_exhaustive() - } -} - -/// Evaluates one round function. -/// -/// Keyed SipHash-2-4 of the right half, truncated by the caller to the half width. -#[expect( - clippy::cast_possible_truncation, - reason = "the caller masks to the half width; the narrowing keeps the used bits" -)] -#[expect( - clippy::little_endian_bytes, - reason = "the round function hashes one pinned byte order; the codec never crosses hosts" -)] -fn round(key: &[u8; 16], half: u32) -> u32 { - let mut hasher = SipHasher24::new_with_key(key); - hasher.write(&half.to_le_bytes()); - hasher.finish() as u32 -} diff --git a/libs/@local/graph/atlas/src/serve/colour.rs b/libs/@local/graph/atlas/src/serve/colour.rs deleted file mode 100644 index ee9f87b71b2..00000000000 --- a/libs/@local/graph/atlas/src/serve/colour.rs +++ /dev/null @@ -1,191 +0,0 @@ -//! Type colouring. -//! -//! Resolving `coloredTypeIds` to per-type memberships with descendant expansion, the `TYPE_MASK` -//! column's serving source. -//! -//! Each requested id arrives as a [`VersionedUrl`] - the transport parses the request body, so it -//! rejects a malformed entry before assembly begins. [`Palette::of`] derives each entry's ontology -//! identity once at the assembly boundary, and everything after it carries those trivially copyable -//! identities. Every resolution reads the generation's snapshot. The identity joins the -//! generation's ontology identity table, and the resulting row names a closure-map row whose set -//! bits are every type the request matches - the type itself and all its descendants. The tile path -//! never consults the live store. -//! -//! Failure to resolve stays legal. A well-formed id this generation never ingested (a client -//! ontology newer than the snapshot, or a corpus whose ontology ids are not store identities) reads -//! as zero bits in every point's mask and never as an error. - -use hashql_core::id::bit_vec::{BitRelations as _, RowRef}; -use type_system::ontology::id::VersionedUrl; - -use super::Atlas; -use crate::{ - bitset::DenseBitSlice, - identity::{BasePosition, OntologyRowId}, - postgres::id::ArchivedOntologyTypeUuid, - salt::{ - fit::prepare::identity::IdentityTableArchive, - postings::{artifact::PostingsArchive, closure::ClosureMap}, - }, -}; - -/// One request's colour palette. -/// -/// Each slot is one `coloredTypeIds` entry as an ontology identity. Slot order is the request's: -/// `TYPE_MASK` bit `i`, mask source `i`, and palette slot `i` are one request entry. -#[derive(Debug)] -pub(super) struct Palette { - entries: Vec, -} - -impl Palette { - /// Derives the palette of the request's ids, one slot per entry. - pub(super) fn of(ids: &[VersionedUrl]) -> Self { - Self { - entries: ids.iter().map(ArchivedOntologyTypeUuid::from_url).collect(), - } - } - - /// Returns whether the palette carries no entry. - pub(super) const fn is_empty(&self) -> bool { - self.entries.is_empty() - } - - /// Returns whether `url` names a palette entry. - /// - /// Comparison is by ontology identity - any [`VersionedUrl`] naming the same versioned type - /// covers it, matching the mask resolution's own derivation. - pub(super) fn covers(&self, url: &VersionedUrl) -> bool { - self.entries - .contains(&ArchivedOntologyTypeUuid::from_url(url)) - } -} - -/// The mask sources of one request's `coloredTypeIds`, in request order. -/// -/// Materialized unions live here so the [`Membership`] views the encoder consumes can borrow them -/// beside the mapped postings. -/// -/// [`Membership`]: crate::salt::postings::artifact::Membership -#[derive(Debug)] -pub(super) struct MaskSet { - sources: Vec, -} - -/// One requested id's resolved membership source. -#[derive(Debug)] -enum MaskSource { - /// The id resolved to a type without proper descendants: the stored membership serves directly. - Stored(OntologyRowId), - /// The id resolved to a type with proper descendants. - /// - /// The dense union of the closure row's memberships, as one bit set frame - the postings' own - /// dense vocabulary, so stored and materialized masks serve through one view. - Union(Box>), - /// The id resolved to no type in this generation: zero bits. - Unresolved, -} - -impl MaskSet { - /// Views the sources as the encoder's membership slice, in request order. - pub(super) fn memberships<'doc>( - &'doc self, - postings: &'doc PostingsArchive, - ) -> Vec> { - use crate::salt::postings::artifact::Membership; - - self.sources - .iter() - .map(|source| match source { - MaskSource::Stored(row) => postings - .membership(*row) - .expect("resolved rows lie inside the postings' type domain"), - MaskSource::Union(words) => Membership::Dense(words), - MaskSource::Unresolved => Membership::List(&[]), - }) - .collect() - } -} - -impl Atlas { - /// Resolves one request's palette into mask sources, in slot order. - pub(super) fn resolve_masks(&self, palette: &Palette) -> MaskSet { - resolve_masks(&self.postings, &self.closure, &self.ontology_ids, palette) - } -} - -/// Materializes the dense union of every set type's membership. -/// -/// One bit set frame over base positions: the postings' own dense vocabulary, so stored and -/// materialized masks serve through one view. -fn union_membership( - postings: &PostingsArchive, - descendants: RowRef<'_, OntologyRowId>, -) -> Box> { - use crate::salt::postings::artifact::Membership; - - let points = usize::try_from(postings.points()).expect("point domains fit usize"); - let mut bitmap = DenseBitSlice::new_empty(points); - - for type_row in &descendants { - let membership = postings - .membership(type_row) - .expect("closure rows lie inside the postings' type domain"); - - match membership { - Membership::Dense(set) => { - bitmap.union(set); - } - Membership::List(positions) => { - for &position in positions { - bitmap.insert(position); - } - } - } - } - - bitmap -} - -/// Resolves one palette identity: ontology row to the closure row's membership union. -fn resolve_mask( - postings: &PostingsArchive, - closure: &ClosureMap, - table: &IdentityTableArchive, - key: ArchivedOntologyTypeUuid, -) -> MaskSource { - let Some(row) = table.row_of(key) else { - return MaskSource::Unresolved; - }; - - let descendants = closure - .descendants(row) - .expect("identity rows share the postings' type domain"); - - // Every closure row carries its own bit, so one set bit means no - // proper descendant exists and the stored membership is the whole - // match set. - if descendants.count() == 1 { - return MaskSource::Stored(row); - } - - MaskSource::Union(union_membership(postings, descendants)) -} - -/// Resolves a palette against one generation's postings. -/// -/// Closure map, and ontology identities, in slot order. -pub(super) fn resolve_masks( - postings: &PostingsArchive, - closure: &ClosureMap, - table: &IdentityTableArchive, - palette: &Palette, -) -> MaskSet { - MaskSet { - sources: palette - .entries - .iter() - .map(|&key| resolve_mask(postings, closure, table, key)) - .collect(), - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/consumer.rs b/libs/@local/graph/atlas/src/serve/delta/consumer.rs deleted file mode 100644 index c739e97f356..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/consumer.rs +++ /dev/null @@ -1,503 +0,0 @@ -//! The poll arm feeding a generation's delta publications. -//! -//! [`DeltaConsumer`] is the one writer behind a [`DeltaCell`], a long-lived task built beside the -//! cell when serving starts. Each poll reads the entity feed window and folds it into the register, -//! and a fresh snapshot publishes whenever resolution changed. Requests keep loading whatever the -//! cell holds, so a slow or failing poll degrades freshness and nothing else. -//! -//! After the fold, a poll classifies its unclassified arrivals in one batched read at the store's -//! present. The read fails closed. A failed batch logs a warning and stays unclassified in the -//! register, so the next poll retries it, while publication proceeds and the watermark advances, -//! because the watermark tracks folded feed events and the register still holds the identities. -//! A link verdict with an incomplete attachment pair registers no edge under a warning, never -//! falling back to the node pipeline, because the type test already excludes it from the node -//! scope's law. Retries need no budget: the read is one indexed batch per poll with no spend to -//! bound. -//! -//! Each poll also drains the placement channel the staging arm feeds, folding every projected -//! placement into the register before the publication decision. The consumer stays the register's -//! one writer, and a drained placement publishes at the same poll that received it, so an -//! arrival's coordinate reaches serving within one poll interval of its projection. -//! -//! Each poll reads events after the held watermark minus the safety lag. The lag covers the feed's -//! commit-visibility window. An event becomes visible up to a whole write request after its -//! recorded time, with concurrent writers adding their clock skew on top, so a poll reading only -//! past its newest seen time would miss late commits. The trailing window re-delivers events the -//! register already applied, and the fold ignores them by version. The clock model lives with the -//! feed, [`PostgresStore::entity_events_since`]. -//! -//! The first poll is the init replay. Its watermark starts at the fit's own transaction-time point, -//! so the read covers every change the serving generation cannot know about. The replay logs its -//! event count, its duration, and the register's resident-byte estimate - the telemetry that sizes -//! replay against the persistent-store escape. -//! -//! A failed poll publishes nothing and leaves the held snapshot serving. The next tick retries from -//! the same watermark, so an error skips no events. -//! -//! [`PostgresStore::entity_events_since`]: hash_graph_postgres_store::store::PostgresStore::entity_events_since - -use alloc::sync::Arc; -use core::{fmt, pin::pin, time::Duration}; -use std::time::Instant; - -use error_stack::Report; -use futures::TryStreamExt as _; -use hash_graph_postgres_store::store::{ - AsClient, EntityEvent, PostgresStorePool, error::StoreError, -}; -use hash_graph_store::{error::QueryError, pool::StorePool as _}; -use hash_graph_temporal_versioning::{Timestamp, TransactionTime}; -use hashql_core::collections::FastHashMap; -use tokio::{sync::mpsc::Receiver, time::MissedTickBehavior}; - -use super::{ - DeltaCell, DeltaEvent, DeltaRegister, DeltaRevision, Disposition, IdentityTables, - ProjectedArrival, -}; -use crate::postgres::{ - Classification, classify_entities, edition_display::DisplayParts, id::ArchivedEntityId, - read_edition_displays, -}; - -/// The consumer's polling knobs. -/// -/// The serve flags read their defaults from here, so the values live in exactly one place and -/// `--help` renders them. -#[derive(Debug, Copy, Clone, Default)] -pub(crate) struct DeltaPolling { - /// How long the consumer waits between polls. - pub interval: Duration = Duration::from_secs(5), - /// How far behind its own watermark a poll starts reading. - /// - /// The feed's contract requires the lag to exceed the - /// longest write request plus the largest cross-writer clock skew. Nobody has measured either - /// bound, so the default stands wide until a measurement revises it downward. Query - /// affordability cannot establish event completeness, so a tighter lag is an accepted-risk - /// decision rather than a performance tuning. - pub safety_lag: Duration = Duration::from_secs(60), - /// How many consecutive staging cycles read for a pending arrival's embedding on each side - /// of its ensure. - /// - /// One budget governs both sides, so the reading budget's - /// exhaustion submits the ensure and the post-ensure budget's exhaustion parks the arrival - /// until reconciliation or refit. - pub retry_polls: u32 = 12, - /// How many pending arrivals the consumer will backlog before parking. - pub placement_backlog: usize = 1024, -} - -/// One poll failed against the store. -#[derive(Debug)] -pub(crate) enum PollError { - /// No connection was available for the poll. - Connect(Report), - /// The feed statement failed or one of its rows did not decode. - Feed(Report), -} - -impl fmt::Display for PollError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Connect(report) => { - write!(fmt, "the delta poll reached no store connection: {report}") - } - Self::Feed(report) => write!(fmt, "the entity feed read failed: {report}"), - } - } -} - -impl core::error::Error for PollError {} - -/// What one successful poll did. -#[derive(Debug, Copy, Clone)] -struct PollReading { - /// The events the feed window delivered. - events: usize, - /// Whether the poll published a fresh snapshot. - published: bool, -} - -/// One poll's running fold over the feed stream. -#[derive(Debug, Default)] -pub(super) struct PollOutcome { - /// The events the feed window delivered. - events: usize, - /// Whether any application changed publication's resolution input. - changed: bool, - /// The newest event time read, or [`None`] for a quiet window. - watermark: Option>, -} - -impl PollOutcome { - /// Returns how many events the feed window delivered. - #[must_use] - pub(super) const fn events(&self) -> usize { - self.events - } - - /// Returns whether any application changed publication's resolution input. - #[must_use] - pub(super) const fn changed(&self) -> bool { - self.changed - } - - /// Returns the newest event time read, or [`None`] for a quiet window. - #[must_use] - pub(super) const fn watermark(&self) -> Option> { - self.watermark - } - - /// Folds one event into `register`, counting it whether or not the fold keeps it. - /// - /// The poll watermark moves on every event read, because an event the register ignores was - /// still delivered by the window, and re-reading it buys nothing. - pub(super) fn fold(&mut self, register: &mut DeltaRegister, event: &EntityEvent) { - let event = DeltaEvent::from(event); - - self.changed |= register.apply(event); - self.events += 1; - self.watermark = Some( - self.watermark - .map_or(event.version, |held| held.max(event.version)), - ); - } -} - -/// The long-lived poll arm owning one generation's delta fold. -/// -/// One consumer serves one generation for the serving process's lifetime. It holds the mutable -/// [`DeltaRegister`], and every other party reads the [`DeltaCell`] it publishes into. Dropping the -/// task drops the fold with it, and a restart rebuilds it by replay, so the process keeps no delta -/// state anywhere durable. -#[derive(Debug)] -pub(crate) struct DeltaConsumer { - /// The store pool the feed reads through. - pool: Arc, - /// The serving generation's identity tables, resolving the fold into row space. - tables: Arc, - /// The cell requests load snapshots from. - cell: Arc, - /// The polling knobs. - polling: DeltaPolling, - /// The channel carrying the staging arm's projected arrivals, bounded at - /// [`DeltaPolling::placement_backlog`]. - placements: Receiver<(ArchivedEntityId, ProjectedArrival)>, - /// The fold of every event applied so far. - register: DeltaRegister, - /// The newest event time folded through, the next poll's base. - watermark: Timestamp, - /// The next publication's name in publication order. - revision: DeltaRevision, - /// Whether the consumer has published any snapshot yet. - published: bool, -} - -impl DeltaConsumer -where - T: IdentityTables, -{ - /// Builds the consumer over a generation fitted at `fitted`. - /// - /// `fitted` is the transaction-time point the generation's dataset observed. The first poll - /// replays the feed from it, minus the safety lag, so the fold covers every store change the - /// fit could not see. `register` is the empty fold the caller built over the generation's - /// row bounds, where every allocation starts. - pub(crate) const fn new( - pool: Arc, - tables: Arc, - cell: Arc, - fitted: Timestamp, - register: DeltaRegister, - polling: DeltaPolling, - placements: Receiver<(ArchivedEntityId, ProjectedArrival)>, - ) -> Self { - Self { - pool, - tables, - cell, - polling, - placements, - register, - watermark: fitted, - revision: DeltaRevision::FIRST, - published: false, - } - } - - /// Polls until the owner drops the task. - /// - /// One tick per [`DeltaPolling::interval`], and a poll running past its tick delays the next - /// rather than bursting to catch up. A failed poll leaves the held snapshot serving, and the - /// next tick retries from the same watermark. - pub(crate) async fn run(mut self) -> ! { - let mut ticks = tokio::time::interval(self.polling.interval); - ticks.set_missed_tick_behavior(MissedTickBehavior::Delay); - - loop { - ticks.tick().await; - - let replaying = !self.published; - let started = Instant::now(); - match self.poll().await { - Ok(reading) if replaying => { - tracing::info!( - events = reading.events, - seconds = started.elapsed().as_secs_f64(), - resident_bytes = self.register.resident_estimate(), - "replayed the entity feed" - ); - } - Ok(reading) => { - if reading.events > 0 { - tracing::debug!( - events = reading.events, - published = reading.published, - resident_bytes = self.register.resident_estimate(), - "folded the entity feed window" - ); - } - } - Err(error) => { - tracing::warn!(%error, "a delta poll failed, the held snapshot keeps serving"); - } - } - } - } - - /// Classifies the register's unclassified arrivals, returning whether any verdict changed - /// publication's resolution input. - /// - /// One batched read against the store's present. Failure changes nothing beyond a warning: - /// the arrivals stay unclassified in the register and the next poll retries them, with the - /// poll itself still succeeding. An identity the read answers nothing about stays - /// unclassified the same way, until the feed's own withdrawal resolves it. - async fn classify_arrivals(&mut self, store: &impl AsClient) -> bool { - let unclassified: Vec<_> = self.register.unclassified(self.tables.as_ref()).collect(); - if unclassified.is_empty() { - return false; - } - - let verdicts = match classify_entities(store, unclassified.iter().copied()).await { - Ok(verdicts) => verdicts, - Err(error) => { - tracing::warn!( - %error, - arrivals = unclassified.len(), - "the classification read failed, unclassified arrivals retry next poll" - ); - return false; - } - }; - - // The statement relies on edge multiplicity the writers enforce rather than the schema: - // a duplicated edge row would fan one request into two verdict rows, so an answer count - // above the request count is the same broken-store signal as one below it. - if verdicts.len() > unclassified.len() { - tracing::warn!( - answers = verdicts.len(), - requests = unclassified.len(), - "the classification read answered more identities than the request named, a store \ - invariant the statement relies on broke" - ); - } - - let unanswered = unclassified.len().saturating_sub(verdicts.len()); - if unanswered > 0 { - tracing::debug!( - unanswered, - "identities without a current edition stay unclassified until the feed resolves \ - them" - ); - } - - let mut disposition = Disposition::AlreadyHeld; - for (entity, verdict) in verdicts { - if let Classification::Edge { source, target } = verdict - && (source.is_none() || target.is_none()) - { - tracing::warn!( - ?entity, - "a link arrival's attachment pair is incomplete, no edge registers" - ); - } - - match self.register.classify(entity, verdict) { - Ok(held) => disposition |= held, - Err(exhausted) => tracing::warn!( - ?entity, - %exhausted, - "the edge row allocator refused a verdict, the identity stays unclassified" - ), - } - } - disposition.changes_resolution() - } - - /// Captures the legends the register lists as pending, returning whether any capture landed. - /// - /// One batched read against the edition cache, keyed by edition, because an edition id - /// names one immutable row. Failure changes nothing beyond a warning: the captures stay - /// pending and the next poll retries them, with the poll itself still succeeding. The - /// statement answers every requested edition exactly once, so a short answer is a broken - /// store invariant rather than a lookup miss - warned loudly, because a permanently - /// unanswered link edition means publication withholds that link with no other signal. - async fn capture_displays(&mut self, store: &impl AsClient) -> bool { - let required: Vec<_> = self - .register - .pending_captures(self.tables.as_ref()) - .collect(); - if required.is_empty() { - return false; - } - - let answers = match read_edition_displays( - store, - required.iter().map(|&(_, edition)| edition), - ) - .await - { - Ok(answers) => answers, - Err(error) => { - tracing::warn!( - %error, - pending = required.len(), - "the display read failed, pending captures retry next poll" - ); - return false; - } - }; - - // The statement answers every requested edition exactly once - `UNNEST` yields one - // row per element, and both outer joins match at most one row each, through - // `entity_edition_cache`'s primary key and `ontology_ids`' unique (base_url, version) - // pair - a mismatched count is therefore a broken store invariant rather than a - // lookup miss. - if answers.len() != required.len() { - tracing::warn!( - answers = answers.len(), - pending = required.len(), - "the display read answered a different edition count than the listing named" - ); - } - - // An answer without a resolved representative type stays pending, so the next poll - // reads it again: a legend cannot exist until the representative resolves. - let displays: FastHashMap<_, _> = answers.into_iter().collect(); - let mut captured = false; - for (entity, edition) in required { - if let Some(Some(DisplayParts { - label, - icon, - representative, - })) = displays.get(&edition) - { - match self.register.capture_display( - entity, - edition, - label, - icon, - *representative, - self.tables.as_ref(), - ) { - Ok(()) => captured = true, - Err(exhausted) => tracing::warn!( - ?entity, - %exhausted, - "the ontology row allocator refused a capture, it stays pending" - ), - } - } - } - - captured - } - - /// Drains the staging arm's placement channel, returning whether any placement changed - /// publication's resolution input. - /// - /// Every queued placement folds into the register in channel order, which is what makes the - /// row assignment follow placement order. A placement for an already-placed identity - /// changes nothing, because the first coordinate never moves, and a placement for a - /// withdrawn identity records without publishing, so the drain never grows the served set on - /// its own. A refused allocation warns and drops the placement: the arrival stays staged, - /// and the refusal repeats at every later projection until a refit retires the register. - fn drain_placements(&mut self) -> bool { - let mut disposition = Disposition::AlreadyHeld; - - while let Ok((entity, arrival)) = self.placements.try_recv() { - match self.register.place(entity, &arrival, self.tables.as_ref()) { - Ok(placed) => disposition |= placed, - Err(exhausted) => { - tracing::warn!( - ?entity, - %exhausted, - "the row allocator refused a placement, the arrival stays staged" - ); - } - } - } - - disposition.changes_resolution() - } - - /// Runs one poll, folding the feed window and publishing when resolution changed. - /// - /// The first successful poll publishes unconditionally, so an empty fold still states its - /// watermark, and a cell holding [`None`] means no poll has completed rather than nothing - /// withdrawn. - /// - /// # Errors - /// - /// Returns [`PollError::Connect`] when no connection was available for the poll, and - /// [`PollError::Feed`] when the feed statement failed or one of its rows did not decode. - async fn poll(&mut self) -> Result { - let safety_lag = time::Duration::try_from(self.polling.safety_lag).expect( - "the safety lag should be in the range of minutes, which should be able to be \ - converted to its time equivalent", - ); - let store = self - .pool - .acquire(None) - .await - .map_err(|report| PollError::Connect(report.change_context(StoreError)))?; - - let mut outcome = PollOutcome::default(); - { - // The feed executes as a fresh prepare on every call: a reused prepared statement - // flips to a generic plan around its sixth execution, and this cadence crosses that - // count within half a minute of startup. - let mut events = pin!(store.entity_events_since(self.watermark - safety_lag)); - while let Some(event) = events.try_next().await.map_err(PollError::Feed)? { - outcome.fold(&mut self.register, &event); - } - } - - let classified = self.classify_arrivals(&store).await; - let captured = self.capture_displays(&store).await; - - // Nothing further reads the store: return the connection before resolving the publication. - drop(store); - - let placed = self.drain_placements(); - - if let Some(watermark) = outcome.watermark() { - self.watermark = watermark; - } - - let publishing = outcome.changed() || classified || captured || placed || !self.published; - if publishing { - self.cell.publish(self.register.snapshot( - self.tables.as_ref(), - self.revision, - self.watermark, - )); - self.revision = self.revision.next(); - self.published = true; - } - - Ok(PollReading { - events: outcome.events(), - published: publishing, - }) - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/mod.rs b/libs/@local/graph/atlas/src/serve/delta/mod.rs deleted file mode 100644 index 3abfbb3a61d..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/mod.rs +++ /dev/null @@ -1,366 +0,0 @@ -//! The generation-local register folding the entity feed against the serving generation. -//! -//! A generation serves immutably from its fit-time snapshot while the store keeps moving underneath -//! it. The entity feed ([`EntityEvent`]) reports each change to an entity's published present, and -//! this module folds those events into one standing per entity. Publication resolves the fold -//! against the serving generation, so a deleted entity stops rendering and a new one can stage for -//! placement without waiting for a refit. A refit retires the register, because a fresh fit already -//! reflects everything the feed reported. -//! -//! An entity's *standing* is what its newest feed event implies for serving. Events that remove the -//! entity from the served corpus - a purge, a present ending without a successor, or an archived -//! edition becoming current - write [`Standing::Withdrawn`], and every other event writes -//! [`Standing::Live`] carrying the edition it observed. Because an unarchive is an ordinary update -//! whose edition is not archived, it replaces a tombstone with a live standing through the same -//! fold, with no special case. -//! -//! [`DeltaRegister`] holds the newest standing per identity, last-writer-wins on the event's -//! transaction time. The fold takes the maximum by version, breaking ties by standing rank and then -//! by edition between two live standings. The maximum makes the fold order-independent, so a -//! post-restart replay converges to the incremental register it replaces however equal-key events -//! interleave. -//! -//! An arrival's *classification* is the node-versus-link verdict -//! ([`Classification`](crate::postgres::Classification)) a batched -//! store lookup returns for it. The register holds the first verdict per identity for the -//! process's lifetime and never replaces it. Endpoint rows are immutable and an edition change -//! cannot flip the link category, so a re-read buys nothing. An arrival without a verdict serves -//! nothing and enters no pipeline, and the lookup retries at the next poll, so a failed read -//! degrades arrival freshness and nothing else while publication proceeds and the watermark -//! advances. -//! -//! An arrival's *placement* ([`ProjectedArrival`]) is the wire coordinate the staging arm -//! projects for it through the generation's own publish path. The register keeps the first -//! placement per identity and never replaces it, so a later edition moves the coordinate nowhere. -//! The refit recalibrates placements exactly as it recalibrates fitted rows, whose coordinates -//! are also fit-time content and often editions old. A placement recorded while the identity -//! stands withdrawn serves nothing, and an unarchive republishes the recorded coordinate at its -//! next publication. -//! -//! An arrival's *slot* is the row id its first placement takes: the next id past the accepted -//! [`Universe`](crate::serve::codec::Universe), assigned in placement order. The register never -//! reassigns or reuses a slot, -//! withdrawal included, so a row a proof admitted cannot change meaning while that proof can -//! answer, and an unarchived identity serves from its own former slot. Slot order and -//! level-of-detail ranking order are distinct contracts, and neither derives from the other. The -//! wire permutation covers the whole `u32` range, so widening the accepted universe preserves -//! every existing wire id. A placement that would grow the universe past that range refuses as -//! [`UniverseExhausted`](self::register::UniverseExhausted) and the arrival stays staged. -//! -//! A complete link's *edge row* is the id its first classification takes: the next row past the -//! accepted edge universe, assigned at the verdict's hold under the same never-reassign law. -//! The wire therefore speaks rows in the edge domain exactly as in the node domain. A hold that -//! would grow the edge universe past the codec's range refuses the whole verdict as -//! [`UniverseExhausted`](self::register::UniverseExhausted), and the identity stays unclassified -//! for the next poll to retry. -//! -//! A [`DeltaEpoch`] names one register's lifetime. Slot assignment is process-local and a -//! replay does not reproduce placement order, so a wire id allocated under one register must not -//! resolve under another. Consumer initialization draws a fresh epoch, the token authority seals -//! it into every token it issues, and a token sealed under any other epoch receives the uniform -//! authorization refusal. A restart is therefore a new epoch by construction: old slot mappings -//! die with the process instead of quietly renaming entities. -//! -//! [`DeltaRegister::snapshot`] publishes the fold as an immutable [`DeltaSnapshot`], resolving each identity against the generation's [`IdentityTables`]: -//! -//! - Withdrawn and fitted: the identity enters the withdrawn set, and its node or edge row enters -//! the matching row bitset - what admission subtraction reads per request and a scoped cache -//! entry folds out of its masks at resolution. -//! - Withdrawn and unfitted: the identity enters the withdrawn set alone. No generation row exists -//! to subtract, and a retained cohort can still hold the identity, so membership in the set never -//! depends on generation fitness. -//! - Live and fitted: nothing resolves. The generation already publishes the entity, and the -//! register entry exists to outrank older events and to clear a former tombstone. -//! - Live and unfitted: an arrival, resolved through its held classification. A node without a -//! placement stages with its edition, a node with one publishes its recorded wire coordinate -//! under its newest edition, and a complete link publishes with its endpoint identities. An -//! arrival holding no verdict, and a link missing an endpoint, publish nowhere. -//! -//! # Determinism -//! -//! A request loads one snapshot through [`DeltaCell::load`] and reads that one at every admission -//! in its answer, so the delta-sensitive assembly stays a pure function of the generation, the -//! request, the visibility proof, and the snapshot. A cache promising current delta semantics -//! applies the snapshot after lookup rather than carrying [`DeltaSnapshot::revision`] in its key. -//! -//! # Freshness -//! -//! A snapshot reflects the feed up to [`DeltaSnapshot::watermark`] rather than the store's present. -//! The clock model and the safety lag a consumer subtracts live with the feed, -//! [`PostgresStore::entity_events_since`], together with the writers the feed cannot represent. -//! Erase and snapshot restore move the store in ways a watermark never revisits, so reconciliation -//! against the store - today, the refit - bounds how long a fold can stay wrong. -//! -//! [`PostgresStore::entity_events_since`]: hash_graph_postgres_store::store::PostgresStore::entity_events_since -#![expect( - clippy::empty_enums, - reason = "zerocopy's FromBytes derive expands to an empty enum for its validation machinery" -)] - -use alloc::sync::Arc; - -use arc_swap::{ArcSwapOption, Guard}; -use hash_graph_postgres_store::store::EntityEvent; -use hash_graph_temporal_versioning::{Timestamp, TransactionTime}; -use rand::TryCryptoRng; -use type_system::knowledge::entity::id::EntityEditionId; - -use crate::{ - dataset::auxiliary::{OwnedIcon, OwnedLabel, OwnedLegend}, - identity::{EdgeRowId, NodeRowId, OntologyRowId}, - math::Vec2, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, -}; - -pub(crate) mod consumer; -pub(crate) mod overlay; -mod placement; -mod register; -mod snapshot; -pub(crate) mod staging; - -#[expect( - unused_imports, - reason = "published for the replay report's CLI adapter, which binds Placer::project's \ - outcomes; its registration consumes them" -)] -pub(crate) use self::placement::{NonFiniteProjection, Projection}; -pub(crate) use self::{ - placement::{PlacementError, Placer}, - register::{DeltaRegister, Disposition}, - snapshot::{DeltaSnapshot, PlacementCohort}, -}; - -#[cfg(test)] -mod tests; - -/// Where an entity stands in the served corpus, per its newest feed event. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum Standing { - /// The entity's published present is current under the carried edition. - Live { - /// The edition current at the event, read in the same store snapshot that observed the - /// change. - edition: EntityEditionId, - }, - /// The entity left the served corpus: purged, its present ended, or its current edition - /// archived. - Withdrawn, -} - -/// One feed event resolved into the standing it implies for serving. -/// -/// Conversion is the one place feed vocabulary becomes register vocabulary. A purge and an ended -/// present withdraw the entity at the event's own time, and an update decides between a live -/// standing and a withdrawal by the archived flag its edition carries, resolved inside the feed -/// statement itself. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct DeltaEvent { - /// The entity whose standing the event states. - entity: ArchivedEntityId, - /// When the change occurred, on the transaction-time axis. - version: Timestamp, - /// The standing the event leaves the entity in. - standing: Standing, -} - -impl From<&EntityEvent> for DeltaEvent { - fn from(event: &EntityEvent) -> Self { - match event { - EntityEvent::Updated(update) => Self { - entity: update.entity.into(), - version: update.changed_at, - standing: if update.archived { - Standing::Withdrawn - } else { - Standing::Live { - edition: update.edition, - } - }, - }, - EntityEvent::Ended(end) => Self { - entity: end.entity.into(), - version: end.ended_at, - standing: Standing::Withdrawn, - }, - EntityEvent::Deleted(deletion) => Self { - entity: deletion.entity.into(), - version: deletion.provenance.deleted_at_transaction_time, - standing: Standing::Withdrawn, - }, - } - } -} - -/// One register lifetime's name, sealed into every authority token issued while it lives. -/// -/// The value is random rather than derived, so two register lifetimes over one generation match -/// only by a collision of 128 fresh bits. Equality is the whole interface: the token authority -/// seals the epoch it holds and refuses any other at open. -#[derive( - Debug, - Copy, - Clone, - zerocopy::ByteEq, - zerocopy::ByteHash, - zerocopy::IntoBytes, - zerocopy::FromBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, -)] -#[repr(transparent)] -pub(crate) struct DeltaEpoch([u8; 16]); - -impl DeltaEpoch { - /// Draws a fresh epoch. - /// - /// # Errors - /// - /// Returns the generator's error when the draw fails: entropy failure refuses initialization - /// rather than starting a register lifetime under a predictable name. - pub(crate) fn fresh(rng: &mut R) -> Result { - let mut bytes = [0_u8; 16]; - rng.try_fill_bytes(&mut bytes)?; - Ok(Self(bytes)) - } -} - -/// One publication of the register, in publication order. -/// -/// The revision names a publication in the determinism contract: response bytes are identical for -/// an identical generation, request, visibility proof, and snapshot revision. -#[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)] -pub(crate) struct DeltaRevision(u64); - -impl DeltaRevision { - /// The first publication's revision. - pub(crate) const FIRST: Self = Self(0); - - /// Returns the revision following this one. - #[must_use] - pub(crate) const fn next(self) -> Self { - Self(self.0 + 1) - } -} - -/// The serving generation's identity tables, as publication resolves against them. -/// -/// The register keys by entity identity while subtraction runs in row space, so publication -/// resolves each identity into the generation's row domains. An identity the generation never -/// fitted resolves in neither domain. The node and edge domains are disjoint, the fit's own scope -/// law excluding link-typed entities from the node scope, so an identity resolves in at most one. -pub(crate) trait IdentityTables { - /// Returns the node row carrying `id`, or [`None`] when the generation fitted none. - fn node_row_of(&self, id: ArchivedEntityId) -> Option; - - /// Returns the edge row carrying `id`, or [`None`] when the generation fitted none. - fn edge_row_of(&self, id: ArchivedEntityId) -> Option; - - /// Returns the ontology row carrying `id`, or [`None`] when the generation tabulated none. - fn ontology_row_of(&self, id: ArchivedOntologyTypeUuid) -> Option; -} - -/// One arrival's projection, recorded at its first success and awaiting its row. -/// -/// The coordinate is in the wire frame, the domain every served response speaks. The edition is -/// the one whose stored embedding produced the coordinate, which a later edition never moves, so -/// the pair records exactly what the projection read. The display parts travel beside the -/// coordinate, read from the same edition's cached row, and share the coordinate's staleness -/// class: a later edition moves neither, and the refit repairs both. The representative type -/// stays a store uuid here, because the register resolves it into its ontology row at placement, -/// where the row fact is established. -#[derive(Debug, Clone, PartialEq)] -pub(crate) struct ProjectedArrival { - /// The edition whose embedding the projection read. - pub edition: EntityEditionId, - /// The projected coordinate, normalized into the wire frame. - pub position: Vec2, - /// The display label read beside the coordinate. - pub label: OwnedLabel, - /// The representative type's nearest declared icon, read beside the coordinate. - /// - /// It rides to the register so an allocation for a type the generation never tabulated - /// records the icon beside the row it allocates. - pub icon: OwnedIcon, - /// The representative type read beside the coordinate. - pub representative: ArchivedOntologyTypeUuid, -} - -/// A placed arrival as publication serves it, carrying its row, coordinate, and legend. -/// -/// The coordinate is the identity's first successful projection, which a later edition never -/// moves. The edition is the newest the feed observed, the key detail reads resolve through. -/// The row is the one the identity's first placement allocated, fixed for the register's -/// lifetime. The legend is the placement's capture, sharing the coordinate's staleness class. -#[derive(Debug, Clone, PartialEq)] -pub(crate) struct DeltaNode { - /// The arrival's row id in the extended universe. - pub id: NodeRowId, - /// The arrival's newest feed edition. - pub edition: EntityEditionId, - /// The projected coordinate, normalized into the wire frame. - pub position: Vec2, - /// The display payload captured at placement. - pub legend: OwnedLegend, -} - -impl DeltaNode { - fn with_edition(self, edition: EntityEditionId) -> Self { - Self { edition, ..self } - } -} - -/// A live post-fit link as publication serves it, carrying its allocated edge row and its -/// row-typed endpoints. -/// -/// The row is the one the link's classification allocated, fixed for the register's lifetime, -/// so the wire speaks one edge-row domain across fitted and delta links. The endpoints are node -/// rows in the accepted universe - a fitted endpoint's generation row, or the slot an arrival's -/// placement took - resolved at publication, so a link publishes once both endpoints hold -/// rows. The edition is the link's newest feed edition, the key its detail reads resolve -/// through. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct DeltaEdge { - /// The link's edge row in the extended universe. - pub id: EdgeRowId, - /// The link's newest feed edition. - pub edition: EntityEditionId, - /// The left attachment's endpoint row. - pub source: NodeRowId, - /// The right attachment's endpoint row. - pub target: NodeRowId, -} - -/// The published snapshot, as requests load it. -/// -/// One cell serves one generation. A publication swaps the held snapshot whole, and a load clones -/// the [`Arc`], so a request reads one snapshot across its whole answer however many publications -/// land while it runs. A cell holding [`None`] means no poll has completed, and a request subtracts -/// nothing and pays nothing. -#[derive(Debug, Default)] -pub(crate) struct DeltaCell { - /// The current publication. - snapshot: ArcSwapOption, -} - -impl DeltaCell { - /// Returns the current publication, or [`None`] before the first. - pub(crate) fn load(&self) -> Guard>> { - self.snapshot.load() - } - - /// Returns an owned handle on the current publication, or [`None`] before the first. - /// - /// The owned handle pins one publication for as long as the caller holds it, the shape a - /// whole request reads through. A read that ends before its next await point takes - /// [`Self::load`] instead, whose guard is cheaper than the reference-count round trip. - pub(crate) fn load_full(&self) -> Option> { - self.snapshot.load_full() - } - - /// Publishes `snapshot`, replacing the held publication. - pub(crate) fn publish(&self, snapshot: DeltaSnapshot) { - self.snapshot.store(Some(Arc::new(snapshot))); - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/overlay.rs b/libs/@local/graph/atlas/src/serve/delta/overlay.rs deleted file mode 100644 index 96593be8ec7..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/overlay.rs +++ /dev/null @@ -1,120 +0,0 @@ -//! Identity-table extensions: rows the delta allocates past a generation's baked bound. -//! -//! A generation's identity tables are immutable files, and the delta observes identities the fit -//! never saw. Each row domain the delta grows gets one overlay, which owns everything past the -//! baked bound - the allocated identities' rows with their reverse index - and derives the -//! accepted universe the two sides span. Rows below the bound stay the generation's own, so a -//! lookup composes the table's answer with the overlay's and a reader learns nothing about which -//! side answered. -//! -//! Growth is the register's alone. [`IdentityTableOverlay::resolve`] hands an identity its row, -//! allocating the next row past the current universe on first sight, so the extension is dense, -//! insert-only, and reproducible from its allocation order. Publication clones the overlay into -//! the snapshot, which pins the extension for every read taken against that publication. - -use hashql_core::{ - collections::{FastHashMap, fast_hash_map}, - id::Id, -}; - -use crate::serve::codec::Universe; - -/// One row domain's extension past its baked identity table. -/// -/// The bound at construction is the baked table's length, and every allocation grows the -/// universe by one row, so [`universe`](Self::universe) always spans the baked rows and the -/// allocated rows with no second counter. The identity type `K` matches the table this overlay -/// extends, and the row type `R` names the domain. -#[derive(Debug, Clone)] -pub(crate) struct IdentityTableOverlay { - /// The accepted row universe, the baked bound grown by one per allocation. - universe: Universe, - /// The allocated identities' rows, keyed by identity. - delta: FastHashMap, - /// The allocated identities in allocation order, indexed by row past the baked bound. - delta_reverse: Vec, -} - -impl IdentityTableOverlay -where - K: Copy + Eq + core::hash::Hash, - R: Id, -{ - /// Opens the overlay past `bound` with an empty extension. - pub(crate) fn new(bound: Universe) -> Self { - Self { - universe: bound, - delta: fast_hash_map(), - delta_reverse: Vec::new(), - } - } - - /// Returns the allocated row carrying `id`, or [`None`] when no allocation holds it. - /// - /// The baked rows answer from the generation's own table, so a caller resolving across both - /// sides consults the table first and this second. - #[must_use] - pub(crate) fn row_of(&self, id: K) -> Option { - self.delta.get(&id).copied() - } - - /// Returns the identity of the allocated row `row`, or [`None`] outside the extension. - /// - /// Rows below the baked bound answer [`None`] here and their identity from the generation's - /// own table, so the two sides partition the universe. - #[must_use] - pub(crate) fn id_of(&self, row: R) -> Option { - let bound = self.bound(); - let index = usize::try_from(row.as_u64().checked_sub(bound)?).ok()?; - self.delta_reverse.get(index).copied() - } - - /// Returns the row carrying `id`, allocating the next row past the universe on first sight. - /// - /// [`None`] is the row domain's own end: the id type has no next value to allocate. The wire - /// codec's `u32` row domain is narrower, and its holder enforces it at the allocation call - /// site, because domains this type serves without a wire codec carry no such bound. - pub(crate) fn resolve(&mut self, id: K) -> Option { - if let Some(row) = self.row_of(id) { - return Some(row); - } - - let (universe, row) = self.universe.grow()?; - self.universe = universe; - self.delta.insert(id, row); - self.delta_reverse.push(id); - Some(row) - } - - /// Returns the accepted row universe: the baked rows and every allocated row. - #[must_use] - pub(crate) const fn universe(&self) -> Universe { - self.universe - } - - /// Estimates the extension's resident bytes: the forward map and the reverse index. - #[must_use] - pub(crate) fn resident_estimate(&self) -> usize { - self.delta.allocation_size() + self.delta_reverse.capacity() * size_of::() - } - - /// Returns the baked bound: the first row the extension may hold. - const fn bound(&self) -> u64 - where - R: [const] Id, - { - self.universe.size() as u64 - self.delta_reverse.len() as u64 - } -} - -impl PartialEq for IdentityTableOverlay -where - K: Copy + Eq + core::hash::Hash, - R: Id, -{ - fn eq(&self, other: &Self) -> bool { - self.universe == other.universe - && self.delta_reverse == other.delta_reverse - && self.delta == other.delta - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/placement.rs b/libs/@local/graph/atlas/src/serve/delta/placement.rs deleted file mode 100644 index 9894c0b8256..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/placement.rs +++ /dev/null @@ -1,680 +0,0 @@ -//! The online projection placing arrivals through the generation's own publish path. -//! -//! A fitted row's published coordinate is not the checkpoint forward alone. The fit selects the -//! canonical step and applies that step's recorded similarity to every projected point, and the -//! aligned field then normalizes through the world frame onto the wire. [`Placer`] repeats -//! exactly that construction for one arrival. The checkpoint opens against the architecture the -//! metadata document echoes, the forward runs at the recorded canonical condition, and the -//! recorded alignment and world frame carry the point onto the wire. A generation whose ladder -//! never measured - a corpus without relation force - published the zero-condition frame -//! directly, and the placer repeats that arm the same way, forwarding at zero and skipping the -//! alignment step. -//! -//! The published generation records every input the construction needs. The architecture -//! and the forward slice bound come from the configuration echo, the canonical condition and its -//! alignment from the ladder evidence, and the world frame from the level-of-detail measurements, -//! so the placer needs no configuration of its own and cannot drift from the fit that published -//! the coordinates it extends. -//! -//! Construction certifies the replay before any arrival trusts it. The placer projects a sample -//! of the generation's own fitted rows and compares the aligned result against the published -//! coordinate column, per component in world units, under the ladder report's own reproduction -//! bound. A checkpoint, an echo, or a backend that does not reproduce the published bytes within -//! that bound would place arrivals on a lookalike map, so certification failure refuses the -//! serve. -//! -//! Every refusal fails closed, each under its own log line. [`Placer::open`] answers `Ok(None)` -//! for a baseline-placed generation alone, the one shape that promises no publish path, and -//! serving without a placer stages arrivals forever, the same disposition as a deployment -//! without a Temporal client. A generation that stages a projector checkpoint must reopen it: -//! every reopening failure is a [`PlacementError`], and the serve refuses to start rather than -//! silently staging every arrival. At runtime, an arrival projecting outside the fitted world -//! frame stays unplaced under a warning, because a clamped coordinate would serve a lie about -//! position. The frame comes from fit-time data, and the next refit recalibrates it. - -use std::fs::File; - -use hashql_core::id::{Id as _, IdSlice}; - -use crate::{ - dataset::PROJECTOR_DIMENSIONS, - device::{Inference, PhysicalDevice}, - file::{array::ArrayFile, generation::Generation}, - math::{AlignedVecN, Bounds2, MatrixN, NonNegative, Similarity, Vec2}, - salt::{ - fit::PlacementOptions, - lod::stage::WIRE_FRAME, - projector::{ - artifact::{self, CERTIFICATE_TOLERANCE}, - model::{NodeRole, Projector}, - train::{batch::NodeColumns, refresh}, - }, - }, -}; - -hashql_core::id::newtype! { - /// A row of one projection batch. - /// - /// The ordinal is batch-local: it names a position in the slice handed to one forward call - /// and nothing beyond it. - #[id(const)] - pub struct BatchRow(u32) -} - -/// Rows per projection scratch buffer. -/// -/// The bound sizes the aligned staging copy one forward call reads from. A larger batch loops. -const SCRATCH_ROWS: usize = 256; - -/// Fitted rows the construction certificate projects. -/// -/// The sample spreads evenly over the row domain, and rows project independently, so each sampled -/// row is its own reproduction check. The count keeps certification to a fraction of a second -/// while still crossing the whole domain. -const CERTIFICATE_ROWS: usize = 1024; - -/// One arrival's projection outcome. -#[derive(Debug, Copy, Clone, PartialEq)] -pub(crate) enum Projection { - /// The world coordinate lies inside the fitted frame, normalized onto the wire. - Placed { - /// The projected coordinate in the wire frame. - wire: Vec2, - }, - /// The world coordinate lies outside the fitted frame, so the arrival stays unplaced until a - /// refit recalibrates the frame. - OutOfFrame { - /// The aligned coordinate in world units, for the caller's log line. - world: Vec2, - }, -} - -/// One projection batch failed on a non-finite coordinate. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct NonFiniteProjection { - /// The failing row's position in the batch the caller handed over. - pub row: usize, -} - -/// The refusals that stop a staged projector checkpoint from reopening. -/// -/// Each case logs its own line at the refusal site. The error names the case for the serve -/// refusal that carries it. A baseline-placed generation is not a refusal: it promises no -/// publish path, and [`Placer::open`] answers `Ok(None)` for it. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub enum PlacementError { - /// The configuration echo records a baseline placement while the generation stages a - /// projector checkpoint. - EchoDisagrees, - /// The generation stages a projector checkpoint without training evidence. - MissingEvidence, - /// The ladder evidence names a canonical step outside its own schedule. - CanonicalStep, - /// The projector checkpoint does not open, or does not decode against the echoed - /// architecture. - Checkpoint, - /// The reopened publish path does not reproduce the generation's own published coordinates. - Certificate, -} - -impl core::fmt::Display for PlacementError { - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::EchoDisagrees => fmt.write_str( - "the generation stages a projector checkpoint while its configuration echo \ - records a baseline placement", - ), - Self::MissingEvidence => fmt.write_str( - "the generation stages a projector checkpoint without training evidence", - ), - Self::CanonicalStep => { - fmt.write_str("the ladder evidence names a canonical step outside its own schedule") - } - Self::Checkpoint => fmt.write_str( - "the projector checkpoint does not open or does not decode against the echoed \ - architecture", - ), - Self::Certificate => fmt.write_str( - "the reopened publish path does not reproduce the generation's own published \ - coordinates", - ), - } - } -} - -impl core::error::Error for PlacementError {} - -/// The online half of the placement stage, bound to one published generation. -/// -/// Holds the reopened checkpoint together with the recorded canonical condition, alignment, and -/// world frame, so [`project`](Self::project) is a pure function from stored embedding prefixes -/// to wire coordinates - the same function the fit applied to every fitted row. -pub(crate) struct Placer { - /// The reopened checkpoint on the placement backend. - model: Projector, - /// The device every projection runs on. - device: PhysicalDevice, - /// The canonical step's condition, or zero for a generation without a measured ladder. - condition: NonNegative, - /// The canonical step's recorded similarity, or [`None`] where the zero frame published - /// directly. - alignment: Option, - /// The recorded world frame the wire normalization maps from. - world: Bounds2, - /// The forward slice bound the configuration echo records. - forward_rows: core::num::NonZero, -} - -impl Placer { - /// Opens the generation's publish path for online projection. - /// - /// Answers `Ok(None)` for a baseline-placed generation, the one shape that promises no - /// publish path. Its arrivals stage until a refit. - /// - /// # Errors - /// - /// Refusals fail closed, each under its own log line at the refusal site: - /// - /// - [`PlacementError::EchoDisagrees`] when the placement discriminant and the configuration - /// echo disagree - /// - [`PlacementError::MissingEvidence`] when the checkpoint stages without training evidence - /// - [`PlacementError::CanonicalStep`] when the ladder evidence names a step outside its own - /// schedule - /// - [`PlacementError::Checkpoint`] when the checkpoint does not open or does not decode - /// - [`PlacementError::Certificate`] when the reopened path does not reproduce the generation's - /// own published coordinates - pub(crate) fn open( - generation: &Generation, - device: PhysicalDevice, - ) -> Result, PlacementError> { - let repository = generation.repository(); - let metadata = &repository.metadata; - - let Some(checkpoint) = &repository.files.projector else { - tracing::info!( - "the generation placed rows by landmark baseline, arrivals stage until a refit" - ); - return Ok(None); - }; - let PlacementOptions::Projector(options) = &metadata.reproducibility.config.placement - else { - tracing::warn!( - "the generation stages a projector checkpoint while its configuration echo \ - records a baseline placement" - ); - return Err(PlacementError::EchoDisagrees); - }; - let Some(evidence) = &metadata.evidence.projector else { - tracing::warn!( - "the generation stages a projector checkpoint without training evidence" - ); - return Err(PlacementError::MissingEvidence); - }; - - let (condition, alignment) = match &evidence.ladder { - Some(ladder) => { - let Some(step) = ladder.steps.get(ladder.canonical_index) else { - tracing::warn!( - canonical_index = ladder.canonical_index, - steps = ladder.steps.len(), - "the ladder evidence names a canonical step outside its own schedule" - ); - return Err(PlacementError::CanonicalStep); - }; - (ladder.canonical, Some(step.alignment)) - } - None => (NonNegative::ZERO, None), - }; - - let checkpoint = match File::open(generation.path_of(&checkpoint.name())) { - Ok(file) => file, - Err(error) => { - tracing::warn!(%error, "the projector checkpoint does not open"); - return Err(PlacementError::Checkpoint); - } - }; - let model: Projector = - match artifact::open_model(checkpoint, options.architecture, &device) { - Ok(model) => model, - Err(error) => { - tracing::warn!( - %error, - "the projector checkpoint does not decode against the echoed architecture" - ); - return Err(PlacementError::Checkpoint); - } - }; - - let placer = Self { - model, - device, - condition, - alignment, - world: metadata.evidence.lod.world, - forward_rows: options.forward_rows, - }; - if placer.certify(generation) { - Ok(Some(placer)) - } else { - Err(PlacementError::Certificate) - } - } - - /// Certifies the reopened publish path against the published coordinate column. - /// - /// Projects an even sample of the generation's own fitted rows and compares each aligned - /// coordinate against the published column, per component in world units, under the ladder - /// report's reproduction bound. Returns whether every sampled row reproduces, logging the - /// verdict either way with the sample width and the largest error observed. - fn certify(&self, generation: &Generation) -> bool { - let files = &generation.repository().files; - - let representations = - match ArrayFile::open(generation.path_of(&files.representations.name())) { - Ok(file) => file, - Err(error) => { - tracing::warn!(%error, "the representation matrix does not open"); - return false; - } - }; - let Some(representations) = representations.vectors::() else { - tracing::warn!("the representation matrix does not read as projector-width rows"); - return false; - }; - - let coordinates = match ArrayFile::open(generation.path_of(&files.coordinates.name())) { - Ok(file) => file, - Err(error) => { - tracing::warn!(%error, "the coordinate column does not open"); - return false; - } - }; - let Some(coordinates) = coordinates.points() else { - tracing::warn!("the coordinate column does not read as points"); - return false; - }; - - let rows = representations.len(); - if rows == 0 || rows != coordinates.len() { - tracing::warn!( - representations = rows, - coordinates = coordinates.len(), - "the representation matrix and the coordinate column do not describe one \ - populated corpus" - ); - return false; - } - - let width = CERTIFICATE_ROWS.min(rows); - #[expect( - clippy::integer_division, - clippy::integer_division_remainder_used, - reason = "the floored stride spreads the sample evenly over the row domain" - )] - let sampled: Vec = (0..width).map(|index| index * rows / width).collect(); - let sampled_rows = sampled.iter().map(|&row| &representations[row]); - let aligned = match self.forward_aligned(sampled_rows) { - Ok(aligned) => aligned, - Err(failure) => { - tracing::warn!( - row = sampled[failure.row], - "a fitted row's reprojection is non-finite" - ); - return false; - } - }; - - let mut max_error = 0.0_f64; - for (&row, reprojected) in sampled.iter().zip(&aligned) { - let published = coordinates[row]; - let error = f64::from((reprojected.x() - published.x()).abs()) - .max(f64::from((reprojected.y() - published.y()).abs())); - max_error = max_error.max(error); - } - - if max_error < f64::from(CERTIFICATE_TOLERANCE) { - tracing::info!( - samples = width, - rows, - max_error, - "the online projection reproduces the published coordinates" - ); - true - } else { - tracing::warn!( - samples = width, - rows, - max_error, - tolerance = %CERTIFICATE_TOLERANCE, - "the online projection does not reproduce the published coordinates" - ); - false - } - } - - /// Projects one batch of arrivals through the publish path. - /// - /// Each embedding is a stored whole-entity embedding's l2-normalized projector prefix, and - /// each outcome is that row's wire coordinate or its out-of-frame world coordinate, in the - /// batch's own order. The construction repeats the fit's own publish path over the batch, so - /// a placed arrival and a fitted row take their coordinates from one function. - /// - /// # Errors - /// - /// Returns [`NonFiniteProjection`] naming the first row whose forward produced a non-finite - /// coordinate. Rows after it were not projected, and the caller retries them. - pub(crate) fn project( - &self, - embeddings: impl IntoIterator>>, - ) -> Result, NonFiniteProjection> { - let world = self.forward_aligned(embeddings)?; - let wire = self.world.normalize_into(WIRE_FRAME, &world); - - Ok(world - .iter() - .zip(wire) - .map(|(&point, wire)| { - if self.world.contains(point) { - Projection::Placed { wire } - } else { - Projection::OutOfFrame { world: point } - } - }) - .collect()) - } - - /// Forwards a batch through the checkpoint and aligns it into the baseline frame. - /// - /// The rows copy into an aligned scratch matrix in bounded slices, every row projects at the - /// placer's condition under the knowledge-entity role, and the recorded alignment maps each - /// point. The result is in world units, in the batch's own order. - /// - /// # Errors - /// - /// Returns [`NonFiniteProjection`] naming the first row whose forward produced a non-finite - /// coordinate. - fn forward_aligned( - &self, - rows: impl IntoIterator>>, - ) -> Result, NonFiniteProjection> { - let mut rows = rows.into_iter(); - let mut aligned = Vec::with_capacity(rows.size_hint().0); - let mut scratch = MatrixN::::zeroed(SCRATCH_ROWS); - let roles = vec![NodeRole::KnowledgeEntity; SCRATCH_ROWS]; - - let mut base = 0; - loop { - let mut filled = 0; - for (slot, row) in scratch.rows_mut().iter_mut().zip(&mut rows) { - slot.as_array_mut().copy_from_slice(row.as_ref().as_array()); - filled += 1; - } - if filled == 0 { - break; - } - - let columns: NodeColumns<'_, BatchRow> = NodeColumns { - representations: IdSlice::from_raw(&scratch.rows()[..filled]), - roles: IdSlice::from_raw(&roles[..filled]), - }; - - let frame = refresh::forward( - &self.model, - columns, - self.condition, - self.forward_rows, - &self.device, - ) - .map_err(|error| match error { - refresh::RefreshError::Diverged { row, .. } - | refresh::RefreshError::NonFiniteScale { row, .. } => NonFiniteProjection { - row: base + row.as_usize(), - }, - })?; - - match self.alignment { - Some(alignment) => { - aligned.extend(frame.iter().map(|&point| alignment.apply(point))); - } - None => aligned.extend(frame.iter().copied()), - } - - if filled < SCRATCH_ROWS { - break; - } - base += filled; - } - - Ok(aligned) - } -} - -impl core::fmt::Debug for Placer { - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - fmt.debug_struct("Placer") - .field("condition", &self.condition) - .field("alignment", &self.alignment) - .field("world", &self.world) - .field("forward_rows", &self.forward_rows) - .finish_non_exhaustive() - } -} - -#[cfg(test)] -mod tests { - use core::num::NonZero; - - use rand::SeedableRng as _; - use rand_xoshiro::Xoshiro256PlusPlus; - - use super::*; - use crate::{ - device::Device, - math::{BoxedVecN, Rotation, non_negative, positive}, - salt::projector::model::Architecture, - }; - - /// Batch rows spanning two scratch chunks plus a remainder. - const CHUNKED_ROWS: usize = SCRATCH_ROWS * 2 + 2; - - fn architecture() -> Architecture { - Architecture { - width: NonZero::new(8).expect("fixture width is non-zero"), - residual_blocks: NonZero::new(1).expect("fixture depth is non-zero"), - representation_dimensions: NonZero::new(PROJECTOR_DIMENSIONS) - .expect("the projector width is non-zero"), - role_dimensions: NonZero::new(4).expect("fixture role width is non-zero"), - condition_dimensions: NonZero::new(1).expect("fixture condition width is non-zero"), - } - } - - fn model() -> Projector { - Projector::new( - architecture(), - &Device::Cpu.pin(0).resolve(), - Xoshiro256PlusPlus::seed_from_u64(7), - ) - } - - fn alignment() -> Similarity { - Similarity::new( - positive!(2.0), - Rotation::from_radians(0.5), - Vec2::new(3.0, -4.0), - ) - .expect("the fixture scale is normal") - } - - fn placer(alignment: Option, world: Bounds2) -> Placer { - Placer { - model: model(), - device: Device::Cpu.pin(0).resolve(), - condition: non_negative!(0.25), - alignment, - world, - forward_rows: NonZero::new(64).expect("the fixture slice bound is non-zero"), - } - } - - /// A distinct unit-scale embedding per row. - #[expect( - clippy::cast_precision_loss, - reason = "the fixture ordinals stay far below the mantissa width" - )] - fn embedding(row: usize) -> BoxedVecN { - let mut components = [0.0_f32; PROJECTOR_DIMENSIONS]; - for (index, component) in components.iter_mut().enumerate() { - *component = ((row * PROJECTOR_DIMENSIONS + index) as f32).sin() * 0.05; - } - BoxedVecN::from(components) - } - - /// Projects `embeddings` through the fit's own leaf calls: forward, alignment, wire frame. - fn publish_path( - placer: &Placer, - embeddings: &[BoxedVecN], - ) -> (Vec, Vec) { - let mut scratch = BoxedVecN::<{ CHUNKED_ROWS * PROJECTOR_DIMENSIONS }>::zero(); - for (slot, row) in embeddings.iter().enumerate() { - scratch.as_array_mut()[slot * PROJECTOR_DIMENSIONS..][..PROJECTOR_DIMENSIONS] - .copy_from_slice(row.as_array()); - } - let staged = - AlignedVecN::from_slice(&scratch.as_array()[..embeddings.len() * PROJECTOR_DIMENSIONS]) - .expect("boxed scratch storage is aligned for whole projector rows"); - let roles = vec![NodeRole::KnowledgeEntity; embeddings.len()]; - let columns: NodeColumns<'_, BatchRow> = NodeColumns { - representations: IdSlice::from_raw(staged), - roles: IdSlice::from_raw(&roles), - }; - - let frame = refresh::forward( - &placer.model, - columns, - placer.condition, - placer.forward_rows, - &Device::Cpu.pin(0).resolve(), - ) - .expect("the fixture forward is finite"); - let world: Vec = placer.alignment.map_or_else( - || frame.iter().copied().collect(), - |alignment| frame.iter().map(|&point| alignment.apply(point)).collect(), - ); - let wire = placer.world.normalize_into(WIRE_FRAME, &world); - (world, wire) - } - - /// World bounds covering every point of `world` with margin. - fn covering(world: &[Vec2]) -> Bounds2 { - let mut bounds = Bounds2::new(world[0], world[0]).expect("a fixture point is finite"); - for &point in world { - bounds = bounds.union(Bounds2::new(point, point).expect("a fixture point is finite")); - } - Bounds2::new( - Vec2::new(bounds.min().x() - 1.0, bounds.min().y() - 1.0), - Vec2::new(bounds.max().x() + 1.0, bounds.max().y() + 1.0), - ) - .expect("widened fixture bounds stay ordered") - } - - #[test] - fn the_online_projection_byte_equals_the_publish_path() { - let embeddings: Vec<_> = (0..CHUNKED_ROWS).map(embedding).collect(); - - // The probe learns where the fixture model puts the batch, and the placer under test - // pins a frame that contains every point. - let probe = placer(Some(alignment()), WIRE_FRAME); - let (world, _) = publish_path(&probe, &embeddings); - let placer = placer(Some(alignment()), covering(&world)); - let (_, expected) = publish_path(&placer, &embeddings); - - let refs: Vec<&BoxedVecN> = embeddings.iter().collect(); - let projections = placer - .project(&refs) - .expect("the fixture forward is finite"); - - assert_eq!(projections.len(), expected.len()); - for (projection, wire) in projections.iter().zip(&expected) { - let Projection::Placed { wire: served } = projection else { - panic!("every fixture point lies inside the covering frame"); - }; - assert_eq!(served.x().to_bits(), wire.x().to_bits()); - assert_eq!(served.y().to_bits(), wire.y().to_bits()); - } - } - - #[test] - fn a_ladder_less_generation_projects_at_zero_without_alignment() { - let embeddings = vec![embedding(3), embedding(11)]; - - let probe = Placer { - condition: NonNegative::ZERO, - ..placer(None, WIRE_FRAME) - }; - let (world, _) = publish_path(&probe, &embeddings); - let placer = Placer { - condition: NonNegative::ZERO, - ..placer(None, covering(&world)) - }; - let (_, expected) = publish_path(&placer, &embeddings); - - let refs: Vec<&BoxedVecN> = embeddings.iter().collect(); - let projections = placer - .project(&refs) - .expect("the fixture forward is finite"); - - for (projection, wire) in projections.iter().zip(&expected) { - let Projection::Placed { wire: served } = projection else { - panic!("every fixture point lies inside the covering frame"); - }; - assert_eq!(served.x().to_bits(), wire.x().to_bits()); - assert_eq!(served.y().to_bits(), wire.y().to_bits()); - } - } - - #[test] - fn an_out_of_frame_coordinate_is_held_with_its_world_point() { - let embeddings = vec![embedding(5)]; - - let probe = placer(Some(alignment()), WIRE_FRAME); - let (world, _) = publish_path(&probe, &embeddings); - - // A frame strictly past the projected point excludes it. - let excluding = Bounds2::new( - Vec2::new(world[0].x() + 10.0, world[0].y() + 10.0), - Vec2::new(world[0].x() + 20.0, world[0].y() + 20.0), - ) - .expect("the excluding fixture bounds are ordered"); - let placer = placer(Some(alignment()), excluding); - - let refs: Vec<&BoxedVecN> = embeddings.iter().collect(); - let projections = placer - .project(&refs) - .expect("the fixture forward is finite"); - - assert_eq!( - projections, - vec![Projection::OutOfFrame { world: world[0] }] - ); - } - - #[test] - fn chunked_batches_project_each_row_as_its_own_batch_would() { - let embeddings: Vec<_> = (0..CHUNKED_ROWS).map(embedding).collect(); - let probe = placer(Some(alignment()), WIRE_FRAME); - let (world, _) = publish_path(&probe, &embeddings); - let placer = placer(Some(alignment()), covering(&world)); - - let refs: Vec<&BoxedVecN> = embeddings.iter().collect(); - let batched = placer - .project(&refs) - .expect("the fixture forward is finite"); - - for row in [0, SCRATCH_ROWS - 1, SCRATCH_ROWS, CHUNKED_ROWS - 1] { - let alone = placer - .project(&refs[row..=row]) - .expect("the fixture forward is finite"); - assert_eq!(alone, vec![batched[row]], "row {row} moved under chunking"); - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/register.rs b/libs/@local/graph/atlas/src/serve/delta/register.rs deleted file mode 100644 index de589e7a87a..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/register.rs +++ /dev/null @@ -1,659 +0,0 @@ -//! The mutable fold the consumer writes: standings, verdicts, captures, and placements. -//! -//! [`DeltaRegister`] is the poll arm's single-writer state. A standing folds last-writer-wins -//! by version, while a classification or a placement holds first-delivery for the process's -//! lifetime, because neither can change meaning under the never-reassign row law. Every -//! mutation answers a [`Disposition`], and a hold that would grow a row domain past the -//! codec's range refuses as [`UniverseExhausted`]. The contracts behind these rules live in -//! the parent module's doc. - -use core::{ - cmp::Ordering, - fmt, - ops::{BitOr, BitOrAssign}, -}; - -use hash_graph_temporal_versioning::{Timestamp, TransactionTime}; -use hashql_core::collections::{FastHashMap, FastHashMapEntry, fast_hash_map, fast_hash_set}; -use type_system::knowledge::entity::id::EntityEditionId; - -use super::{ - DeltaEdge, DeltaEvent, DeltaNode, DeltaRevision, IdentityTables, ProjectedArrival, Standing, - overlay::IdentityTableOverlay, snapshot::DeltaSnapshot, -}; -use crate::{ - bitset::CompressedBitSet, - dataset::auxiliary::{Icon, Label, OwnedIcon, OwnedLegend}, - identity::{EdgeRowId, NodeRowId, OntologyRowId}, - postgres::{ - Classification, - id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - }, - serve::{Atlas, codec::Universe}, -}; - -/// The newest applied event's version and standing, for one identity. -#[derive(Debug, Copy, Clone)] -struct AppliedEvent { - /// The applied event's transaction time. - version: Timestamp, - /// The applied event's standing. - standing: Standing, -} - -impl AppliedEvent { - /// Returns whether this event replaces `incumbent` under the fold. - /// - /// The comparison is the fold's total order: version, then standing rank, then edition between - /// two live standings. An event never supersedes an equal one, which is what makes - /// re-delivery idempotent. - fn supersedes(&self, incumbent: &Self) -> bool { - match self.version.cmp(&incumbent.version) { - Ordering::Greater => true, - Ordering::Less => false, - Ordering::Equal => match (self.standing, incumbent.standing) { - (Standing::Withdrawn, Standing::Live { .. }) => true, - (Standing::Withdrawn | Standing::Live { .. }, Standing::Withdrawn) => false, - ( - Standing::Live { edition }, - Standing::Live { - edition: incumbent_edition, - }, - ) => edition > incumbent_edition, - }, - } - } -} - -/// One identity's captured legend, keyed to the edition the capture read. -/// -/// The edition decides staleness. A register edition past the captured one lists the identity -/// for a fresh capture at the next poll, and the newest capture serves meanwhile, exactly as a -/// placement's coordinate serves until refit. -#[derive(Debug, Clone)] -struct EditionLegend { - /// The edition the capture read. - edition: EntityEditionId, - /// The legend the read answered. - legend: OwnedLegend, -} - -/// A refusal to allocate a row past the domain its holder can carry. -/// -/// For node and edge rows the bound is the wire codec's `u32` row domain, enforced at the -/// register's allocation sites. For every row domain the id type's own end bounds allocation -/// last. The refusal fails closed. The allocation records nothing and the arrival stays staged -/// or unclassified. Every later first allocation refuses the same way until a refit retires -/// the register. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct UniverseExhausted; - -impl fmt::Display for UniverseExhausted { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - fmt.write_str("the accepted universe is at its row domain's bound") - } -} - -impl core::error::Error for UniverseExhausted {} - -/// The register's disposition of one delivered classification verdict or placement. -/// -/// Publication's resolution input changes on [`Disposition::Resolving`] alone, and -/// [`Disposition::changes_resolution`] reads exactly that. Dispositions join through `|` into -/// the strongest one delivered, with [`Disposition::AlreadyHeld`] as the neutral element. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum Disposition { - /// Newly held for a live arrival, so publication's resolution input changed. - Resolving, - /// Newly held for an identity not standing live, so resolution stays unchanged until the - /// feed reports the identity live. - Dormant, - /// A holding for the identity already stood, so the delivery recorded nothing. - AlreadyHeld, -} - -impl Disposition { - /// Returns whether this disposition changed publication's resolution input. - #[must_use] - pub(crate) const fn changes_resolution(self) -> bool { - matches!(self, Self::Resolving) - } -} - -impl BitOr for Disposition { - type Output = Self; - - /// Joins two dispositions into the stronger one, by resolution strength. - /// - /// [`Disposition::Resolving`] absorbs, [`Disposition::AlreadyHeld`] is the neutral element, - /// and [`Disposition::Dormant`] sits between, so a fold over a batch answers whether any - /// delivery resolved while still recording that a new holding exists. The joined value's - /// consumed meaning is [`Disposition::changes_resolution`] alone; the Dormant-over-held - /// preference keeps the new-holding fact in the value, which no consumer reads yet. - fn bitor(self, rhs: Self) -> Self { - match (self, rhs) { - (Self::Resolving, _) | (_, Self::Resolving) => Self::Resolving, - (Self::Dormant, _) | (_, Self::Dormant) => Self::Dormant, - (Self::AlreadyHeld, Self::AlreadyHeld) => Self::AlreadyHeld, - } - } -} - -impl BitOrAssign for Disposition { - fn bitor_assign(&mut self, rhs: Self) { - *self = *self | rhs; - } -} - -/// The mutable fold of the feed since the generation's fit-time snapshot. -/// -/// One [`AppliedEvent`] per identity, last-writer-wins on the event's transaction time. The map -/// grows with the distinct identities the feed has reported, and a refit retires it along with -/// the generation whose delta it states. -#[derive(Debug)] -pub(crate) struct DeltaRegister { - /// The newest applied event per identity. - applied: FastHashMap, - /// The held classification verdict per identity, insert-only. - classifications: FastHashMap, - /// The published node payload per placed identity, insert-only. - placements: FastHashMap, - /// The captured legend per identity, replaced when a newer edition's capture lands. - legends: FastHashMap, - /// Every allocated node row beside the accepted universe the allocations grew. - node_rows: IdentityTableOverlay, - /// The edge rows allocated for complete-pair links at their classification hold. - edge_rows: IdentityTableOverlay, - /// The ontology rows allocated for types the generation never tabulated. - ontology_rows: IdentityTableOverlay, - /// The icons recorded at the allocated ontology rows, written once at allocation. - ontology_icons: FastHashMap, -} - -impl DeltaRegister { - /// Builds an empty register over a generation whose row universes are `nodes`, `edges` and - /// `ontology`. - /// - /// Row allocation starts at each bound, so the first placement takes the first node row past - /// the generation's fitted rows, the first complete link verdict the first edge row past the - /// generation's fitted edges, and the first unknown type the first ontology row past the - /// generation's tabulated types. - pub(crate) fn new( - nodes: Universe, - edges: Universe, - ontology: Universe, - ) -> Self { - Self { - applied: fast_hash_map(), - classifications: fast_hash_map(), - placements: fast_hash_map(), - legends: fast_hash_map(), - node_rows: IdentityTableOverlay::new(nodes), - edge_rows: IdentityTableOverlay::new(edges), - ontology_rows: IdentityTableOverlay::new(ontology), - ontology_icons: fast_hash_map(), - } - } - - /// Creates a new delta register from the atlas's node, edge, and ontology universes. - pub(crate) fn from_atlas(atlas: &Atlas) -> Self { - Self::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ) - } - - /// Resolves a representative type into its ontology row, allocating past the baked bound - /// for a type the generation never tabulated. - /// - /// The generation's table answers first and the extension second, so a reader learns - /// nothing about which side answered. A type neither holds allocates the next row, and the - /// allocation records `icon` at exactly that moment. Tabulated types resolve their icons - /// through the baked closure artifact, which leaves the extension recording icons for its - /// own rows alone. A later capture for the same type replaces nothing: the refit repairs - /// icon staleness exactly as it repairs coordinates. - /// - /// [`None`] is the ontology row domain's own end. - fn resolve_representative( - &mut self, - representative: ArchivedOntologyTypeUuid, - icon: &Icon, - tables: &impl IdentityTables, - ) -> Option { - if let Some(row) = tables.ontology_row_of(representative) { - return Some(row); - } - if let Some(row) = self.ontology_rows.row_of(representative) { - return Some(row); - } - - let row = self.ontology_rows.resolve(representative)?; - self.ontology_icons.insert(row, icon.to_owned()); - Some(row) - } - - /// Applies one event, returning whether publication's resolution input changed. - /// - /// The fold is last-writer-wins per identity. An event that supersedes the held register under - /// the total order replaces it. Every other event changes nothing, because the register already - /// reflects a later state of the same entity, and re-delivery of an already-applied event is - /// idempotent. - /// - /// The return value is the publication decision's input: `true` when an identity arrives, a - /// standing flips, or a live standing's edition moves, and `false` when only the version moved. - /// The signal over-approximates, because a new live identity the generation fitted resolves to - /// nothing, so a poll acting on it publishes a snapshot equivalent to the held one. A poll - /// whose applications all return `false` can skip publishing. - pub(crate) fn apply( - &mut self, - DeltaEvent { - entity, - version, - standing, - }: DeltaEvent, - ) -> bool { - let challenger = AppliedEvent { version, standing }; - - match self.applied.entry(entity) { - FastHashMapEntry::Vacant(slot) => { - slot.insert(challenger); - true - } - FastHashMapEntry::Occupied(mut slot) => { - let incumbent = *slot.get(); - if !challenger.supersedes(&incumbent) { - return false; - } - - slot.insert(challenger); - incumbent.standing != standing - } - } - } - - /// Iterates the live arrivals holding no classification verdict. - /// - /// An arrival is a live identity the generation never fitted. A withdrawn identity never - /// lists, because it serves nothing whatever its category, and the feed's own withdrawal is - /// what resolves an arrival whose present ended between its event and a lookup. A fitted - /// identity never lists, because the generation's tables already decide its category. - pub(crate) fn unclassified( - &self, - tables: &impl IdentityTables, - ) -> impl Iterator { - self.applied - .iter() - .filter(|&(entity, applied)| { - matches!(applied.standing, Standing::Live { .. }) - && !self.classifications.contains_key(entity) - && tables.node_row_of(*entity).is_none() - && tables.edge_row_of(*entity).is_none() - }) - .map(|(&entity, _)| entity) - } - - /// Holds one classification verdict, returning the register's disposition of it. - /// - /// The first verdict per identity holds for the process's lifetime, and a later verdict for - /// the same identity changes nothing, so re-delivery of a verdict is idempotent and comes - /// back [`Disposition::AlreadyHeld`]. A complete link verdict allocates the link's edge row - /// at the hold. That row is the next past the accepted edge universe and stands for the - /// process's lifetime, so a published link speaks the row domain fitted edges speak. A newly - /// held verdict is [`Disposition::Resolving`] exactly when the identity stands live, - /// because only a live arrival resolves through its classification at publication, and - /// [`Disposition::Dormant`] otherwise. - /// - /// # Errors - /// - /// Returns [`UniverseExhausted`] when one more edge row would grow the accepted edge - /// universe past the wire codec's `u32` row domain, or when the edge row domain has no next - /// row to allocate. The hold records nothing and the identity stays unclassified, so the - /// next poll retries it and meets the same refusal until a refit retires the register. - pub(crate) fn classify( - &mut self, - entity: ArchivedEntityId, - verdict: Classification, - ) -> Result { - if self.classifications.contains_key(&entity) { - return Ok(Disposition::AlreadyHeld); - } - - if let Classification::Edge { - source: Some(_), - target: Some(_), - } = verdict - { - self.edge_rows.resolve(entity).ok_or(UniverseExhausted)?; - } - - self.classifications.insert(entity, verdict); - Ok( - if matches!( - self.applied.get(&entity), - Some(AppliedEvent { - standing: Standing::Live { .. }, - .. - }) - ) { - Disposition::Resolving - } else { - Disposition::Dormant - }, - ) - } - - /// Lists the identities whose current edition no captured legend matches, each with the - /// edition to read. - /// - /// An identity lists when its legend capture is stale or absent and serving consults it or - /// will. Fitted identities list, since a post-fit edition revises the baked legend, and so - /// does a link-classified arrival with a complete attachment pair, since publication - /// withholds an uncaptured link. A fitted node's capture has no serving reader yet - the - /// edges trailer asks only for links - and exists for the tile and locate label overlay that - /// will read request-time legends through the register, so narrowing the listing to links - /// would leave that overlay nothing to read. Node-classified arrivals never list, because - /// staging reads their displays before placement, and neither does a withdrawn identity, - /// which serves nothing. - pub(crate) fn pending_captures( - &self, - tables: &impl IdentityTables, - ) -> impl Iterator { - self.applied.iter().filter_map(move |(&entity, applied)| { - let Standing::Live { edition } = applied.standing else { - return None; - }; - if let Some(captured) = self.legends.get(&entity) - && captured.edition == edition - { - return None; - } - - let fitted = - tables.node_row_of(entity).is_some() || tables.edge_row_of(entity).is_some(); - let serving_link = matches!( - self.classifications.get(&entity), - Some(Classification::Edge { - source: Some(_), - target: Some(_), - }) - ); - - (fitted || serving_link).then_some((entity, edition)) - }) - } - - /// Holds one captured legend, keyed to the edition its read answered. - /// - /// The newest capture replaces the held one, because a capture for an edition the register - /// has moved past lists the identity again at the next poll, and the held legend serves - /// meanwhile. The representative type resolves into its ontology row here. The generation's - /// table answers a tabulated type, and for one the generation never saw the register's own - /// extension answers or allocates, recording `icon` beside a freshly allocated row. - /// - /// # Errors - /// - /// Returns [`UniverseExhausted`] when the ontology row domain has no next row to allocate. - /// The capture records nothing and the identity lists again at the next poll. - pub(crate) fn capture_display( - &mut self, - entity: ArchivedEntityId, - edition: EntityEditionId, - label: &Label, - icon: &Icon, - representative: ArchivedOntologyTypeUuid, - tables: &impl IdentityTables, - ) -> Result<(), UniverseExhausted> { - let representative = self - .resolve_representative(representative, icon, tables) - .ok_or(UniverseExhausted)?; - - self.legends.insert( - entity, - EditionLegend { - edition, - legend: OwnedLegend::new(representative, label), - }, - ); - Ok(()) - } - - /// Holds one placement, returning the register's disposition of it. - /// - /// The first placement per identity takes the next row past the accepted universe and holds - /// for the process's lifetime. A later placement for the same identity changes nothing, - /// because the coordinate and the row never move, so re-delivery of a placement is - /// idempotent and comes back [`Disposition::AlreadyHeld`]. A placement for an identity - /// standing withdrawn records all the same, so an unarchive republishes the recorded - /// coordinate on its former row. A newly held placement is [`Disposition::Resolving`] - /// exactly when the identity stands live, because only a live arrival resolves through its - /// placement at publication, and [`Disposition::Dormant`] otherwise. - /// - /// The arrival's representative type resolves into its ontology row here, exactly as - /// [`Self::capture_display`] resolves one, so the published node's legend speaks the row - /// domain every fitted legend speaks. - /// - /// # Errors - /// - /// Returns [`UniverseExhausted`] when one more node row would grow the accepted universe past - /// the wire codec's `u32` row domain, or when the ontology row domain has no next row to - /// allocate. The placement records nothing and the arrival stays staged. - pub(crate) fn place( - &mut self, - entity: ArchivedEntityId, - &ProjectedArrival { - edition, - position, - ref label, - ref icon, - representative: representative_type_uuid, - }: &ProjectedArrival, - tables: &impl IdentityTables, - ) -> Result { - if self.placements.contains_key(&entity) { - return Ok(Disposition::AlreadyHeld); - } - - // The codec permutes node rows over u32, so a row past that domain could never encode. - // The refusal sits before any allocation, which keeps a refused placement free of side - // effects in the node domain. - if self.node_rows.universe().size() > u32::MAX as usize { - return Err(UniverseExhausted); - } - - let representative = self - .resolve_representative(representative_type_uuid, icon, tables) - .ok_or(UniverseExhausted)?; - let id = self.node_rows.resolve(entity).ok_or(UniverseExhausted)?; - - self.placements.insert( - entity, - DeltaNode { - id, - edition, - position, - legend: OwnedLegend::new(representative, label), - }, - ); - - Ok( - if matches!( - self.applied.get(&entity), - Some(AppliedEvent { - standing: Standing::Live { .. }, - .. - }) - ) { - Disposition::Resolving - } else { - Disposition::Dormant - }, - ) - } - - /// Estimates the fold's resident bytes. - /// - /// A weighted estimate for replay telemetry rather than an allocator-faithful ceiling: the - /// event, classification, placement, and legend maps' heap allocations, the captured legends' - /// bytes on both holders, and the row-domain extensions. - #[must_use] - pub(crate) fn resident_estimate(&self) -> usize { - let placement_legends: usize = self - .placements - .values() - .map(|node| node.legend.heap_bytes()) - .sum::() - .saturating_cast(); - let captured_legends: usize = self - .legends - .values() - .map(|captured| captured.legend.heap_bytes()) - .sum::() - .saturating_cast(); - let allocated_icons: usize = self - .ontology_icons - .values() - .map(|icon| icon.as_ref().len() as u64) - .sum::() - .saturating_cast(); - - self.applied.allocation_size() - + self.classifications.allocation_size() - + self.placements.allocation_size() - + self.legends.allocation_size() - + self.node_rows.resident_estimate() - + self.edge_rows.resident_estimate() - + self.ontology_rows.resident_estimate() - + self.ontology_icons.allocation_size() - + placement_legends - + captured_legends - + allocated_icons - } - - /// Publishes the fold as an immutable snapshot resolved against `tables`. - /// - /// `revision` and `watermark` are the publisher's: the revision names this publication in - /// publication order, and the watermark is the feed position the publisher had folded through - /// when it published. - /// - /// # Panics - /// - /// This panics when `tables` resolves an identity to a row above the row bitsets' representable - /// domain, which is an implementor bug rather than feed data. - #[must_use] - pub(crate) fn snapshot( - &self, - tables: &impl IdentityTables, - revision: DeltaRevision, - watermark: Timestamp, - ) -> DeltaSnapshot { - let mut withdrawn = fast_hash_set(); - let mut withdrawn_nodes = CompressedBitSet::new(); - let mut withdrawn_edges = CompressedBitSet::new(); - let mut staged = fast_hash_map(); - let mut nodes = fast_hash_map(); - let mut edges = fast_hash_map(); - let mut legends = fast_hash_map(); - - for (&entity, applied) in &self.applied { - match applied.standing { - Standing::Withdrawn => { - withdrawn.insert(entity); - - if let Some(row) = tables.node_row_of(entity) { - withdrawn_nodes.insert(row); - } else if let Some(row) = tables.edge_row_of(entity) { - withdrawn_edges.insert(row); - } else { - // An unfitted withdrawal has no generation row to subtract, and the - // identity set above already carries it. - } - } - Standing::Live { edition } => { - if tables.node_row_of(entity).is_some() || tables.edge_row_of(entity).is_some() - { - // A fitted identity with a post-fit edition revises its baked legend: - // the captured legend publishes, and holders read it map-first. A - // revision the read has not yet answered keeps answering from what - // the holder already reads - the previous capture, or the fit-time - // payload when the register never captured one - until its capture - // read lands. - if let Some(captured) = self.legends.get(&entity) { - legends.insert(entity, captured.legend.clone()); - } - } else { - match self.classifications.get(&entity) { - Some(Classification::Node) => { - if let Some(node) = self.placements.get(&entity) { - nodes.insert(entity, node.clone().with_edition(edition)); - } else { - staged.insert(entity, edition); - } - } - Some(&Classification::Edge { - source: Some(source), - target: Some(target), - }) => { - let resolve = |endpoint: ArchivedEntityId| { - tables - .node_row_of(endpoint) - .or_else(|| self.node_rows.row_of(endpoint)) - }; - - if let Some(captured) = self.legends.get(&entity) - && let Some(source) = resolve(source) - && let Some(target) = resolve(target) - { - let id = self.edge_rows.row_of(entity).expect( - "classification allocated the edge row at the hold", - ); - - edges.insert( - entity, - DeltaEdge { - id, - edition, - source, - target, - }, - ); - legends.insert(entity, captured.legend.clone()); - } - } - Some(Classification::Edge { .. }) | None => { - // An unclassified arrival and a link with an incomplete - // attachment pair publish nowhere: nothing serves until a - // verdict supplies the data to serve it. - } - } - } - } - } - } - - // Locate's link-label pass consumes this containment as an `unreachable!` arm: the - // edge insert and the legend insert above share one `if let` block, so a published - // edge always carries a legend. The check lives here, where the walk establishes the - // invariant, rather than at the consumer. - debug_assert!( - edges.keys().all(|entity| legends.contains_key(entity)), - "every published edge carries its captured legend", - ); - - DeltaSnapshot { - revision, - watermark, - withdrawn, - withdrawn_nodes, - withdrawn_edges, - staged, - nodes, - edges, - legends, - node_rows: self.node_rows.clone(), - edge_rows: self.edge_rows.clone(), - ontology_rows: self.ontology_rows.clone(), - ontology_icons: self.ontology_icons.clone(), - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/snapshot.rs b/libs/@local/graph/atlas/src/serve/delta/snapshot.rs deleted file mode 100644 index 7412a47e8b4..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/snapshot.rs +++ /dev/null @@ -1,351 +0,0 @@ -//! The immutable publication requests read, and the cohort view over it. -//! -//! [`DeltaSnapshot`] is one publication of the register's fold, resolved against the serving -//! generation and swapped whole into the cell, so a reader never sees a half-applied poll. -//! [`PlacementCohort`] wraps the snapshot a scope resolution read, and a request's assembly -//! paths answer from that one publication however the cell moves meanwhile. - -use hash_graph_temporal_versioning::{Timestamp, TransactionTime}; -use hashql_core::collections::{FastHashMap, FastHashSet}; -use type_system::knowledge::entity::id::EntityEditionId; - -use super::{DeltaEdge, DeltaNode, DeltaRevision, overlay::IdentityTableOverlay}; -use crate::{ - bitset::CompressedBitSet, - dataset::auxiliary::{Icon, Legend, OwnedIcon, OwnedLegend}, - identity::{EdgeRowId, NodeRowId, OntologyRowId}, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - serve::codec::Universe, -}; - -/// An immutable publication of the register, resolved against the serving generation. -/// -/// The withdrawn identities live here in two forms. The identity set answers for every withdrawn -/// identity whether or not the generation fitted it, because a retained cohort can hold identities -/// no generation bitset can name. The row bitsets are the fitted withdrawals resolved into the -/// generation's row domains at publication, so an admission path tests membership per admitted row -/// rather than consulting a map per candidate. The staged identities are the node-classified -/// arrivals, keyed to the edition the feed last observed, and the links are the link-classified -/// arrivals with complete attachment pairs. An unclassified arrival joins neither, so it serves -/// nothing until a verdict arrives. The published nodes carry their recorded wire coordinates -/// and rows, ready to serve wherever the caller's cohort admits them. -#[derive(Debug, PartialEq)] -pub(crate) struct DeltaSnapshot { - /// This publication's position in publication order. - pub revision: DeltaRevision, - /// The feed position the register had folded through at publication. - pub watermark: Timestamp, - /// Every withdrawn identity, fitted or not. - pub withdrawn: FastHashSet, - /// The withdrawn identities' node rows in the serving generation. - pub withdrawn_nodes: CompressedBitSet, - /// The withdrawn identities' edge rows in the serving generation. - pub withdrawn_edges: CompressedBitSet, - /// The node-classified arrivals awaiting placement, keyed to their newest feed edition. - pub staged: FastHashMap, - /// The published nodes, carrying their recorded wire coordinates. - pub nodes: FastHashMap, - /// The link-classified arrivals with complete attachment pairs, published row-typed once - /// both endpoints hold rows. - pub edges: FastHashMap, - /// One captured legend per published link and per live fitted identity whose capture read - /// has answered. A revised fitted identity with no capture stays absent here and answers - /// from the fit-time payload. - pub legends: FastHashMap, - /// Every allocated node row beside the accepted universe, as of this publication. - pub node_rows: IdentityTableOverlay, - /// Every allocated edge row beside the accepted edge universe, as of this publication. - pub edge_rows: IdentityTableOverlay, - /// The ontology rows allocated for types the generation never tabulated, as of this - /// publication. - pub ontology_rows: IdentityTableOverlay, - /// The icons recorded at the allocated ontology rows, one per row the extension holds. - /// - /// A tabulated type's icon resolves through the generation's baked closure artifact, so - /// this map carries exactly the rows past the baked bound, written once at allocation and - /// repaired by refit. - pub ontology_icons: FastHashMap, -} - -impl DeltaSnapshot { - /// Returns the accepted row universe at publication. - /// - /// The bound covers every allocated row whatever its holder's standing, so a wire id a - /// retained proof admitted keeps decoding to the same row. A request takes this one value at - /// every encode and decode in its answer. - #[must_use] - pub(crate) const fn universe(&self) -> Universe { - self.node_rows.universe() - } - - /// Returns whether the snapshot withdraws the entity `id` names, fitted or not. - #[must_use] - pub(crate) fn withdraws(&self, id: ArchivedEntityId) -> bool { - self.withdrawn.contains(&id) - } - - /// Returns the captured current legend of the entity `id` names. - /// - /// Map first, artifact second: a holder consults this before the generation's baked legend, - /// so a revised fitted identity answers with its most recently captured legend - the current - /// edition's once its capture read answers - and a fitted identity with no capture answers - /// from the fit-time payload. Every published link answers here, because publication - /// withholds a link until its legend captures. - #[must_use] - pub(crate) fn legend_of(&self, id: ArchivedEntityId) -> Option<&Legend> { - self.legends.get(&id).map(AsRef::as_ref) - } - - /// Returns whether the snapshot withdraws any identity at all, fitted or not. - /// - /// An empty set lets an arrival-bearing admission walk skip whole, the identity-domain - /// counterpart of [`Self::withdraws_any_node`]. - #[must_use] - pub(crate) fn withdraws_any(&self) -> bool { - !self.withdrawn.is_empty() - } - - /// Returns whether the snapshot captured any display at all, link or fitted. - /// - /// An empty capture map answers no overlay read, so a detail pass consults this once - /// instead of paying a per-row identity lookup that can never hit. - #[must_use] - pub(crate) fn captures_any(&self) -> bool { - !self.legends.is_empty() - } - - /// Returns whether the snapshot withdraws any fitted node at all. - /// - /// An empty projection lets the admission walk skip whole, so a snapshot subtracting no - /// node rows costs a tile request nothing beyond this question. - #[must_use] - pub(crate) fn withdraws_any_node(&self) -> bool { - !self.withdrawn_nodes.is_empty() - } - - /// Returns whether the snapshot withdraws the node in `row`. - #[must_use] - pub(crate) fn withdraws_node(&self, row: NodeRowId) -> bool { - self.withdrawn_nodes.contains(row) - } - - /// Returns whether the snapshot withdraws the edge in `row`. - #[must_use] - pub(crate) fn withdraws_edge(&self, row: EdgeRowId) -> bool { - self.withdrawn_edges.contains(row) - } - - /// Iterates the withdrawn node rows, the fitted withdrawals in the node domain. - /// - /// The edges route subtracts these from its bounding set, which is what tiles rendered, so - /// the two routes keep answering from one delivered world. - pub(crate) fn withdrawn_node_rows(&self) -> impl Iterator + '_ { - self.withdrawn_nodes.iter() - } - - /// Iterates the withdrawn edge rows, the fitted withdrawals in the edge domain. - /// - /// An entry fold subtracts these from a scoped proof's edge mask at resolution, exactly as - /// [`Self::withdrawn_node_rows`] feeds the node mask, so the folded proof and the admission - /// checks answer from one withdrawn set. - pub(crate) fn withdrawn_edge_rows(&self) -> impl Iterator + '_ { - self.withdrawn_edges.iter() - } - - /// Returns every staged arrival, keyed to its newest feed edition. - #[must_use] - pub(crate) const fn staged_arrivals(&self) -> &FastHashMap { - &self.staged - } - - /// Returns the published edge `id` names, or [`None`] for an identity with no published edge. - #[must_use] - pub(crate) fn edge(&self, id: ArchivedEntityId) -> Option { - self.edges.get(&id).copied() - } - - /// Returns the published node `id` names, or [`None`] for an identity with no published node. - #[must_use] - pub(crate) fn node(&self, id: ArchivedEntityId) -> Option<&DeltaNode> { - self.nodes.get(&id) - } - - /// Returns the published node holding `row`, or [`None`] for a row this publication does not - /// serve. - /// - /// [`Self::node`] reversed: wire-domain ingress decodes to an allocated row and resolves the - /// identity serving it here. The extension answers every allocated row, and the node map - /// filters it to the published holders, so a dormant holder's row resolves to [`None`]. - #[must_use] - pub(crate) fn node_at(&self, row: NodeRowId) -> Option<(ArchivedEntityId, &DeltaNode)> { - let identity = self.node_rows.id_of(row)?; - let node = self.nodes.get(&identity)?; - - Some((identity, node)) - } - - /// Returns every published node, carrying its recorded wire coordinate. - #[must_use] - pub(crate) const fn nodes(&self) -> &FastHashMap { - &self.nodes - } - - /// Returns every published edge, carrying its endpoint identities. - #[must_use] - pub(crate) const fn edges(&self) -> &FastHashMap { - &self.edges - } - - /// Returns the identity of the allocated ontology row `row`, or [`None`] below the baked - /// bound, whose rows the generation's own table answers. - #[must_use] - pub(crate) fn ontology_id_of(&self, row: OntologyRowId) -> Option { - self.ontology_rows.id_of(row) - } - - /// Returns the icon recorded at the allocated ontology row `row`, or [`None`] below the - /// baked bound, whose icons the generation's closure artifact resolves. - /// - /// Every allocated row answers, because allocation records an icon beside the row it - /// allocates - the empty icon when the type's chain declares none. - #[must_use] - pub(crate) fn allocated_icon_of(&self, row: OntologyRowId) -> Option<&Icon> { - self.ontology_icons.get(&row).map(|icon| &**icon) - } -} - -/// The arrivals snapshot one scope resolution read, as a request borrows it. -/// -/// A scope resolution reads exactly one snapshot and resolves its proof against that snapshot's -/// placed set, and the cache entry binds the snapshot for its lifetime - the entry's placement -/// cohort. Every arrival-sensitive read takes slots, placement payload, and the accepted row -/// universe from this value, so a publication landing mid-window moves nothing a held entry -/// serves. The request's ingress capture stays the admission-time withdrawal authority: the -/// current withdrawn identity set filters what a retained cohort serves. The entry's masks -/// already fold this snapshot's own withdrawals, so the capture's admission work is the -/// residue that published after the entry resolved. -/// -/// An empty cohort is the resolution that read no publication - a serve that starts no consumer, -/// or a scope resolved before the first poll completes. No arrival serves through it, and the -/// universe stays the generation's own. -#[derive(Debug, Copy, Clone)] -pub(crate) struct PlacementCohort<'scope> { - /// The snapshot the resolution read, absent when it read none. - snapshot: Option<&'scope DeltaSnapshot>, -} - -impl<'scope> PlacementCohort<'scope> { - /// The cohort of a resolution that read no publication. - pub(crate) const EMPTY: Self = Self { snapshot: None }; - - /// Borrows `snapshot` as one resolution's cohort. - pub(crate) const fn of(snapshot: Option<&'scope DeltaSnapshot>) -> Self { - Self { snapshot } - } - - /// Returns the captured current legend of the entity `id` names, [`None`] for a cohort - /// that read no publication. - /// - /// The read is [`DeltaSnapshot::legend_of`]'s, out of the one snapshot the entry's whole - /// resolution bound. A published link answers here, and a fitted identity answers with its - /// captured legend once its capture read has answered. A fitted identity holding no capture - /// answers [`None`], and the holder serves the generation's baked legend. - pub(crate) fn legend_of(self, id: ArchivedEntityId) -> Option<&'scope Legend> { - self.snapshot?.legend_of(id) - } - - /// Returns whether this cohort captured any display at all. - /// - /// A cohort that read no publication answers `false`. Detail passes hoist this ahead of - /// their per-row overlay reads: a captureless cohort answers [`None`] from - /// [`Self::legend_of`] for every identity, so a pass checks once per response instead of - /// once per delivered row. - #[must_use] - #[expect( - clippy::missing_const_for_fn, - reason = "false positive: the delegate reads a hash map's emptiness, which is not const" - )] - pub(crate) fn captures_any(self) -> bool { - self.snapshot.is_some_and(DeltaSnapshot::captures_any) - } - - /// Returns the published node `id` names in this cohort, [`None`] for every other identity. - /// - /// An identity placed after the cohort's snapshot published answers [`None`], so a caller - /// meets it at its next resolution rather than mid-window. - #[must_use] - pub(crate) fn node(self, id: ArchivedEntityId) -> Option<&'scope DeltaNode> { - self.snapshot?.node(id) - } - - /// Returns the published node holding `row` in this cohort, [`None`] for every other row. - /// - /// The wire-ingress counterpart of [`Self::node`]: a decoded row at or past the generation's - /// fitted bound names an allocated row, and this answers the identity serving it. - #[must_use] - pub(crate) fn node_at(self, row: NodeRowId) -> Option<(ArchivedEntityId, &'scope DeltaNode)> { - self.snapshot?.node_at(row) - } - - /// Iterates the cohort's published nodes, empty for the empty cohort. - /// - /// Iteration order is the map's own. A consumer whose output must be deterministic orders - /// the nodes itself, by identity. - pub(crate) fn nodes(self) -> impl Iterator { - self.snapshot - .into_iter() - .flat_map(|snapshot| snapshot.nodes().iter().map(|(&id, node)| (id, node))) - } - - /// Returns the published edge `id` names in this cohort, [`None`] for every other identity. - /// - /// An edge published after the cohort's snapshot answers [`None`], so a caller meets it at - /// its next resolution rather than mid-window. - #[must_use] - pub(crate) fn edge(self, id: ArchivedEntityId) -> Option { - self.snapshot?.edge(id) - } - - /// Iterates the cohort's published edges, empty for the empty cohort. - /// - /// Iteration order is the map's own. A consumer whose output must be deterministic orders - /// the edges itself, by identity. - pub(crate) fn edges(self) -> impl Iterator { - self.snapshot - .into_iter() - .flat_map(|snapshot| snapshot.edges().iter().map(|(&id, &edge)| (id, edge))) - } - - /// Returns the identity of the allocated ontology row `row`, [`None`] below the baked bound - /// or for a cohort that read no publication. - /// - /// Rows below the bound answer from the generation's own table, so a caller resolving a - /// representative row consults the table first and this second. - #[must_use] - pub(crate) fn ontology_id_of(self, row: OntologyRowId) -> Option { - self.snapshot?.ontology_id_of(row) - } - - /// Returns the icon recorded at the allocated ontology row `row`, [`None`] below the baked - /// bound or for a cohort that read no publication. - /// - /// The tile trailer resolves an arrival legend's representative here when its row lies - /// past the generation's tabulated types, whose icons the baked closure artifact resolves - /// instead. - #[must_use] - pub(crate) fn allocated_icon_of(self, row: OntologyRowId) -> Option<&'scope Icon> { - self.snapshot?.allocated_icon_of(row) - } - - /// Returns the accepted row universe reads under this cohort take, `base` for an empty one. - /// - /// The bound is the snapshot's own. Every slot the cohort can name lies inside it, and a - /// slot allocated after the snapshot published refuses at decode. - #[must_use] - pub(crate) const fn universe(self, base: Universe) -> Universe { - match self.snapshot { - Some(snapshot) => snapshot.universe(), - None => base, - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/staging.rs b/libs/@local/graph/atlas/src/serve/delta/staging.rs deleted file mode 100644 index 5ff4d80aefb..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/staging.rs +++ /dev/null @@ -1,749 +0,0 @@ -//! The staging arm walking classified arrivals toward a placeable embedding. -//! -//! [`StagingArm`] is a long-lived task beside the poll arm, built when serving starts. The poll -//! arm owns the feed fold and withdrawal publication, and this arm owns the arrivals pipeline: -//! each cycle it mirrors the published snapshot's staged arrivals, batch-reads their stored -//! embeddings, and walks each arrival's retry state. The poll arm and this arm share the -//! published [`DeltaCell`] alone, so a stalled ensure or a slow embedding read degrades arrival -//! freshness and nothing else - withdrawal publication continues however long a staging cycle -//! runs. -//! -//! One arrival's phases, and every move between them: -//! -//! ```text -//! enter - the published staged set names a fresh identity -//! | -//! v -//! Reading ------ a granted ensure after the ------> Ensured -//! | spent reading budget | \ -//! | | \ the spent post-ensure budget -//! |<----- the embedding read answers, either side --+ \ -//! v '--> Exhausted -//! Ready, display pending (reconciliation or refit) -//! | -//! | the display read answers -//! v -//! Ready, display captured -//! | -//! +---- an in-frame projection the publisher accepts ----> HandedOff -//! | -//! +---- an out-of-frame or non-finite projection --------> OutOfFrame (until refit) -//! -//! leave - an identity the staged set stops naming drops from any phase -//! ``` -//! -//! The pipeline restates the accepted embedding-retry contract, whose one observable is the -//! store read. Entity insert and patch commit before their embedding workflow starts, and the -//! worker writes the embeddings table without a second feed event, so the indexed read is what -//! notices the write. A pending arrival reads for [`DeltaPolling::retry_polls`] consecutive -//! cycles. The cycle that spends the reading budget submits a deduplicated ensure, only after -//! its own read missed - the contract's final read before creating any work. The ensure names -//! the workflow by identity plus edition, the Temporal server's refusal of a concurrent -//! duplicate start is the deduplication, and an already-started refusal counts as a successful -//! ensure. After the ensure, the read continues for the same budget again, and pending ends at a -//! returned row or at exhaustion, never at workflow terminality. A start is not completion. An -//! already-started refusal returns no run to await, so terminality is unobservable by -//! construction. -//! -//! Every failure fails closed and stays pending. A failed ensure start logs and retries at the -//! next cycle. A deployment with no Temporal client stages arrivals and never ensures. An -//! exhausted arrival logs and stays unplaced until reconciliation or refit, the feed contract's -//! own disposition for its losses, and an embedding written after exhaustion reads the same way -//! until the refit repairs it. -//! -//! The pipeline mirrors the published staged set. An identity that leaves it - a withdrawal - -//! drops its pipeline state, and an identity that returns - an unarchive the register never -//! placed - re-enters pending with a fresh budget. A pending arrival's edition tracks the -//! newest feed edition, so an ensure always names the current one. -//! -//! A completed arrival places in the same cycle. The arm projects every ready embedding through -//! the generation's own publish path ([`Placer`]) and hands each in-frame coordinate to the poll -//! arm's publisher over the placement channel, keeping the register's single writer. An -//! out-of-frame or non-finite projection parks the arrival under a warning until a refit -//! recalibrates the frame, and a serving process without a placer - a baseline-placed -//! generation, the one shape that serves without one - stages arrivals forever. - -use alloc::sync::Arc; -use core::fmt; -use std::collections::HashMap; - -use error_stack::Report; -use hash_graph_authorization::policies::store::PrincipalStore; -use hash_graph_postgres_store::store::{AsClient, PostgresStorePool, error::StoreError}; -use hash_graph_store::pool::StorePool as _; -use hash_temporal_client::TemporalClient; -use hashql_core::collections::{FastHashMap, FastHashMapEntry, fast_hash_map, fast_hash_set}; -use tokio::{ - sync::mpsc::{Sender, error::TrySendError}, - time::MissedTickBehavior, -}; -use type_system::{ - knowledge::entity::id::{EntityEditionId, EntityId}, - ontology::id::BaseUrl, - principal::actor::ActorEntityUuid, -}; - -use super::{ - DeltaCell, ProjectedArrival, - consumer::DeltaPolling, - placement::{NonFiniteProjection, Placer, Projection}, -}; -use crate::{ - dataset::{PROJECTOR_DIMENSIONS, postgres::PostgresDatasetError}, - math::BoxedVecN, - postgres::{ - edition_display::DisplayParts, id::ArchivedEntityId, read_edition_displays, - read_projector_embeddings, - }, -}; - -/// The client half of the deduplicated embedding ensure. -/// -/// A deployment with no Temporal client configured builds no value of this type: its arrivals -/// stage and never ensure, which fails closed. -#[derive(Debug)] -pub struct EmbeddingWorkflow { - /// The client the ensure starts workflows through. - pub temporal: TemporalClient, - /// The property exclusions every ensured workflow receives. - /// - /// The exclusions match the store's own workflow starts, read from the same - /// filter-protection configuration, because an ensure without them would embed protected - /// properties. - pub exclusions: HashMap>, -} - -/// One pending arrival's position in the retry contract. -#[derive(Debug)] -enum Phase { - /// The embedding read has this many cycles left before the ensure submits. - /// - /// Zero means the reading budget ran out without a successful ensure, so every further - /// missed read submits it again. - Reading { - /// The pre-ensure read cycles remaining. - reads_left: u32, - }, - /// The ensure succeeded, and the read has this many cycles left before exhaustion. - Ensured { - /// The post-ensure read cycles remaining. - reads_left: u32, - }, - /// The read returned a row, and the embedding awaits placement. - Ready { - /// The stored whole-entity embedding's l2-normalized projector prefix. - embedding: BoxedVecN, - /// The display parts read for the recorded edition, absent until its read answers. - /// - /// Placement waits for the read, so a placed arrival always carries the label, icon, - /// and representative type its store row stated at the hand-off. - display: Option, - }, - /// The publisher holds the placement, and the published staged set retires the entry. - HandedOff, - /// The embedding projects outside the fitted world frame or to a non-finite point, and the - /// arrival stays unplaced until a refit recalibrates the frame. - OutOfFrame, - /// Both budgets ran out, and the arrival stays unplaced until reconciliation or refit. - Exhausted, -} - -/// One staged arrival's pipeline state. -#[derive(Debug)] -struct StagingEntry { - /// The arrival's newest feed edition, the ensure's deduplication component. - edition: EntityEditionId, - /// The arrival's position in the retry contract. - phase: Phase, -} - -/// What one missed read obliges for its arrival. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(super) enum MissAction { - /// A budget still stands, and the next cycle reads again. - Wait, - /// The reading budget ran out, so the arm submits the deduplicated ensure for the carried - /// edition. - Ensure(EntityEditionId), - /// The post-ensure budget ran out, and the arrival parks until reconciliation or refit. - Park, -} - -/// The retry state per staged arrival, mirroring the published staged set. -#[derive(Debug)] -pub(super) struct StagingPipeline { - /// The per-side retry budget, in cycles. - budget: u32, - /// The pipeline state per staged identity. - entries: FastHashMap, -} - -impl StagingPipeline { - /// Builds an empty pipeline with `budget` read cycles on each side of the ensure. - pub(super) fn new(budget: u32) -> Self { - Self { - budget, - entries: fast_hash_map(), - } - } - - /// Mirrors the published staged set into the pipeline. - /// - /// A fresh identity enters pending with the full reading budget. An identity absent from - /// `staged` - a withdrawal - drops with its state, so an unarchived identity re-enters - /// pending from the start. A held identity keeps its phase while its edition tracks the - /// newest feed edition, so an ensure always names the current one. - pub(super) fn sync(&mut self, staged: &FastHashMap) { - self.entries - .retain(|identity, _| staged.contains_key(identity)); - - for (&identity, &edition) in staged { - match self.entries.entry(identity) { - FastHashMapEntry::Vacant(slot) => { - slot.insert(StagingEntry { - edition, - phase: Phase::Reading { - reads_left: self.budget, - }, - }); - } - FastHashMapEntry::Occupied(mut slot) => { - slot.get_mut().edition = edition; - } - } - } - } - - /// Lists every arrival whose next read is due, pre-ensure and post-ensure alike. - pub(super) fn pending(&self) -> Vec { - self.entries - .iter() - .filter(|&(_, entry)| { - matches!(entry.phase, Phase::Reading { .. } | Phase::Ensured { .. }) - }) - .map(|(&identity, _)| identity) - .collect() - } - - /// Completes one arrival: the read returned its row, and the embedding awaits placement. - /// - /// A completion for an identity the pipeline holds nothing pending for changes nothing, so a - /// row racing a withdrawal drops rather than resurrects. - pub(super) fn complete( - &mut self, - identity: ArchivedEntityId, - embedding: BoxedVecN, - ) { - if let Some(entry) = self.entries.get_mut(&identity) - && matches!(entry.phase, Phase::Reading { .. } | Phase::Ensured { .. }) - { - entry.phase = Phase::Ready { - embedding, - display: None, - }; - } - } - - /// Lists the completed arrivals lacking a captured display, keyed to their recorded editions. - pub(super) fn uncaptured(&self) -> Vec<(ArchivedEntityId, EntityEditionId)> { - self.entries - .iter() - .filter(|&(_, entry)| matches!(entry.phase, Phase::Ready { display: None, .. })) - .map(|(&identity, entry)| (identity, entry.edition)) - .collect() - } - - /// Records one completed arrival's display parts. - /// - /// An answer for an identity not awaiting one changes nothing, so an answer racing a - /// withdrawal drops rather than resurrects. - pub(super) fn captured(&mut self, identity: ArchivedEntityId, payload: DisplayParts) { - if let Some(entry) = self.entries.get_mut(&identity) - && let Phase::Ready { display, .. } = &mut entry.phase - && display.is_none() - { - *display = Some(payload); - } - } - - /// Records one missed read, returning what the miss obliges. - /// - /// A miss inside a budget waits for the next cycle. The miss that spends the reading budget - /// obliges the ensure, and every later miss without a successful ensure obliges it again, so - /// a failed start retries each cycle. The miss that spends the post-ensure budget parks the - /// arrival. A miss for an identity not pending changes nothing and waits. - pub(super) fn miss(&mut self, identity: ArchivedEntityId) -> MissAction { - let Some(entry) = self.entries.get_mut(&identity) else { - return MissAction::Wait; - }; - - match &mut entry.phase { - Phase::Reading { reads_left } => { - *reads_left = reads_left.saturating_sub(1); - if *reads_left == 0 { - MissAction::Ensure(entry.edition) - } else { - MissAction::Wait - } - } - Phase::Ensured { reads_left } => { - *reads_left = reads_left.saturating_sub(1); - if *reads_left == 0 { - entry.phase = Phase::Exhausted; - MissAction::Park - } else { - MissAction::Wait - } - } - Phase::Ready { .. } | Phase::HandedOff | Phase::OutOfFrame | Phase::Exhausted => { - MissAction::Wait - } - } - } - - /// Grants the post-ensure budget after a successful ensure. - pub(super) fn ensured(&mut self, identity: ArchivedEntityId) { - if let Some(entry) = self.entries.get_mut(&identity) - && matches!(entry.phase, Phase::Reading { .. }) - { - entry.phase = Phase::Ensured { - reads_left: self.budget, - }; - } - } - - /// Lists the completed arrivals holding a captured display, awaiting placement. - pub(super) fn ready( - &self, - ) -> impl Iterator< - Item = ( - ArchivedEntityId, - EntityEditionId, - &BoxedVecN, - &DisplayParts, - ), - > { - self.entries - .iter() - .filter_map(|(&identity, entry)| match &entry.phase { - Phase::Ready { - embedding, - display: Some(display), - } => Some((identity, entry.edition, embedding, display)), - Phase::Ready { display: None, .. } - | Phase::Reading { .. } - | Phase::Ensured { .. } - | Phase::HandedOff - | Phase::OutOfFrame - | Phase::Exhausted => None, - }) - } - - /// Retires one arrival whose placement the publisher now holds. - /// - /// The entry stays until the published staged set stops naming the identity, so a cycle - /// running between the hand-off and the next publication neither re-reads nor re-places it. - pub(super) fn handed_off(&mut self, identity: ArchivedEntityId) { - if let Some(entry) = self.entries.get_mut(&identity) { - entry.phase = Phase::HandedOff; - } - } - - /// Parks one arrival whose embedding projects outside the fitted world frame or to a - /// non-finite point. - /// - /// Only a refit moves the frame, so the same embedding projects outside it at every retry - /// and the arrival stays unplaced until one runs. - pub(super) fn out_of_frame(&mut self, identity: ArchivedEntityId) { - if let Some(entry) = self.entries.get_mut(&identity) { - entry.phase = Phase::OutOfFrame; - } - } -} - -/// One staging cycle failed against the store. -#[derive(Debug)] -pub(crate) enum StagingError { - /// No connection was available for the cycle. - Connect(Report), - /// The embedding read failed or one of its rows did not decode. - Read(PostgresDatasetError), -} - -impl fmt::Display for StagingError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Connect(report) => { - write!( - fmt, - "the staging cycle reached no store connection: {report}" - ) - } - Self::Read(error) => write!(fmt, "the embedding read failed: {error}"), - } - } -} - -impl core::error::Error for StagingError {} - -/// One ready arrival's owned hand-off key: its identity, edition and display parts. -type PlacementKey = (ArchivedEntityId, EntityEditionId, DisplayParts); - -/// Every row's placement from one projected batch, or the batch's first non-finite row. -type ProjectionOutcome = Result, NonFiniteProjection>; - -/// The long-lived task owning one generation's arrivals pipeline. -/// -/// One arm serves one generation beside its poll arm. Dropping the task drops the pipeline with -/// it, and a restart rebuilds pending state from the published staged set, so the process keeps -/// no staging state anywhere durable. -#[derive(Debug)] -pub(crate) struct StagingArm { - /// The store pool the embedding reads and the actor resolution go through. - pool: Arc, - /// The cell carrying the poll arm's publications. - cell: Arc, - /// The polling knobs, shared with the poll arm. - polling: DeltaPolling, - /// The ensure client and its exclusions, or [`None`] to stage without ensuring. - workflow: Option, - /// The generation's publish path, or [`None`] for a baseline-placed generation, which - /// stages without placing. - placer: Option, - /// The channel carrying projected arrivals to the poll arm's publisher, bounded at - /// [`DeltaPolling::placement_backlog`]. - placements: Sender<(ArchivedEntityId, ProjectedArrival)>, - /// The retry state per staged arrival. - pipeline: StagingPipeline, - /// The resolved ensure actor, cached at first use. - actor: Option, -} - -impl StagingArm { - /// Builds the arm over the cell the poll arm publishes into. - /// - /// `placements` carries every projected arrival to the poll arm's publisher, and `placer` - /// being [`None`] stages arrivals without ever placing them, the fail-closed disposition its - /// construction already logged. - pub(crate) fn new( - pool: Arc, - cell: Arc, - polling: DeltaPolling, - workflow: Option, - placer: Option, - placements: Sender<(ArchivedEntityId, ProjectedArrival)>, - ) -> Self { - let pipeline = StagingPipeline::new(polling.retry_polls); - Self { - pool, - cell, - polling, - workflow, - placer, - placements, - pipeline, - actor: None, - } - } - - /// Cycles until the owner drops the task. - /// - /// One tick per [`DeltaPolling::interval`], and a cycle running past its tick delays the - /// next rather than bursting to catch up. A failed cycle moves no budget, and the next tick - /// retries every pending arrival. - pub(crate) async fn run(mut self) -> ! { - let mut ticks = tokio::time::interval(self.polling.interval); - ticks.set_missed_tick_behavior(MissedTickBehavior::Delay); - - loop { - ticks.tick().await; - - if let Err(error) = self.cycle().await { - tracing::warn!(%error, "a staging cycle failed, pending arrivals retry next cycle"); - } - } - } - - /// Runs one staging cycle. - /// - /// A cycle mirrors the published staged set into the pipeline and reads every pending - /// arrival's embedding in one batch. Each miss then walks its retry state. - /// - /// # Errors - /// - /// Returns [`StagingError::Connect`] when no connection was available for the cycle, and - /// [`StagingError::Read`] when the embedding read failed or one of its rows did not decode. - /// Neither failure moves a budget, because a budget counts reads the store answered rather - /// than cycles the infrastructure lost. - async fn cycle(&mut self) -> Result<(), StagingError> { - { - let guard = self.cell.load(); - let Some(snapshot) = guard.as_deref() else { - return Ok(()); - }; - self.pipeline.sync(snapshot.staged_arrivals()); - } - - let pending = self.pipeline.pending(); - if !pending.is_empty() || !self.pipeline.uncaptured().is_empty() { - let mut store = self - .pool - .acquire(None) - .await - .map_err(|report| StagingError::Connect(report.change_context(StoreError)))?; - - if !pending.is_empty() { - let answers = read_projector_embeddings(&store, pending.iter().copied()) - .await - .map_err(StagingError::Read)?; - - let mut answered = fast_hash_set(); - for (identity, embedding) in answers { - answered.insert(identity); - self.pipeline.complete(identity, embedding); - } - - for identity in pending { - if answered.contains(&identity) { - continue; - } - - match self.pipeline.miss(identity) { - MissAction::Wait => {} - MissAction::Ensure(edition) => { - self.submit_ensure(&mut store, identity, edition).await; - } - MissAction::Park => tracing::warn!( - ?identity, - "an arrival's retry budget ran out, it stays unplaced until \ - reconciliation or refit" - ), - } - } - } - - // The capture runs after the completions, so an arrival that became ready this - // cycle captures and places in the same cycle. - self.capture_displays(&store).await; - } - - self.place_ready(); - - Ok(()) - } - - /// Captures the display payload of every completed arrival that lacks one. - /// - /// One batched read keyed by recorded edition, on the cycle's held connection. Failure - /// changes nothing beyond a warning: the arrivals stay uncaptured and their placements - /// defer to the next cycle's retry, so a failed capture degrades arrival freshness and - /// nothing else. - async fn capture_displays(&mut self, store: &impl AsClient) { - let uncaptured = self.pipeline.uncaptured(); - if uncaptured.is_empty() { - return; - } - - let answers = match read_edition_displays( - store, - uncaptured.iter().map(|&(_, edition)| edition), - ) - .await - { - Ok(answers) => answers, - Err(error) => { - tracing::warn!( - %error, - arrivals = uncaptured.len(), - "the display read failed, uncaptured arrivals defer placement to the \ - next cycle" - ); - return; - } - }; - - // The statement answers every requested edition exactly once, so a count below the - // request count is a broken store invariant rather than a lookup miss. - if answers.len() != uncaptured.len() { - tracing::warn!( - answers = answers.len(), - requests = uncaptured.len(), - "the display read answered a different edition count than the request named" - ); - } - - // An answer without a resolved representative type stays uncaptured, so the next cycle - // reads it again: a legend cannot exist until the representative resolves. - let by_edition: FastHashMap = answers - .into_iter() - .filter_map(|(edition, display)| display.map(|display| (edition, display))) - .collect(); - for (identity, edition) in uncaptured { - if let Some(display) = by_edition.get(&edition) { - self.pipeline.captured(identity, display.clone()); - } - } - } - - /// Projects every ready arrival and hands each placement to the publisher. - /// - /// Placement is one batched projection through the publish path. An in-frame coordinate - /// travels the placement channel and retires its pipeline entry, an out-of-frame coordinate - /// parks the arrival until refit, and a non-finite projection parks its own row while the - /// rows behind it retry at the next cycle. A full channel and a closed one both leave every - /// remaining entry ready, so the next cycle retries the hand-off. The entry retires on the - /// send that landed, never on the projection, and the bounded channel therefore loses no - /// placement. - fn place_ready(&mut self) { - let Some((keyed, outcome)) = self.project_ready() else { - return; - }; - - match outcome { - Ok(projections) => { - for ( - ( - identity, - edition, - DisplayParts { - label, - icon, - representative, - }, - ), - projection, - ) in keyed.into_iter().zip(projections) - { - match projection { - Projection::Placed { wire } => { - match self.placements.try_send(( - identity, - ProjectedArrival { - edition, - position: wire, - label, - icon, - representative, - }, - )) { - Ok(()) => {} - Err(TrySendError::Full(_)) => { - tracing::warn!( - "the placement channel is full, the remaining placements \ - retry next cycle" - ); - return; - } - Err(TrySendError::Closed(_)) => { - tracing::warn!( - "the placement channel closed, placements retry next cycle" - ); - return; - } - } - self.pipeline.handed_off(identity); - } - Projection::OutOfFrame { world } => { - tracing::warn!( - ?identity, - ?world, - "an arrival projects outside the fitted world frame, it stays \ - unplaced until a refit recalibrates the frame" - ); - self.pipeline.out_of_frame(identity); - } - } - } - } - Err(failure) => { - let (identity, _, _) = keyed[failure.row]; - tracing::warn!( - ?identity, - "an arrival's projection is non-finite, it stays unplaced until a refit" - ); - self.pipeline.out_of_frame(identity); - } - } - } - - /// Projects the ready batch through the publish path, keyed for the hand-off. - /// - /// The keys own their identity, edition and display, so the caller walks the outcome while - /// mutating the pipeline. [`None`] means the arm holds no placer or nothing is ready. - fn project_ready(&self) -> Option<(Vec, ProjectionOutcome)> { - let placer = self.placer.as_ref()?; - - let batch: Vec<_> = self.pipeline.ready().collect(); - if batch.is_empty() { - return None; - } - - let keyed = batch - .iter() - .map(|&(identity, edition, _, display)| (identity, edition, display.clone())) - .collect(); - let outcome = placer.project(batch.iter().map(|&(_, _, embedding, _)| embedding)); - - Some((keyed, outcome)) - } - - /// Submits the deduplicated ensure for one arrival whose reading budget ran out. - /// - /// Every failure leaves the arrival pending. A missing Temporal client never ensures. An - /// unresolved actor and a failed start both log, and the next cycle retries them. A - /// successful start and an already-started refusal both grant the post-ensure budget. - async fn submit_ensure( - &mut self, - store: &mut impl PrincipalStore, - identity: ArchivedEntityId, - edition: EntityEditionId, - ) { - let Some(workflow) = &self.workflow else { - return; - }; - - let actor = match self.actor { - Some(actor) => actor, - None => match store.get_or_create_system_machine("h").await { - Ok(machine) => { - let actor = ActorEntityUuid::from(machine); - self.actor = Some(actor); - actor - } - Err(error) => { - tracing::warn!( - %error, - ?identity, - "the ensure actor did not resolve, the arrival stays pending" - ); - return; - } - }, - }; - - let entity = EntityId::from(identity); - let workflow_id = format!("atlas-embedding-{entity}-{}", edition.as_uuid()); - match workflow - .temporal - .ensure_update_entity_embeddings_workflow( - workflow_id, - actor, - entity, - &workflow.exclusions, - ) - .await - { - Ok(start) => { - tracing::debug!(?identity, ?start, "the embedding ensure is in flight"); - self.pipeline.ensured(identity); - } - Err(error) => { - tracing::warn!( - %error, - ?identity, - "the embedding ensure did not start, the arrival stays pending" - ); - } - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/delta/tests.rs b/libs/@local/graph/atlas/src/serve/delta/tests.rs deleted file mode 100644 index 756924da57f..00000000000 --- a/libs/@local/graph/atlas/src/serve/delta/tests.rs +++ /dev/null @@ -1,1333 +0,0 @@ -use hash_graph_postgres_store::store::{EntityDeletion, EntityEnd, EntityEvent, EntityUpdate}; -use hash_graph_temporal_versioning::{Timestamp, TransactionTime}; -use hashql_core::collections::FastHashMap; -use type_system::knowledge::entity::{ - EntityId, - id::{EntityEditionId, EntityUuid}, - provenance::EntityDeletionProvenance, -}; -use uuid::Uuid; - -use super::{ - DeltaCell, DeltaEdge, DeltaEvent, DeltaNode, DeltaRegister, DeltaRevision, Disposition, - IdentityTables, ProjectedArrival, Standing, - consumer::{DeltaPolling, PollOutcome}, - register::UniverseExhausted, - staging::{MissAction, StagingPipeline}, -}; -use crate::{ - dataset::auxiliary::{Icon, Label, OwnedIcon, OwnedLabel, OwnedLegend}, - identity::{EdgeRowId, NodeRowId, OntologyRowId}, - math::{BoxedVecN, Vec2}, - postgres::{ - Classification, - edition_display::DisplayParts, - id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedOntologyTypeUuid}, - }, - serve::codec::Universe, -}; - -/// The fixture generation's base row bound, past every fitted row the tables name. -const BASE: u32 = 100; - -/// The fixture generation's tabulated type bound, where ontology row allocation starts. -const ONTOLOGY_BASE: u64 = 8; - -/// The fixture generation's fitted edge bound, where edge row allocation starts. -const EDGE_BASE: u64 = 40; - -/// An empty register whose row allocation starts at the fixture's bounds. -fn register() -> DeltaRegister { - DeltaRegister::new( - Universe::new(slot(BASE)), - Universe::new(EdgeRowId::new(EDGE_BASE)), - Universe::new(OntologyRowId::new(ONTOLOGY_BASE)), - ) -} - -/// The row id `n` names in the fixture's allocated node domain. -fn slot(n: u32) -> NodeRowId { - NodeRowId::new(u64::from(n)) -} - -/// Identity tables over a handful of fitted rows, standing in for a generation. -#[derive(Debug, Default)] -struct Tables { - nodes: Vec<(ArchivedEntityId, NodeRowId)>, - edges: Vec<(ArchivedEntityId, EdgeRowId)>, - ontology: Vec<(ArchivedOntologyTypeUuid, OntologyRowId)>, -} - -impl IdentityTables for Tables { - fn node_row_of(&self, id: ArchivedEntityId) -> Option { - self.nodes - .iter() - .find(|&&(key, _)| key == id) - .map(|&(_, row)| row) - } - - fn edge_row_of(&self, id: ArchivedEntityId) -> Option { - self.edges - .iter() - .find(|&&(key, _)| key == id) - .map(|&(_, row)| row) - } - - fn ontology_row_of(&self, id: ArchivedOntologyTypeUuid) -> Option { - self.ontology - .iter() - .find(|&&(key, _)| key == id) - .map(|&(_, row)| row) - } -} - -fn entity(n: u128) -> ArchivedEntityId { - ArchivedEntityId { - web_id: Uuid::from_u128(0xAB).into(), - entity_uuid: ArchivedEntityUuid::from_bytes(Uuid::from_u128(n).into_bytes()), - } -} - -fn at(seconds: i64) -> Timestamp { - Timestamp::from_unix_timestamp(seconds) -} - -fn edition(n: u128) -> EntityEditionId { - EntityEditionId::new(Uuid::from_u128(n)) -} - -fn live(n: u128) -> Standing { - Standing::Live { - edition: edition(n), - } -} - -/// The fixture's shared representative type, unknown to the tables, so the register's own -/// extension allocates its row. -fn type_uuid() -> ArchivedOntologyTypeUuid { - ArchivedOntologyTypeUuid::from(Uuid::from_u128(0xE0)) -} - -/// A display read answer carrying `text`, the fixture icon, and the fixture's shared -/// representative type. -fn display(text: &str) -> DisplayParts { - DisplayParts { - label: OwnedLabel::from(text), - icon: OwnedIcon::from("capture-icon"), - representative: type_uuid(), - } -} - -/// The legend the fixture's display read produces: the extension's first ontology row. -fn legend(text: &str) -> OwnedLegend { - OwnedLegend::new(OntologyRowId::new(ONTOLOGY_BASE), Label::new(text)) -} - -/// Captures `text` for the identity through `tables`, panicking where the fixture cannot -/// exhaust the ontology domain. -fn capture( - register: &mut DeltaRegister, - entity_n: u128, - edition_n: u128, - text: &str, - tables: &Tables, -) { - let DisplayParts { - label, - icon, - representative, - } = display(text); - register - .capture_display( - entity(entity_n), - edition(edition_n), - &label, - &icon, - representative, - tables, - ) - .expect("the fixture ontology domain has room"); -} - -fn link_between(source: u128, target: u128) -> Classification { - Classification::Edge { - source: Some(entity(source)), - target: Some(entity(target)), - } -} - -fn store_id(n: u128) -> EntityId { - EntityId { - web_id: type_system::principal::actor_group::WebId::new(Uuid::from_u128(0xAB)), - entity_uuid: EntityUuid::new(Uuid::from_u128(n)), - draft_id: None, - } -} - -fn updated(entity_n: u128, seconds: i64, edition_n: u128, archived: bool) -> EntityEvent { - EntityEvent::Updated(EntityUpdate { - entity: store_id(entity_n), - edition: edition(edition_n), - archived, - changed_at: at(seconds), - }) -} - -fn event(entity_n: u128, seconds: i64, standing: Standing) -> DeltaEvent { - DeltaEvent { - entity: entity(entity_n), - version: at(seconds), - standing, - } -} - -/// Publishes under fixed publication inputs, so two snapshots compare on resolution alone. -fn snapshot(register: &DeltaRegister, tables: &Tables) -> super::DeltaSnapshot { - register.snapshot(tables, DeltaRevision::FIRST, at(100)) -} - -#[test] -fn newer_event_wins() { - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - assert!(register.apply(event(1, 2, Standing::Withdrawn))); - - assert!(snapshot(®ister, &Tables::default()).withdraws(entity(1))); -} - -#[test] -fn unarchive_replaces_tombstone() { - let tables = Tables { - nodes: vec![(entity(1), NodeRowId::new(4))], - ..Tables::default() - }; - let mut register = register(); - - assert!(register.apply(event(1, 1, Standing::Withdrawn))); - assert!(register.apply(event(1, 2, live(10)))); - - // The live standing cleared the tombstone, and a fitted live identity resolves to nothing. - let snapshot = snapshot(®ister, &tables); - assert!(!snapshot.withdraws(entity(1))); - assert!(!snapshot.withdraws_node(NodeRowId::new(4))); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); -} - -#[test] -fn older_event_is_ignored() { - let mut register = register(); - - assert!(register.apply(event(1, 2, Standing::Withdrawn))); - assert!(!register.apply(event(1, 1, live(10)))); - - assert!(snapshot(®ister, &Tables::default()).withdraws(entity(1))); -} - -#[test] -fn redelivery_idempotent() { - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - assert!(!register.apply(event(1, 1, live(10)))); - assert!(register.apply(event(2, 1, Standing::Withdrawn))); - assert!(!register.apply(event(2, 1, Standing::Withdrawn))); -} - -#[test] -fn equal_version_withdrawn_bias() { - let tables = Tables::default(); - - let mut withdrawal_last = register(); - assert!(withdrawal_last.apply(event(1, 1, live(10)))); - assert!(withdrawal_last.apply(event(1, 1, Standing::Withdrawn))); - - let mut withdrawal_first = register(); - assert!(withdrawal_first.apply(event(1, 1, Standing::Withdrawn))); - assert!(!withdrawal_first.apply(event(1, 1, live(10)))); - - assert!(snapshot(&withdrawal_last, &tables).withdraws(entity(1))); - assert!(snapshot(&withdrawal_first, &tables).withdraws(entity(1))); -} - -#[test] -fn version_moves_without_flipping_standing() { - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - // The same standing at a newer version replaces the register without changing resolution. - assert!(!register.apply(event(1, 3, live(10)))); - // The version moved even though resolution did not, so an older withdrawal still loses. - assert!(!register.apply(event(1, 2, Standing::Withdrawn))); - - assert!(!snapshot(®ister, &Tables::default()).withdraws(entity(1))); -} - -#[test] -fn equal_version_live_editions_converge() { - let tables = Tables::default(); - - let mut ascending = register(); - assert!(ascending.apply(event(1, 1, live(10)))); - assert!(ascending.apply(event(1, 1, live(11)))); - assert_eq!( - ascending.classify(entity(1), Classification::Node), - Ok(Disposition::Resolving) - ); - - let mut descending = register(); - assert!(descending.apply(event(1, 1, live(11)))); - assert!(!descending.apply(event(1, 1, live(10)))); - assert_eq!( - descending.classify(entity(1), Classification::Node), - Ok(Disposition::Resolving) - ); - - assert_eq!( - snapshot(&ascending, &tables) - .staged - .get(&entity(1)) - .copied(), - Some(edition(11)) - ); - assert_eq!( - snapshot(&ascending, &tables), - snapshot(&descending, &tables) - ); -} - -#[test] -fn withdrawn_fitted_subtract() { - let tables = Tables { - nodes: vec![(entity(1), NodeRowId::new(4))], - edges: vec![(entity(2), EdgeRowId::new(7))], - ..Tables::default() - }; - let mut register = register(); - - assert!(register.apply(event(1, 1, Standing::Withdrawn))); - assert!(register.apply(event(2, 1, Standing::Withdrawn))); - - let snapshot = snapshot(®ister, &tables); - assert!(snapshot.withdraws(entity(1))); - assert!(snapshot.withdraws(entity(2))); - assert!(snapshot.withdraws_node(NodeRowId::new(4))); - assert!(snapshot.withdraws_edge(EdgeRowId::new(7))); -} - -#[test] -fn withdrawn_unfitted_identity_only() { - let tables = Tables { - nodes: vec![(entity(1), NodeRowId::new(4))], - ..Tables::default() - }; - let mut register = register(); - - assert!(register.apply(event(9, 1, Standing::Withdrawn))); - - let snapshot = snapshot(®ister, &tables); - assert!(snapshot.withdraws(entity(9))); - assert!(!snapshot.withdraws_node(NodeRowId::new(4))); - assert_eq!(snapshot.staged.get(&entity(9)).copied(), None); -} - -#[test] -fn live_fitted_identity_resolves_to_nothing() { - let tables = Tables { - nodes: vec![(entity(1), NodeRowId::new(4))], - ..Tables::default() - }; - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - - let snapshot = snapshot(®ister, &tables); - assert!(!snapshot.withdraws(entity(1))); - assert!(!snapshot.withdraws_node(NodeRowId::new(4))); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); -} - -#[test] -fn arrival_stages_on_node_verdict() { - let tables = Tables::default(); - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - assert!(register.apply(event(1, 2, live(11)))); - assert!(register.apply(event(2, 1, Standing::Withdrawn))); - - // Unclassified, the arrival publishes nowhere while everything else proceeds. - let unclassified = snapshot(®ister, &tables); - assert_eq!(unclassified.staged.get(&entity(1)).copied(), None); - assert_eq!(unclassified.edge(entity(1)), None); - assert!(unclassified.withdraws(entity(2))); - - assert_eq!( - register.classify(entity(1), Classification::Node), - Ok(Disposition::Resolving) - ); - - assert_eq!( - snapshot(®ister, &tables).staged.get(&entity(1)).copied(), - Some(edition(11)) - ); -} - -#[test] -fn unclassified_live_unfitted_only() { - let tables = Tables { - nodes: vec![(entity(3), NodeRowId::new(4))], - ..Tables::default() - }; - let mut register = register(); - - // A live arrival, a withdrawn identity, a live fitted identity, a classified arrival. - assert!(register.apply(event(1, 1, live(10)))); - assert!(register.apply(event(2, 1, Standing::Withdrawn))); - assert!(register.apply(event(3, 1, live(30)))); - assert!(register.apply(event(4, 1, live(40)))); - assert_eq!( - register.classify(entity(4), Classification::Node), - Ok(Disposition::Resolving) - ); - - assert_eq!( - register.unclassified(&tables).collect::>(), - [entity(1)] - ); -} - -#[test] -fn classification_final() { - let tables = Tables::default(); - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - assert_eq!( - register.classify(entity(1), Classification::Node), - Ok(Disposition::Resolving) - ); - - // A later conflicting verdict changes nothing: the first verdict holds. - assert_eq!( - register.classify(entity(1), link_between(8, 9)), - Ok(Disposition::AlreadyHeld) - ); - - let snapshot = snapshot(®ister, &tables); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), Some(edition(10))); - assert_eq!(snapshot.edge(entity(1)), None); -} - -#[test] -fn resolving_changes_resolution() { - assert!(Disposition::Resolving.changes_resolution()); - assert!(!Disposition::Dormant.changes_resolution()); - assert!(!Disposition::AlreadyHeld.changes_resolution()); -} - -#[test] -fn dispositions_join_by_resolution_strength() { - let all = [ - Disposition::Resolving, - Disposition::Dormant, - Disposition::AlreadyHeld, - ]; - - // The nine-case table by exhaustion: AlreadyHeld is the neutral element and Resolving - // absorbs, in either operand order. - for disposition in all { - assert_eq!(disposition | Disposition::AlreadyHeld, disposition); - assert_eq!(Disposition::AlreadyHeld | disposition, disposition); - assert_eq!(disposition | Disposition::Resolving, Disposition::Resolving); - assert_eq!(Disposition::Resolving | disposition, Disposition::Resolving); - } - assert_eq!( - Disposition::Dormant | Disposition::Dormant, - Disposition::Dormant - ); - - let mut folded = Disposition::AlreadyHeld; - folded |= Disposition::Dormant; - assert_eq!(folded, Disposition::Dormant); - folded |= Disposition::Resolving; - assert_eq!(folded, Disposition::Resolving); -} - -#[test] -fn publish_withholds_uncaptured() { - // The link attaches two fitted rows, so publication resolves them row-typed and the - // capture alone decides whether the link publishes. - let tables = Tables { - nodes: vec![ - (entity(8), NodeRowId::new(2)), - (entity(9), NodeRowId::new(3)), - ], - ..Tables::default() - }; - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - assert_eq!( - register.classify(entity(1), link_between(8, 9)), - Ok(Disposition::Resolving) - ); - - // Classified but uncaptured: publication withholds the link, and the pending listing - // carries the link's edition. - let withheld = snapshot(®ister, &tables); - assert_eq!(withheld.edge(entity(1)), None); - assert_eq!(withheld.legend_of(entity(1)), None); - let pending: Vec<_> = register.pending_captures(&tables).collect(); - assert_eq!(pending, [(entity(1), edition(10))]); - - capture(&mut register, 1, 10, "wrote", &tables); - assert_eq!(register.pending_captures(&tables).count(), 0); - - let snapshot = snapshot(®ister, &tables); - assert_eq!( - snapshot.edge(entity(1)), - Some(DeltaEdge { - id: EdgeRowId::new(EDGE_BASE), - edition: edition(10), - source: NodeRowId::new(2), - target: NodeRowId::new(3), - }) - ); - assert_eq!(snapshot.legend_of(entity(1)), Some(&*legend("wrote"))); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); - assert!(!snapshot.withdraws(entity(1))); -} - -#[test] -fn revised_fitted_display() { - let tables = Tables { - nodes: vec![(entity(2), NodeRowId::new(5))], - edges: vec![(entity(3), EdgeRowId::new(7))], - ..Tables::default() - }; - let mut register = register(); - - // A post-fit edition on either fitted domain lists for a capture. - assert!(register.apply(event(2, 1, live(20)))); - assert!(register.apply(event(3, 1, live(30)))); - assert_eq!(register.pending_captures(&tables).count(), 2); - - // An uncaptured revision publishes no legend, so holders answer from the baked - // fit-time payload. - let uncaptured = snapshot(®ister, &tables); - assert_eq!(uncaptured.legend_of(entity(2)), None); - assert_eq!(uncaptured.legend_of(entity(3)), None); - - capture(&mut register, 2, 20, "revised", &tables); - let captured = snapshot(®ister, &tables); - assert_eq!(captured.legend_of(entity(2)), Some(&*legend("revised"))); - assert_eq!(captured.legend_of(entity(3)), None); - - // An edition move re-lists the identity while the held capture keeps serving. - assert!(register.apply(event(2, 2, live(21)))); - assert!( - register - .pending_captures(&tables) - .any(|(id, owed)| id == entity(2) && owed == edition(21)) - ); - let stale = snapshot(®ister, &tables); - assert_eq!(stale.legend_of(entity(2)), Some(&*legend("revised"))); -} - -/// Allocation records the first capture's icon, and a later read replaces nothing. -/// -/// Hand-derivation: `type_uuid()` is unknown to the tables, so the first capture allocates -/// ontology row `ONTOLOGY_BASE` and records its icon beside the row. The second capture names -/// the same type with a different icon and must change nothing, exactly as a later edition -/// moves no placement coordinate: the refit repairs icon staleness. -#[test] -fn allocation_records_first_icon() { - let tables = Tables::default(); - let mut register = register(); - - register - .capture_display( - entity(1), - edition(10), - Label::new("first"), - Icon::new("the first icon"), - type_uuid(), - &tables, - ) - .expect("the fixture ontology domain has room"); - register - .capture_display( - entity(2), - edition(20), - Label::new("second"), - Icon::new("a later icon"), - type_uuid(), - &tables, - ) - .expect("the fixture ontology domain has room"); - - let published = snapshot(®ister, &tables); - assert_eq!( - published.allocated_icon_of(OntologyRowId::new(ONTOLOGY_BASE)), - Some(Icon::new("the first icon")) - ); - // A row below the baked bound answers from the generation's artifacts, never here. - assert_eq!(published.allocated_icon_of(OntologyRowId::new(0)), None); -} - -/// A capture for a tabulated type stores no extension icon, whose resolution stays the baked -/// closure artifact's. -#[test] -fn tabulated_capture_stores_no_extension_icon() { - let known = ArchivedOntologyTypeUuid::from(Uuid::from_u128(0xE1)); - let tables = Tables { - ontology: vec![(known, OntologyRowId::new(3))], - ..Tables::default() - }; - let mut register = register(); - - register - .capture_display( - entity(1), - edition(10), - Label::new("known"), - Icon::new("a fetched icon"), - known, - &tables, - ) - .expect("the tabulated type allocates nothing"); - - let published = snapshot(®ister, &tables); - assert_eq!(published.allocated_icon_of(OntologyRowId::new(3)), None); -} - -#[test] -fn pending_captures_serving_only() { - let tables = Tables { - nodes: vec![(entity(4), NodeRowId::new(6))], - ..Tables::default() - }; - let mut register = register(); - - // A node-classified arrival never lists, because staging captures its display. - assert!(register.apply(event(1, 1, live(10)))); - assert_eq!( - register.classify(entity(1), Classification::Node), - Ok(Disposition::Resolving) - ); - - // An incomplete link never lists, because it never serves. - assert!(register.apply(event(2, 1, live(11)))); - assert_eq!( - register.classify( - entity(2), - Classification::Edge { - source: Some(entity(8)), - target: None, - } - ), - Ok(Disposition::Resolving) - ); - - // An unclassified arrival never lists, because no verdict says its display serves. - assert!(register.apply(event(3, 1, live(12)))); - - // A withdrawn fitted identity never lists, because it serves nothing. - assert!(register.apply(event(4, 1, Standing::Withdrawn))); - - assert_eq!(register.pending_captures(&tables).count(), 0); -} - -#[test] -fn incomplete_link_publishes_nowhere() { - let tables = Tables::default(); - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - // The verdict is new and the identity lives, so publication's input changed even though - // the incomplete pair resolves to nothing served. - assert_eq!( - register.classify( - entity(1), - Classification::Edge { - source: Some(entity(8)), - target: None, - } - ), - Ok(Disposition::Resolving) - ); - - let snapshot = snapshot(®ister, &tables); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); - assert_eq!(snapshot.edge(entity(1)), None); - assert!(!snapshot.withdraws(entity(1))); - // The verdict holds, so the identity never re-lists for classification. - assert_eq!(register.unclassified(&tables).collect::>(), []); -} - -#[test] -fn classify_withdrawn_no_input_change() { - let tables = Tables::default(); - let mut register = register(); - - assert!(register.apply(event(1, 1, Standing::Withdrawn))); - assert_eq!( - register.classify(entity(1), Classification::Node), - Ok(Disposition::Dormant) - ); - - let snapshot = snapshot(®ister, &tables); - assert!(snapshot.withdraws(entity(1))); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); -} - -#[test] -fn withdraw_unarchive_keeps_verdict() { - let tables = Tables::default(); - let mut register = register(); - - assert!(register.apply(event(1, 1, live(10)))); - assert_eq!( - register.classify(entity(1), Classification::Node), - Ok(Disposition::Resolving) - ); - assert!(register.apply(event(1, 2, Standing::Withdrawn))); - - let withdrawn = snapshot(®ister, &tables); - assert!(withdrawn.withdraws(entity(1))); - assert_eq!(withdrawn.staged.get(&entity(1)).copied(), None); - - // The unarchive re-stages through the held verdict, with no second lookup owed. - assert!(register.apply(event(1, 3, live(11)))); - assert_eq!(register.unclassified(&tables).collect::>(), []); - assert_eq!( - snapshot(®ister, &tables).staged.get(&entity(1)).copied(), - Some(edition(11)) - ); -} - -#[test] -fn replay_order_invariance() { - let tables = Tables { - nodes: vec![(entity(1), NodeRowId::new(4))], - edges: vec![(entity(3), EdgeRowId::new(7))], - ..Tables::default() - }; - let events = [ - event(1, 1, live(10)), - event(1, 2, Standing::Withdrawn), - event(2, 1, live(20)), - event(2, 1, Standing::Withdrawn), - event(3, 2, Standing::Withdrawn), - event(3, 3, live(30)), - event(4, 1, live(40)), - event(4, 1, live(41)), - ]; - - let mut forward = register(); - for event in events { - forward.apply(event); - } - - let mut reversed = register(); - for event in events.into_iter().rev() { - reversed.apply(event); - } - - assert_eq!(snapshot(&forward, &tables), snapshot(&reversed, &tables)); -} - -#[test] -fn feed_events_resolve_standing_and_version() { - let id = store_id(1); - - assert_eq!( - DeltaEvent::from(&updated(1, 1, 10, false)), - event(1, 1, live(10)) - ); - assert_eq!( - DeltaEvent::from(&updated(1, 2, 10, true)), - event(1, 2, Standing::Withdrawn) - ); - - let ended = EntityEvent::Ended(EntityEnd { - entity: id, - ended_at: at(3), - }); - assert_eq!(DeltaEvent::from(&ended), event(1, 3, Standing::Withdrawn)); - - let deleted = EntityEvent::Deleted(EntityDeletion { - entity: id, - provenance: EntityDeletionProvenance { - deleted_by_id: type_system::principal::actor::ActorEntityUuid::new(Uuid::from_u128( - 0xCD, - )), - deleted_at_transaction_time: at(4), - deleted_at_decision_time: Timestamp::from_unix_timestamp(4), - }, - }); - assert_eq!(DeltaEvent::from(&deleted), event(1, 4, Standing::Withdrawn)); -} - -#[test] -fn poll_outcome_advances_past_ignored() { - let mut register = register(); - let mut outcome = PollOutcome::default(); - - outcome.fold(&mut register, &updated(1, 5, 10, false)); - assert!(outcome.changed()); - - // The register ignores an older event for the same identity, and the poll still counts it - // while the watermark keeps the newest time read. - outcome.fold(&mut register, &updated(1, 3, 11, false)); - assert_eq!(outcome.events(), 2); - assert_eq!(outcome.watermark(), Some(at(5))); -} - -#[test] -fn poll_outcome_redelivery_no_change() { - let mut register = register(); - - let mut first = PollOutcome::default(); - first.fold(&mut register, &updated(1, 5, 10, false)); - assert!(first.changed()); - - let mut redelivery = PollOutcome::default(); - redelivery.fold(&mut register, &updated(1, 5, 10, false)); - assert!(!redelivery.changed()); - assert_eq!(redelivery.events(), 1); - assert_eq!(redelivery.watermark(), Some(at(5))); -} - -#[test] -fn polling_defaults() { - // The defaults give each side of the ensure twelve staging cycles at the five-second cadence: - // a minute per side and about two minutes end to end, over the sixty-second safety lag. The - // backlog capacity is a channel bound outside this arithmetic and stays unpinned. - let polling = DeltaPolling::default(); - assert_eq!(polling.interval, core::time::Duration::from_secs(5)); - assert_eq!(polling.retry_polls, 12); - assert_eq!(polling.safety_lag, core::time::Duration::from_secs(60)); -} - -#[test] -fn cell_swaps_publications_whole() { - let cell = DeltaCell::default(); - assert!(cell.load().is_none()); - - let register = register(); - cell.publish(register.snapshot(&Tables::default(), DeltaRevision::FIRST, at(1))); - let first = cell.load(); - assert_eq!( - first.as_ref().map(|snapshot| snapshot.revision), - Some(DeltaRevision::FIRST) - ); - - cell.publish(register.snapshot(&Tables::default(), DeltaRevision::FIRST.next(), at(2))); - assert_eq!( - cell.load().as_ref().map(|snapshot| snapshot.revision), - Some(DeltaRevision::FIRST.next()) - ); - // The guard loaded before the swap keeps reading its own publication. - assert_eq!( - first.as_ref().map(|snapshot| snapshot.watermark), - Some(at(1)) - ); -} - -/// The staged map one publication would carry, from `(identity, edition)` pairs. -fn staged_map(pairs: &[(u128, u128)]) -> FastHashMap { - pairs - .iter() - .map(|&(entity_n, edition_n)| (entity(entity_n), edition(edition_n))) - .collect() -} - -#[test] -fn sync_stages_fresh_pending() { - let mut pipeline = StagingPipeline::new(2); - - pipeline.sync(&staged_map(&[(1, 10)])); - - assert_eq!(pipeline.pending(), vec![entity(1)]); - assert_eq!(pipeline.ready().count(), 0); -} - -#[test] -fn budget_read_obliges_ensure() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10)])); - - // The first miss waits, and the second - the budget-spending final read - obliges the - // ensure in its own cycle. - assert_eq!(pipeline.miss(entity(1)), MissAction::Wait); - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(10))); -} - -#[test] -fn unconfirmed_ensure_reobliged() { - let mut pipeline = StagingPipeline::new(1); - pipeline.sync(&staged_map(&[(1, 10)])); - - // The ensure start failed, so `ensured` never confirmed it and each miss retries it. - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(10))); - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(10))); - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(10))); -} - -#[test] -fn post_ensure_exhaustion() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10)])); - assert_eq!(pipeline.miss(entity(1)), MissAction::Wait); - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(10))); - pipeline.ensured(entity(1)); - - // The post-ensure side reads under the same budget: one waiting miss, then the parking one. - assert_eq!(pipeline.miss(entity(1)), MissAction::Wait); - assert_eq!(pipeline.miss(entity(1)), MissAction::Park); - - // An exhausted arrival stops reading, and the mirror keeps it parked while it stays staged. - assert!(pipeline.pending().is_empty()); - pipeline.sync(&staged_map(&[(1, 10)])); - assert!(pipeline.pending().is_empty()); -} - -#[test] -fn exhaustion_survives_edition_move() { - let mut pipeline = StagingPipeline::new(1); - pipeline.sync(&staged_map(&[(1, 10)])); - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(10))); - pipeline.ensured(entity(1)); - assert_eq!(pipeline.miss(entity(1)), MissAction::Park); - - // A parked arrival stays parked until reconciliation or refit: a later edition moves the - // recorded edition and nothing else. - pipeline.sync(&staged_map(&[(1, 11)])); - assert!(pipeline.pending().is_empty()); -} - -#[test] -fn withdrawal_drops_unarchive_restages() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10)])); - assert_eq!(pipeline.miss(entity(1)), MissAction::Wait); - - // The withdrawal removes the identity from the staged set, and its state drops with it. - pipeline.sync(&staged_map(&[])); - assert!(pipeline.pending().is_empty()); - - // The unarchive restages it with the full budget. A held budget of 1 would oblige the - // ensure here, and a fresh budget of 2 waits. - pipeline.sync(&staged_map(&[(1, 10)])); - assert_eq!(pipeline.miss(entity(1)), MissAction::Wait); -} - -#[test] -fn sync_moves_edition_in_place() { - let mut pipeline = StagingPipeline::new(1); - pipeline.sync(&staged_map(&[(1, 10)])); - - // The ensure names the newest feed edition, not the one the arrival staged under. - pipeline.sync(&staged_map(&[(1, 11)])); - assert_eq!(pipeline.miss(entity(1)), MissAction::Ensure(edition(11))); -} - -#[test] -fn hit_completes_either_phase() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10), (2, 20)])); - - // Identity 2 is post-ensure when its row returns, and identity 1 is still pre-ensure. - assert_eq!(pipeline.miss(entity(2)), MissAction::Wait); - assert_eq!(pipeline.miss(entity(2)), MissAction::Ensure(edition(20))); - pipeline.ensured(entity(2)); - - pipeline.complete(entity(1), BoxedVecN::zero()); - pipeline.complete(entity(2), BoxedVecN::zero()); - - assert!(pipeline.pending().is_empty()); - let mut uncaptured = pipeline.uncaptured(); - uncaptured.sort_unstable_by_key(|&(identity, _)| identity); - assert_eq!( - uncaptured, - vec![(entity(1), edition(10)), (entity(2), edition(20))] - ); - - pipeline.captured(entity(1), display("one")); - pipeline.captured(entity(2), display("two")); - let mut ready: Vec<_> = pipeline - .ready() - .map(|(identity, recorded, _, _)| (identity, recorded)) - .collect(); - ready.sort_unstable_by_key(|&(identity, _)| identity); - assert_eq!( - ready, - vec![(entity(1), edition(10)), (entity(2), edition(20))] - ); - - // A completed arrival survives the mirror without re-entering pending. - pipeline.sync(&staged_map(&[(1, 10), (2, 20)])); - assert!(pipeline.pending().is_empty()); - assert_eq!(pipeline.ready().count(), 2); -} - -#[test] -fn placement_waits_for_capture() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10)])); - pipeline.complete(entity(1), BoxedVecN::zero()); - - // The embedding is ready and the display is not, so nothing places yet. - assert_eq!(pipeline.ready().count(), 0); - assert_eq!(pipeline.uncaptured(), vec![(entity(1), edition(10))]); - - pipeline.captured(entity(1), display("labeled arrival")); - assert!(pipeline.uncaptured().is_empty()); - let ready: Vec<_> = pipeline - .ready() - .map(|(identity, recorded, _, payload)| (identity, recorded, payload.clone())) - .collect(); - assert_eq!( - ready, - vec![(entity(1), edition(10), display("labeled arrival"))] - ); - - // A second capture for the same identity changes nothing: the first one stands. - pipeline.captured(entity(1), display("a later read")); - let ready: Vec<_> = pipeline - .ready() - .map(|(_, _, _, payload)| payload.clone()) - .collect(); - assert_eq!(ready, vec![display("labeled arrival")]); -} - -#[test] -fn departed_capture_dropped() { - let mut pipeline = StagingPipeline::new(1); - pipeline.sync(&staged_map(&[])); - - // The display read raced a withdrawal, so the identity left the staged set first. - pipeline.captured(entity(1), display("gone")); - - assert_eq!(pipeline.ready().count(), 0); - assert!(pipeline.uncaptured().is_empty()); -} - -#[test] -fn departed_completion_dropped() { - let mut pipeline = StagingPipeline::new(1); - pipeline.sync(&staged_map(&[])); - - // The row raced a withdrawal, so the identity left the staged set before its read answered. - pipeline.complete(entity(1), BoxedVecN::zero()); - - assert_eq!(pipeline.ready().count(), 0); - assert!(pipeline.pending().is_empty()); -} - -/// A projected arrival at a distinct fixture coordinate. -fn projected(edition_n: u128, x: f32, y: f32) -> ProjectedArrival { - let DisplayParts { - label, - icon, - representative, - } = display("arrival"); - ProjectedArrival { - edition: edition(edition_n), - position: Vec2::new(x, y), - label, - icon, - representative, - } -} - -#[test] -fn captured_label_priced() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - let bare = register.resident_estimate(); - - let label = "a label the estimate carries"; - let arrival = ProjectedArrival { - edition: edition(10), - position: Vec2::new(0.25, -0.5), - label: OwnedLabel::from(label), - icon: OwnedIcon::from("an icon"), - representative: type_uuid(), - }; - assert_eq!(place(&mut register, 1, &arrival), Disposition::Resolving); - - // The estimate grows by at least the label's text, whatever the map allocation adds. - assert!(register.resident_estimate() >= bare + label.len()); -} - -#[test] -fn placed_arrival_leaves_staged() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - assert_eq!( - place(&mut register, 1, &projected(10, 0.25, -0.5)), - Disposition::Resolving - ); - - let snapshot = snapshot(®ister, &Tables::default()); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); - assert_eq!( - snapshot.node(entity(1)), - Some(&DeltaNode { - edition: edition(10), - position: Vec2::new(0.25, -0.5), - id: slot(BASE), - legend: legend("arrival"), - }) - ); -} - -#[test] -fn first_placement_stands() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - assert_eq!( - place(&mut register, 1, &projected(10, 0.25, -0.5)), - Disposition::Resolving - ); - - // A re-delivered placement changes nothing, coordinate and slot included. - assert_eq!( - place(&mut register, 1, &projected(11, 0.75, 0.75)), - Disposition::AlreadyHeld - ); - - // A later edition moves the published edition and never the coordinate. - register.apply(event(1, 2, live(11))); - let snapshot = snapshot(®ister, &Tables::default()); - assert_eq!( - snapshot.node(entity(1)), - Some(&DeltaNode { - edition: edition(11), - position: Vec2::new(0.25, -0.5), - id: slot(BASE), - legend: legend("arrival"), - }) - ); -} - -#[test] -fn post_withdrawal_placement_unserved() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - register.apply(event(1, 2, Standing::Withdrawn)); - - // The completion raced the withdrawal. The register still records the placement while the - // standing keeps the row unserved, so it moves no publication input. - assert_eq!( - place(&mut register, 1, &projected(10, 0.25, -0.5)), - Disposition::Dormant - ); - - let withdrawn = snapshot(®ister, &Tables::default()); - assert!(withdrawn.withdraws(entity(1))); - assert_eq!(withdrawn.staged.get(&entity(1)).copied(), None); - assert_eq!(withdrawn.node(entity(1)), None); - - // The unarchive republishes the recorded coordinate on its former slot without re-entering - // the staged set. - register.apply(event(1, 3, live(11))); - let restored = snapshot(®ister, &Tables::default()); - assert_eq!(restored.staged.get(&entity(1)).copied(), None); - assert_eq!( - restored.node(entity(1)), - Some(&DeltaNode { - edition: edition(11), - position: Vec2::new(0.25, -0.5), - id: slot(BASE), - legend: legend("arrival"), - }) - ); -} - -#[test] -fn unclassified_placement_unpublished() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - - // An arrival without a verdict serves nothing, so the placement waits for the - // classification it cannot precede in any live pipeline, and publication stays fail-closed - // if one ever does. - assert_eq!( - place(&mut register, 1, &projected(10, 0.25, -0.5)), - Disposition::Resolving - ); - - let snapshot = snapshot(®ister, &Tables::default()); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), None); - assert_eq!(snapshot.node(entity(1)), None); -} - -/// Places through the fixture universe, panicking where the fixture cannot exhaust it. -fn place(register: &mut DeltaRegister, entity_n: u128, arrival: &ProjectedArrival) -> Disposition { - register - .place(entity(entity_n), arrival, &Tables::default()) - .expect("the fixture universe has room") -} - -#[test] -fn slots_assign_monotonically_in_placement_order() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - register.apply(event(2, 1, live(20))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - register - .classify(entity(2), Classification::Node) - .expect("a node verdict allocates no edge row"); - - // Placement order assigns the slots, and the base bound is the first one taken. - assert_eq!( - place(&mut register, 2, &projected(20, 0.5, 0.5)), - Disposition::Resolving - ); - assert_eq!( - place(&mut register, 1, &projected(10, 0.25, -0.5)), - Disposition::Resolving - ); - - let snapshot = snapshot(®ister, &Tables::default()); - assert_eq!( - snapshot.node(entity(2)).map(|arrival| arrival.id), - Some(slot(BASE)) - ); - assert_eq!( - snapshot.node(entity(1)).map(|arrival| arrival.id), - Some(slot(BASE + 1)) - ); - assert_eq!(snapshot.universe(), Universe::new(slot(BASE + 2))); -} - -#[test] -fn withdrawn_slot_never_reused() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - assert_eq!( - place(&mut register, 1, &projected(10, 0.25, -0.5)), - Disposition::Resolving - ); - - // The withdrawal leaves the slot allocated, so the next placement takes the one after it - // and the universe still counts both. - register.apply(event(1, 2, Standing::Withdrawn)); - register.apply(event(2, 1, live(20))); - register - .classify(entity(2), Classification::Node) - .expect("a node verdict allocates no edge row"); - assert_eq!( - place(&mut register, 2, &projected(20, 0.5, 0.5)), - Disposition::Resolving - ); - - let snapshot = snapshot(®ister, &Tables::default()); - assert_eq!( - snapshot.node(entity(2)).map(|arrival| arrival.id), - Some(slot(BASE + 1)) - ); - assert_eq!(snapshot.universe(), Universe::new(slot(BASE + 2))); -} - -#[test] -fn universe_base_bound_first() { - let mut register = register(); - register.apply(event(1, 1, live(10))); - - assert_eq!( - snapshot(®ister, &Tables::default()).universe(), - Universe::new(slot(BASE)) - ); -} - -#[test] -fn exhausted_universe_refuses_placement() { - let bound = NodeRowId::new(u64::from(u32::MAX) + 1); - let mut register = DeltaRegister::new( - Universe::new(bound), - Universe::new(EdgeRowId::new(EDGE_BASE)), - Universe::new(OntologyRowId::new(ONTOLOGY_BASE)), - ); - register.apply(event(1, 1, live(10))); - register - .classify(entity(1), Classification::Node) - .expect("a node verdict allocates no edge row"); - - // One more node row would take a value the wire codec cannot encode, so the refusal - // repeats and the bound never moves. - assert_eq!( - register.place(entity(1), &projected(10, 0.25, -0.5), &Tables::default()), - Err(UniverseExhausted) - ); - assert_eq!( - register.place(entity(1), &projected(10, 0.25, -0.5), &Tables::default()), - Err(UniverseExhausted) - ); - - let snapshot = snapshot(®ister, &Tables::default()); - assert_eq!(snapshot.staged.get(&entity(1)).copied(), Some(edition(10))); - assert_eq!(snapshot.node(entity(1)), None); - assert_eq!(snapshot.universe(), Universe::new(bound)); -} - -#[test] -fn handed_off_entry_rests_until_retired() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10)])); - pipeline.complete(entity(1), BoxedVecN::zero()); - pipeline.handed_off(entity(1)); - - // After the hand-off nothing reads and nothing re-places. - assert!(pipeline.pending().is_empty()); - assert_eq!(pipeline.ready().count(), 0); - - // The mirror keeps it resting while the publication still stages the identity. - pipeline.sync(&staged_map(&[(1, 10)])); - assert!(pipeline.pending().is_empty()); - assert_eq!(pipeline.ready().count(), 0); - - // The publication that placed it retires the entry, and a later unarchive restages fresh. - pipeline.sync(&staged_map(&[])); - pipeline.sync(&staged_map(&[(1, 10)])); - assert_eq!(pipeline.pending(), vec![entity(1)]); -} - -#[test] -fn out_of_frame_entry_stays_parked() { - let mut pipeline = StagingPipeline::new(2); - pipeline.sync(&staged_map(&[(1, 10)])); - pipeline.complete(entity(1), BoxedVecN::zero()); - pipeline.out_of_frame(entity(1)); - - // An out-of-frame arrival neither reads nor re-projects, because only a refit moves the - // fitted frame. - assert!(pipeline.pending().is_empty()); - assert_eq!(pipeline.ready().count(), 0); - assert_eq!(pipeline.miss(entity(1)), MissAction::Wait); - - pipeline.sync(&staged_map(&[(1, 10)])); - assert!(pipeline.pending().is_empty()); - assert_eq!(pipeline.ready().count(), 0); -} diff --git a/libs/@local/graph/atlas/src/serve/density/mod.rs b/libs/@local/graph/atlas/src/serve/density/mod.rs deleted file mode 100644 index 4eb13f52c11..00000000000 --- a/libs/@local/graph/atlas/src/serve/density/mod.rs +++ /dev/null @@ -1,436 +0,0 @@ -//! The scope-local delivery cut. -//! -//! A public density band over one authorized view's occupancy. -//! -//! A tile's delivery cut is `d(z) = z + m + k`, where `m` is the generation's span exponent and `k` -//! the offset this module resolves. The recorded schedule fixes `m`. `k` is the one degree of -//! freedom a restricted view has, and it decides how deep that view's tiles reach. -//! -//! `k` reads the authorized view and public constants, and nothing else. An offset derived from a -//! hidden row (a corpus-wide count, an unmasked bucket length, or a per-tile occupancy statistic) -//! would put the mask's own contents back on the wire as delivery depth. A client cannot see the -//! hidden rows, but it can see how deep the server went to accommodate them. Delivery depth is -//! therefore a public decision over the authorized view `V` and never an inheritance from the -//! corpus. -//! -//! The rule is a public inclusive band `[L, U]` over `C(m + k, V)`, the number of distinct -//! depth-`(m + k)` Morton cells `V` occupies. [`DensityPolicy::resolve`] chooses the offset whose -//! count lies nearest the band, coarsest on a tie. The band is a target rather than a promise: -//! occupied-cell counts step by whole subdivisions, so a coarse step can jump the band and the -//! nearest result may sit below or above it. Counts across scopes need not agree. -//! -//! The schedule determines the candidate offsets rather than any configuration. `k` runs over -//! `0..=ceiling`, where the ceiling is the room the recorded schedule leaves inside the 32 -//! subdivisions a 64-bit Morton key resolves, `32 - (m + z_max)`. The cut applies at every zoom, so -//! it is the deepest bucket `d(z_max)` that spends the key width, and `0` - the generation's own -//! schedule - is always available. The ceiling binds on clustering rather than on scale: a small, -//! densely-clustered view never reaches the band, so `resolve` runs to its saturation depth, which -//! sits as deep as the fit separates points - past the ceiling at default settings. A filtered -//! scope is exactly that shape. -//! -//! Occupancy travels as [`ViewOccupancy`], a typed aggregate of `V`, rather than as a precomputed -//! number, so a review can vary what a resolution reads. An opaque scalar would hide exactly the -//! provenance a channel argument turns on. -//! -//! # The aggregate -//! -//! `C(·, V)` at every depth, `Q(V) = C(32, V)` - the distinct complete keys - and `d_sat(V) = min { -//! d | C(d, V) = Q(V) }` all fall out of one pass over the view's sorted keys. -//! -//! Keys share a depth-`d` cell exactly when their leading `2d` bits agree, so distinct keys `a ≠ b` -//! first separate at depth `⌊lz(a ⊕ b) / 2⌋ + 1`, where `lz` counts leading zero bits. Sortedness -//! makes adjacent pairs sufficient, because when every adjacent distinct pair separates at or above -//! depth `d`, so does every pair. One pass therefore histograms the adjacent separation depths, and -//! `C(d, V) = 1 + #{ separations ≤ d }` for a non-empty view - the histogram's running sum, which -//! is what the aggregate stores. -//! -//! The stored form is the count profile alone. A row count is not an input any policy reads, so the -//! aggregate does not carry one. Views whose profiles agree are one value, and a resolution returns -//! the same offset for both. -//! -//! The module is crate-internal. Its examples carry `ignore` and spell each call as an in-crate -//! caller writes it, and the module's tests assert every property the examples show. -#![expect( - clippy::empty_enums, - reason = "zerocopy's FromBytes derive expands to an empty enum for its validation machinery" -)] - -#[cfg(test)] -mod tests; - -use core::{error::Error, fmt, num::NonZero}; - -use crate::{ - math::Log2, - morton::{Depth, MortonKey}, -}; - -/// The inclusive occupied-cell band a scope's delivery aims for. -/// -/// Both bounds are public configured constants, positive and ordered `lower ≤ upper`. A count -/// inside the band lies at distance zero, and outside it the distance is the shortfall or the -/// excess. -/// -/// Unconfigured, the band runs 2,000 through 4,000 occupied cells. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct DensityBand { - lower: NonZero, - upper: NonZero, -} - -impl DensityBand { - /// Validates a configured band. - /// - /// Returns [`None`] when `upper` lies below `lower`, a band that admits no count. - /// - /// # Examples - /// - /// ```ignore - /// use core::num::NonZero; - /// - /// let band = DensityBand::new( - /// NonZero::new(2_000).expect("2,000 is positive"), - /// NonZero::new(4_000).expect("4,000 is positive"), - /// ) - /// .expect("2,000 ≤ 4,000"); - /// - /// assert_eq!(band.distance(3_000), 0); - /// assert_eq!(band.distance(1_500), 500); - /// assert_eq!(band.distance(4_500), 500); - /// ``` - #[must_use] - #[cfg(test)] // The density and manifest tests configure bands directly. - pub(crate) const fn new(lower: NonZero, upper: NonZero) -> Option { - if upper.get() < lower.get() { - return None; - } - - Some(Self { lower, upper }) - } - - /// Returns `count`'s distance to the band, zero inside it. - #[must_use] - pub(crate) const fn distance(self, count: u64) -> u64 { - if count < self.lower.get() { - return self.lower.get() - count; - } - - if count > self.upper.get() { - return count - self.upper.get(); - } - - 0 - } -} - -const impl Default for DensityBand { - fn default() -> Self { - Self { - lower: NonZero::new(2_000).expect("2,000 is positive"), - upper: NonZero::new(4_000).expect("4,000 is positive"), - } - } -} - -/// One delivery-cut offset, the depth every zoom's cut gains for one scope. -/// -/// A density policy resolves the offset at a session's bootstrap and a sealed authority token -/// carries it, so a session serves at one delivery depth for as long as its client holds a token. A -/// view re-bound to another filter keeps that depth unless the new view resolves coarser, which -/// clamps it down - see [`DensityPolicy::rebind`]. Production construction is [`Self::ZERO`], a -/// policy's resolution or rebind, or the authenticated read of a sealed token, so the public-band -/// rule fixes every served offset and a client cannot forge one past the seal. -#[derive( - Debug, - Copy, - Clone, - PartialEq, - Eq, - PartialOrd, - Ord, - zerocopy::IntoBytes, - zerocopy::FromBytes, - zerocopy::Immutable, - zerocopy::Unaligned, - zerocopy::KnownLayout, -)] -#[repr(transparent)] -pub(crate) struct CutOffset(u8); - -impl CutOffset { - /// The generation's own schedule. - /// - /// Every zoom keeps its recorded cut. Every schedule leaves room for this offset, so it is the - /// resolution of a view with no occupancy to aim with. - pub(crate) const ZERO: Self = Self(0); - - /// Returns the offset. - #[must_use] - pub(crate) const fn get(self) -> u8 { - self.0 - } - - /// Wraps a raw offset, for fixtures alone. - #[cfg(test)] - pub(crate) const fn new(offset: u8) -> Self { - Self(offset) - } -} - -/// A generation whose recorded schedule leaves no delivery-cut offset to resolve. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum DensityPolicyError { - /// The recorded schedule already exceeds the key width, so no scope cascade deepens it. - Schedule { - /// The schedule's span exponent. - span: u8, - /// The schedule's deepest served zoom. - max_tile_depth: u8, - }, - /// The generation's root tile is its deepest. - /// - /// A terminal root is the catch-all bucket, and deepening a catch-all delivers no proportional - /// view. - TerminalRoot, -} - -impl fmt::Display for DensityPolicyError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Schedule { - span, - max_tile_depth, - } => write!( - fmt, - "the schedule's deepest bucket {max_tile_depth} + {span} already exceeds the 32 \ - subdivisions a Morton key resolves, so no density policy applies to it" - ), - Self::TerminalRoot => fmt.write_str( - "a generation whose deepest zoom is its root serves one catch-all tile, which no \ - density policy deepens", - ), - } - } -} - -impl Error for DensityPolicyError {} - -/// The public rule resolving one scope's delivery cut. -/// -/// A policy fixes the band, the generation's span exponent, and the offset ceiling the schedule -/// leaves. Those are everything a resolution reads besides the view's own occupancy. Policies -/// differing in any of them are different public policies, and a resolved cut is comparable only -/// within one of them. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct DensityPolicy { - band: DensityBand, - span: Log2, - /// The deepest offset the schedule leaves room for: `32 - (max_tile_depth + span)`. - ceiling: u8, -} - -impl DensityPolicy { - /// Configures a policy for one generation's schedule. - /// - /// The recorded schedule determines the candidate offsets rather than any configuration: - /// `0..=ceiling`, where the ceiling is what it leaves inside the 32 subdivisions a 64-bit - /// Morton key resolves. Every candidate therefore keeps the scope cascade's deepest bucket - - /// the sum of `max_tile_depth`, `span` and the offset - within the key width by construction, - /// and the two failures below are the schedules where no offset at all exists. - /// - /// # Errors - /// - /// Returns [`DensityPolicyError::TerminalRoot`] when `max_tile_depth` is zero, and - /// [`DensityPolicyError::Schedule`] when `max_tile_depth + span` already exceeds the key width. - /// - /// # Examples - /// - /// ```ignore - /// use crate::math::Log2; - /// - /// let span = Log2::new(6).expect("6 lies below the shift width"); - /// let policy = DensityPolicy::new(DensityBand::default(), span, 18)?; - /// ``` - pub(crate) fn new( - band: DensityBand, - span: Log2, - max_tile_depth: u8, - ) -> Result { - if max_tile_depth == 0 { - return Err(DensityPolicyError::TerminalRoot); - } - - let Some(ceiling) = max_tile_depth - .checked_add(span.get()) - .and_then(|deepest| Depth::MAX.get().checked_sub(deepest)) - else { - return Err(DensityPolicyError::Schedule { - span: span.get(), - max_tile_depth, - }); - }; - - Ok(Self { - band, - span, - ceiling, - }) - } - - /// Resolves the delivery-cut offset of one authorized view. - /// - /// The offset in `A(V) = { k | k ≤ min(k_sat(V), ceiling) }` minimizing `(dist(C(m + k, V), [L, - /// U]), k)`: nearest the band, coarsest on a tie. `k_sat(V) = max(0, d_sat(V) - m)` caps the - /// search where deeper cuts stop separating rows - past saturation every occupied cell holds - /// one key, so a deeper cut buys no occupancy and the tie-break keeps the coarser cut. The - /// ceiling caps it where the key width runs out, and it is the binding cap exactly when a view - /// clusters densely enough to saturate below it. - /// - /// Over a contiguous candidate set the saturation cap no longer decides a result on its own. - /// Counts are constant past `d_sat`, so a deeper offset ties and the tie-break already holds - /// the coarser cut. It stays because it makes "no resolution cuts deeper than saturation" true - /// by construction rather than by that argument. - /// - /// An empty view resolves to [`CutOffset::ZERO`] through this same argmin rather than through a - /// case of its own. It occupies no cell at any depth, so every candidate sits the same - /// shortfall from the band and the tie-break keeps the coarsest. No hidden or corpus quantity - /// stands in for the occupancy it lacks. - /// - /// The result may lie outside the band. Counts step by whole subdivisions, so a coarse step can - /// jump the band and a small, co-located, or saturation-capped view can stay below it. - #[must_use] - pub(crate) fn resolve(self, occupancy: &ViewOccupancy) -> CutOffset { - let saturation = occupancy - .saturation_depth() - .get() - .saturating_sub(self.span.get()); - let limit = saturation.min(self.ceiling); - - let mut resolved = CutOffset::ZERO; - let mut distance = u64::MAX; - for offset in 0..=limit { - // The ceiling bounded the search by the key width, so the fallback is unreachable; it - // keeps the resolution total rather than fallible. - let cut = Depth::new(self.span.get() + offset).unwrap_or(Depth::MAX); - // Strict improvement over an ascending walk is the ordered pair's tie-break: an equal - // distance keeps the coarser offset already held. - let candidate = self.band.distance(occupancy.occupied_cells(cut)); - if candidate < distance { - distance = candidate; - resolved = CutOffset(offset); - } - } - - resolved - } - - /// Rebinds a session's offset to a new view, never deepening it. - /// - /// A session serves at the one delivery depth its bootstrap resolved. Re-binding its view to a - /// different filter keeps that depth instead of re-optimizing it. The detail a tile carries at - /// a fixed zoom therefore does not move under a caller that only changed what it selects. - /// - /// The exception is one-way. A new view that resolves coarser than the session's offset wins, - /// so a deep cut resolved over a sparse view never reaches a dense one whose band asks for less - /// depth. - /// - /// Both directions therefore read off the minimum. A new view resolving deeper keeps the - /// carried and coarser cut. A new view resolving coarser clamps the session down to it. An - /// empty new view resolves [`CutOffset::ZERO`] and so clamps to zero. - /// - /// The result stays admissible for the new view. The candidate offsets are contiguous from - /// zero, so a value at or below a resolvable one is itself resolvable, and the key width stays - /// bounded because the same generation's ceiling bounded the carried value. - #[must_use] - pub(crate) fn rebind(self, carried: CutOffset, occupancy: &ViewOccupancy) -> CutOffset { - carried.min(self.resolve(occupancy)) - } -} - -/// The occupancy aggregate of one authorized view: every count a policy may read. -/// -/// The view's occupied-cell count at every depth, from which the distinct-key count and the -/// saturation depth read off. The module doc carries the derivation and the reason the profile is -/// the whole stored form. -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct ViewOccupancy { - /// Entry `d` holds `C(d, V)`: zero throughout for an empty view, and one at [`Depth::MIN`] - - /// the whole domain - for every other. - occupied: [u64; Self::DEPTHS], -} - -impl ViewOccupancy { - /// One entry per depth, the whole domain included. - const DEPTHS: usize = Depth::MAX.get() as usize + 1; - - /// Aggregates the Morton keys of one authorized view. - /// - /// Sorts `keys` in place: the aggregate is a function of their multiset, and sorting is what - /// lets one pass reach every depth's count. - #[must_use] - pub(crate) fn of(keys: &mut [MortonKey]) -> Self { - keys.sort_unstable(); - - let mut occupied = [0_u64; Self::DEPTHS]; - if keys.is_empty() { - return Self { occupied }; - } - - let mut separations = [0_u64; Self::DEPTHS]; - for &[earlier, later] in keys.array_windows::<2>() { - if earlier == later { - continue; - } - - // The keys part one level below their deepest shared grid. - let depth = earlier.shared_depth(later).get() + 1; - separations[usize::from(depth)] += 1; - } - - // Every key shares the whole domain, so the profile starts at one cell and gains each - // depth's separations. - let mut cells = 1; - for (count, separations) in occupied.iter_mut().zip(separations) { - cells += separations; - *count = cells; - } - - Self { occupied } - } - - /// Returns whether the view occupies nothing. - #[must_use] - #[cfg(test)] // The density, serve, and manifest tests assert emptiness directly. - pub(crate) const fn is_empty(&self) -> bool { - self.occupied[Depth::MIN.get() as usize] == 0 - } - - /// Counts the distinct depth-`depth` cells the view occupies: `C(depth, V)`. - /// - /// Zero for an empty view; one for every other view at [`Depth::MIN`], the whole domain. - #[must_use] - pub(crate) const fn occupied_cells(&self, depth: Depth) -> u64 { - self.occupied[depth.get() as usize] - } - - /// Counts the distinct complete keys the view carries: `Q(V) = C(32, V)`. - #[must_use] - pub(crate) const fn distinct_keys(&self) -> u64 { - self.occupied_cells(Depth::MAX) - } - - /// Returns the coarsest depth at which every distinct key occupies its own cell: `d_sat(V)`. - /// - /// [`Depth::MIN`] when the view carries at most one distinct key, since the whole domain - /// already separates them - an empty view included. - #[must_use] - pub(crate) fn saturation_depth(&self) -> Depth { - let saturated = self.distinct_keys(); - - // The profile's deepest entry is the count itself, so the search is total; the fallback is - // that same depth. - Depth::all() - .find(|&depth| self.occupied_cells(depth) == saturated) - .unwrap_or(Depth::MAX) - } -} diff --git a/libs/@local/graph/atlas/src/serve/density/tests.rs b/libs/@local/graph/atlas/src/serve/density/tests.rs deleted file mode 100644 index 1002d99e5ba..00000000000 --- a/libs/@local/graph/atlas/src/serve/density/tests.rs +++ /dev/null @@ -1,467 +0,0 @@ -use core::num::NonZero; -use std::collections::HashSet; - -use proptest::{prop_assert, prop_assert_eq, property_test}; - -use super::{CutOffset, DensityBand, DensityPolicy, DensityPolicyError, ViewOccupancy}; -use crate::{ - math::Log2, - morton::{Depth, MortonCell, MortonKey}, -}; - -/// Returns the policy's band. -const fn band_of(policy: DensityPolicy) -> DensityBand { - policy.band -} - -/// The fixtures' span exponent. -/// -/// A view's cut at offset `k` is depth `1 + k`. -const SPAN: u8 = 1; - -/// The fixtures' deepest served zoom. -const MAX_TILE_DEPTH: u8 = 4; - -/// The offset ceiling the fixtures' schedule leaves: `32 - (4 + 1)`, hand-derived. -const CEILING: u8 = 27; - -fn band(lower: u64, upper: u64) -> DensityBand { - DensityBand::new( - NonZero::new(lower).expect("the fixture band's bounds are positive"), - NonZero::new(upper).expect("the fixture band's bounds are positive"), - ) - .expect("the fixture band is ordered") -} - -fn span(value: u8) -> Log2 { - Log2::new(value).expect("the fixture span lies below the shift width") -} - -fn depth(value: u8) -> Depth { - Depth::new(value).expect("the fixture depth lies within the key width") -} - -fn policy(band: DensityBand) -> DensityPolicy { - DensityPolicy::new(band, span(SPAN), MAX_TILE_DEPTH).expect("the fixture policy is admissible") -} - -/// A view whose occupancy keeps climbing to depth 24. -/// -/// Key `i` carries a single set bit at position `64 - 2i`, so it separates from every coarser key -/// at depth `i` exactly: sorted, the adjacent separations land one per depth over `1..=24`, giving -/// `C(d, V) = 1 + d` up to `d = 24` and `Q(V) = 25`. It is the shape the ceiling exists for - a -/// view that keeps paying for depth long past the room a deep schedule leaves. -fn deep_view() -> ViewOccupancy { - let mut keys = vec![MortonKey::from_bits(0)]; - keys.extend((1..=24_u32).map(|index| MortonKey::from_bits(1_u64 << (64 - 2 * index)))); - - occupancy(&keys) -} - -/// The corner key of one cell of the depth's grid. -fn key(depth: u8, x: u32, y: u32) -> MortonKey { - MortonCell::new( - Depth::new(depth).expect("the fixture depth lies within the key width"), - x, - y, - ) - .expect("the fixture cell lies on the depth's grid") - .min_key() -} - -fn occupancy(keys: &[MortonKey]) -> ViewOccupancy { - let mut keys = keys.to_vec(); - ViewOccupancy::of(&mut keys) -} - -/// A view whose occupancy plateaus once and then splits. -/// -/// The view occupies the depth-3 cells `(0,0)`, `(1,0)`, `(4,0)`, and `(5,0)`. The first two share -/// one depth-1 half and the last two share the other. Each pair shares one depth-2 cell and -/// separates at depth 3, so the counts run `C(1) = 2`, `C(2) = 2`, `C(3) = 4`, saturating at depth -/// 3. -fn plateau_view() -> ViewOccupancy { - occupancy(&[key(3, 0, 0), key(3, 1, 0), key(3, 4, 0), key(3, 5, 0)]) -} - -/// An inverted band refuses construction rather than admitting no count. -/// -/// The band's own ordering is the invariant every distance reading rests on: an unchecked `[4_000, -/// 2_000]` would report a positive distance for every count, including counts a correct -/// configuration calls perfect. -#[test] -fn inverted_band_refuses_construction() { - let lower = NonZero::new(2_000).expect("2,000 is positive"); - let upper = NonZero::new(4_000).expect("4,000 is positive"); - - assert!(DensityBand::new(lower, upper).is_some()); - assert_eq!(DensityBand::new(upper, lower), None); - assert!( - DensityBand::new(lower, lower).is_some(), - "a single-count band is ordered" - ); -} - -/// The band's bounds count as inside it. -/// -/// A half-open reading of the band would make the lower bound a shortfall of zero-plus-one and put -/// the selector one subdivision deeper than the policy asks for. -#[test] -fn band_includes_its_bounds() { - let band = band(2_000, 4_000); - - assert_eq!(band.distance(2_000), 0); - assert_eq!(band.distance(4_000), 0); - assert_eq!(band.distance(1_999), 1); - assert_eq!(band.distance(4_001), 1); -} - -/// Distance measures the shortfall below the band and the excess above it. -#[test] -fn band_distance_grows_with_the_gap_to_its_bounds() { - let band = band(2_000, 4_000); - - assert_eq!(band.distance(3_000), 0); - assert_eq!(band.distance(1_500), 500); - assert_eq!(band.distance(4_500), 500); -} - -/// The key width caps a resolution that would otherwise keep going deeper. -/// -/// A span-6 schedule serving 18 zooms leaves room for 8. The deepest bucket `18 + 6 + 8` is exactly -/// the 32 subdivisions a 64-bit key resolves. One more would wrap a cut the walk then reads. The -/// deep view saturates at depth 24, so `k_sat = 18` and an unreachable band pulls the argmin toward -/// every deeper cut the search offers it. The ceiling alone stops it at 8. -#[test] -fn key_width_caps_a_resolution_that_would_otherwise_deepen() { - let policy = DensityPolicy::new(band(100, 200), span(6), 18) - .expect("a span-6 schedule serving 18 zooms is admissible"); - let view = deep_view(); - - assert_eq!(view.saturation_depth(), depth(24), "k_sat is 24 - 6 = 18"); - assert_eq!( - view.occupied_cells(depth(14)), - 15, - "the cut the ceiling gives" - ); - assert_eq!( - view.occupied_cells(depth(24)), - 25, - "the cut k_sat would give" - ); - - assert_eq!( - policy.resolve(&view).get(), - 8, - "the count climbs the whole way, so only the key width stops the search" - ); -} - -/// Configuration refuses a terminal root and a schedule already past the key width. -/// -/// Both are schedules no offset repairs, so they fail when a caller configures the policy rather -/// than when a view resolves. -#[test] -fn configuration_refuses_a_schedule_no_offset_deepens() { - assert_eq!( - DensityPolicy::new(band(2_000, 4_000), span(6), 0), - Err(DensityPolicyError::TerminalRoot) - ); - assert_eq!( - DensityPolicy::new(band(2_000, 4_000), span(6), 30), - Err(DensityPolicyError::Schedule { - span: 6, - max_tile_depth: 30 - }) - ); -} - -/// Occupancy counts cells, not rows. -/// -/// Co-located rows share one complete key, so a row-counting aggregate would report a view as -/// denser than its geometry and resolve a coarser cut than the band asks for. -#[test] -fn occupancy_counts_cells_not_rows() { - let anchor = key(3, 2, 1); - let view = occupancy(&[anchor, anchor, anchor]); - - assert!(!view.is_empty()); - assert_eq!(view.distinct_keys(), 1); - assert_eq!(view.occupied_cells(Depth::MAX), 1); - assert_eq!(view.saturation_depth(), Depth::MIN); -} - -/// An empty view occupies no cell at any depth. -/// -/// The aggregate's counts start at one for a non-empty view, so an empty view that shared that -/// floor would manufacture an occupied cell out of nothing. -#[test] -fn empty_view_occupies_no_cell() { - let view = occupancy(&[]); - - assert!(view.is_empty()); - assert_eq!(view.distinct_keys(), 0); - for depth in Depth::all() { - assert_eq!(view.occupied_cells(depth), 0, "depth {}", depth.get()); - } -} - -/// Occupancy saturates at the depth that first separates every distinct key. -/// -/// The saturation depth caps the search: reading it one depth too shallow would drop a candidate -/// cut that still splits cells, and one too deep would keep cuts that buy nothing. -#[test] -fn occupancy_saturates_at_the_separating_depth() { - let view = plateau_view(); - - assert_eq!(view.occupied_cells(depth(0)), 1); - assert_eq!(view.occupied_cells(depth(1)), 2); - assert_eq!(view.occupied_cells(depth(2)), 2); - assert_eq!(view.occupied_cells(depth(3)), 4); - assert_eq!(view.occupied_cells(depth(4)), 4); - assert_eq!(view.distinct_keys(), 4); - assert_eq!(view.saturation_depth(), depth(3)); -} - -/// An empty view resolves to the base offset. -/// -/// The same argmin produces this resolution as every other one. An empty view has nothing to aim -/// with, and falling back on a corpus quantity is the channel this policy exists to close. -#[test] -fn empty_view_resolves_to_the_base_offset() { - assert_eq!(policy(band(2, 4)).resolve(&occupancy(&[])), CutOffset::ZERO); -} - -/// A co-located view resolves to the base offset. -/// -/// Its saturation depth is the whole domain, so every deeper cut leaves the search space: no cut -/// separates keys that share one complete key. -#[test] -fn co_located_view_resolves_to_the_base_offset() { - let anchor = key(3, 2, 1); - - assert_eq!( - policy(band(2, 4)).resolve(&occupancy(&[anchor, anchor])), - CutOffset::ZERO - ); -} - -/// The coarsest in-band offset wins when the band admits more than one. -/// -/// Under the plateau view, offset 0 counts 2 and offset 2 counts 4; a band holding both must keep -/// the coarser cut, since a deeper one costs response bytes for no policy gain. -#[test] -fn coarsest_in_band_offset_wins() { - assert_eq!(policy(band(2, 4)).resolve(&plateau_view()).get(), 0); -} - -/// A plateau does not stop the search. -/// -/// The plateau view counts 2 at offsets 0 and 1 and 4 at offset 2. A search that stopped at the -/// first offset failing to improve would resolve 0 and miss the only in-band cut - the reason the -/// argmin runs over the whole candidate range rather than the least positive offset. -#[test] -fn plateau_does_not_stop_the_search() { - assert_eq!(policy(band(3, 4)).resolve(&plateau_view()).get(), 2); -} - -/// An equal distance keeps the coarser offset. -/// -/// Band `[3, 3]` puts offset 0 one below and offset 2 one above. The ordered pair's second -/// component decides it, and it decides for the coarser cut. -#[test] -fn equal_distance_keeps_the_coarser_offset() { - assert_eq!(policy(band(3, 3)).resolve(&plateau_view()).get(), 0); -} - -/// A view below the band takes the closest count it can reach. -/// -/// Every reachable cut of the plateau view stays under a `[10, 20]` band, so the nearest is the -/// deepest reachable - the case where the band is unreachable and the policy still resolves. -#[test] -fn view_below_the_band_takes_the_closest_reachable_count() { - assert_eq!(policy(band(10, 20)).resolve(&plateau_view()).get(), 2); -} - -/// A view already above the band keeps the base offset. -/// -/// Occupancy never falls with depth, so no deeper cut can come back toward the band: the coarsest -/// cut is the closest one, and the tie-break holds it. -#[test] -fn view_already_above_the_band_keeps_the_base_offset() { - assert_eq!(policy(band(1, 1)).resolve(&plateau_view()).get(), 0); -} - -/// No resolution cuts deeper than the view's saturation depth. -/// -/// The plateau view saturates at depth 3, so with span 1 offset 2 is the deepest cut that separates -/// anything, because a deeper one would deliver more buckets and more response for identical -/// occupancy. The -/// band `[10, 20]` is unreachable from above, so nothing but this property stops the argmin at 2 -/// rather than at [`CEILING`], which is 27 and therefore not what caps this resolution. -/// -/// What the test pins is the property rather than one mechanism, and the deletion controls say so. -/// Over a contiguous candidate range two guards cover the resolution, `resolve`'s saturation cap -/// and the tie-break, and removing either one alone leaves this green. Both together resolve 27 -/// here. Counts are constant past saturation, so no view and no band can separate the two guards - -/// the cap is unwitnessable through the output by construction, and it stays for the stated law -/// rather than for a behaviour a test could lose. The single-fault witness for the other cap is -/// [`the_ceiling_caps_a_resolution_at_the_key_width`]. -#[test] -fn resolution_never_runs_deeper_than_saturation() { - let view = plateau_view(); - - assert_eq!(view.saturation_depth(), depth(3)); - assert_eq!(view.occupied_cells(depth(5)), view.occupied_cells(depth(3))); - assert_eq!(policy(band(10, 20)).resolve(&view).get(), 2); -} - -/// The aggregate's counts agree with a direct prefix census at every depth. -/// -/// The histogram derivation replaces one distinct-prefix count per depth with a single pass over -/// sorted keys. The bug class is the derivation itself - an off-by-one in the separation depth -/// would shift whole columns of the count profile, and every count above feeds a cut the client -/// receives. -#[property_test] -fn occupied_cells_match_a_direct_prefix_census( - #[strategy = proptest::collection::vec(0_u64..1_u64 << 12, 0..24_usize)] bits: Vec, -) { - // A narrow key domain packs the sample into few high-depth cells, so separations land across - // the whole depth range instead of only at the deepest bins. - let keys: Vec = bits - .iter() - .map(|&bits| MortonKey::from_bits(bits << 40)) - .collect(); - let view = occupancy(&keys); - - let distinct: HashSet = keys.iter().map(|key| key.to_bits()).collect(); - prop_assert_eq!(view.distinct_keys(), distinct.len() as u64); - prop_assert_eq!(view.is_empty(), keys.is_empty()); - - let mut previous = 0; - for depth in Depth::all() { - let census: HashSet = keys.iter().map(|key| key.prefix(depth)).collect(); - let expected = if keys.is_empty() { - 0 - } else { - census.len() as u64 - }; - - prop_assert_eq!( - view.occupied_cells(depth), - expected, - "depth {}", - depth.get() - ); - prop_assert!( - view.occupied_cells(depth) >= previous, - "occupancy fell at depth {}", - depth.get() - ); - previous = view.occupied_cells(depth); - - // The saturation depth is the coarsest depth reaching the distinct-key count. - prop_assert_eq!( - depth >= view.saturation_depth(), - view.occupied_cells(depth) == view.distinct_keys(), - "depth {} against saturation {}", - depth.get(), - view.saturation_depth().get() - ); - } -} - -/// A resolution reads the band and the schedule, and nothing else about the view. -/// -/// Views with identical count profiles resolve identically whatever key layout produced them, which -/// is the property a hidden row must not be able to break: it can only reach the resolution through -/// the aggregate the policy reads. -#[property_test] -fn resolution_is_a_function_of_the_count_profile( - #[strategy = proptest::collection::vec(0_u64..1_u64 << 8, 1..16_usize)] bits: Vec, -) { - let policy = policy(band(3, 5)); - - let keys: Vec = bits - .iter() - .map(|&bits| MortonKey::from_bits(bits << 48)) - .collect(); - let mut reversed: Vec = keys.iter().rev().copied().collect(); - - let resolved = policy.resolve(&occupancy(&keys)); - prop_assert_eq!(resolved, policy.resolve(&ViewOccupancy::of(&mut reversed))); - - // The resolved offset lies in the candidate range, and it is the argmin the law states. - let cut = depth(SPAN + resolved.get()); - let view = occupancy(&keys); - let distance = band_of(policy).distance(view.occupied_cells(cut)); - prop_assert!(resolved.get() <= CEILING); - for offset in 0..=CEILING { - if offset > view.saturation_depth().get().saturating_sub(SPAN) { - continue; - } - - let candidate = band_of(policy).distance(view.occupied_cells(depth(SPAN + offset))); - prop_assert!( - distance < candidate || (distance == candidate && resolved.get() <= offset), - "offset {} beats the resolved {}", - offset, - resolved.get() - ); - } -} - -/// A re-bind keeps the session's coarser cut when the new view resolves deeper. -/// -/// The band is unreachable for the plateau view, whose deepest cut holds 4 cells, so its bootstrap -/// resolves the deepest offset it has: 2. The deep view reaches the band exactly at offset 18 -/// (`C(19, V) = 20`), so re-optimizing would deepen the session by sixteen subdivisions and change -/// the detail every tile carries at a fixed zoom. The re-bind keeps 2. -#[test] -fn rebind_keeps_the_carried_cut_when_the_new_view_resolves_deeper() { - let policy = policy(band(20, 20)); - let carried = policy.resolve(&plateau_view()); - assert_eq!( - carried, - CutOffset::new(2), - "the plateau view's own resolution" - ); - assert_eq!( - policy.resolve(&deep_view()), - CutOffset::new(18), - "the deep view's own resolution, which the re-bind must not adopt" - ); - - assert_eq!(policy.rebind(carried, &deep_view()), carried); -} - -/// A re-bind clamps the session down when the new view resolves coarser. -/// -/// In the reverse pairing, a session bootstrapped on the deep view holds offset 18, and re-binding -/// to the plateau view would carry that cut into a view whose band asks for 2. Carrying it would -/// deliver a depth-19 cut over four cells, so the coarser resolution wins. -#[test] -fn rebind_clamps_down_when_the_new_view_resolves_coarser() { - let policy = policy(band(20, 20)); - let carried = policy.resolve(&deep_view()); - - assert_eq!( - policy.rebind(carried, &plateau_view()), - CutOffset::new(2), - "the deep session kept its cut over a view the band serves shallower" - ); -} - -/// A re-bind to an empty view clamps to the base offset. -/// -/// An empty view resolves [`CutOffset::ZERO`], and the clamp takes it whatever the session held. -/// A view with no occupancy is never served at a depth an earlier view paid for. -#[test] -fn rebind_to_an_empty_view_clamps_to_the_base_offset() { - let policy = policy(band(20, 20)); - - assert_eq!( - policy.rebind(policy.resolve(&deep_view()), &occupancy(&[])), - CutOffset::ZERO - ); -} diff --git a/libs/@local/graph/atlas/src/serve/edges.rs b/libs/@local/graph/atlas/src/serve/edges.rs deleted file mode 100644 index d048950cdc0..00000000000 --- a/libs/@local/graph/atlas/src/serve/edges.rs +++ /dev/null @@ -1,442 +0,0 @@ -//! Edges delivery. -//! -//! The edges among the listed tiles' delivered rows, answered as `SALTILEE` envelope bytes in -//! ascending link-entity identity order. Fitted edges are the generation's own rows, gathered -//! by the structural walk. The entry cohort's published post-fit links join them wherever both -//! endpoints deliver, and each merges into the same identity order and competes in the same -//! rank union at the cap. - -use core::{error::Error, fmt}; - -use hashql_core::{ - collections::FastHashMap, - id::{IdVec, bit_vec::DenseBitSet}, -}; -use type_system::ontology::id::VersionedUrl; - -use super::{ - Atlas, - delta::PlacementCohort, - grid, - hydrate::{DetailError, EdgeLinkDetails, EdgeSlot, EdgesStore, TypeSlot}, - intern::{Table, TableIndex}, - neighbourhood::{DeliveredBounds, EdgeColumns, EdgeOrigin, EdgeSet}, - schedule::ViewRow, - view::{View, ViewError}, - walk::Walk, -}; -use crate::{ - dataset::auxiliary::{Label, Legend}, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - salt::wire::{ - edges::{EdgesResponse, EdgesTrailer}, - tile::TileCoordinate, - }, -}; - -/// An edges request the atlas rejects, by name. -/// -/// Every variant is a named, data-carrying rejection for the transport layer to map onto its error -/// vocabulary. -#[derive(Debug)] -pub(crate) enum EdgesError { - /// The request lists more tiles than the cap admits. - Tiles { - /// The listed tile count. - count: usize, - /// The cap the manifest publishes as `limits.edgesTiles`. - maximum: u32, - }, - /// A listed zoom exceeds the generation's deepest served tile. - Depth { - /// The requested zoom. - z: u8, - /// The generation's deepest served zoom. - maximum: u8, - }, - /// A listed coordinate lies outside its zoom's `2^z` grid. - Grid { - /// The requested zoom. - z: u8, - /// The requested x index. - x: u32, - /// The requested y index. - y: u32, - }, - /// The delivery view did not bind. - /// - /// A binding refusal converts into this variant through [`From`], so one error union carries a - /// route's binding and assembly rejections together. [`Atlas::edges`] takes the view already - /// bound, so its own rejections are all request-shaped. - View(ViewError), - /// The store half of the detail trailer failed. - /// - /// Only [`Atlas::edges`] answers this, because that path places the hydration order itself, - /// and only a request asking for the detail trailer places one. - Details(DetailError), -} - -impl fmt::Display for EdgesError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Tiles { count, maximum } => { - write!( - fmt, - "the request lists {count} tiles where the cap admits {maximum}" - ) - } - Self::Depth { z, maximum } => { - write!(fmt, "zoom {z} exceeds the deepest served tile {maximum}") - } - Self::Grid { z, x, y } => { - write!(fmt, "({x}, {y}) lies outside the 2^{z} tile grid") - } - Self::View(error) => error.fmt(fmt), - Self::Details(error) => error.fmt(fmt), - } - } -} - -impl Error for EdgesError {} - -impl From for EdgesError { - fn from(value: ViewError) -> Self { - Self::View(value) - } -} - -#[derive(Debug, Copy, Clone, PartialEq, Eq, Default, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase")] -pub(crate) enum EdgesDetail { - #[default] - Minimal, - Auxiliary, -} - -/// The POST body of one edges read. -#[derive(Debug, Clone, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub(crate) struct EdgesRequest { - /// The tiles whose delivered rows bound the edge set. - pub tiles: Vec, - /// Whether the response carries the detail trailer. - #[serde(default)] - pub detail: EdgesDetail, -} - -/// The edges endpoint's request and response limits. -/// -/// Transport configuration with documented defaults, never wire constants: the transport constructs -/// one value and the manifest publishes the same value, so enforcement and advertisement cannot -/// disagree. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct EdgesLimits { - /// Most tiles one request may list. - /// - /// The manifest publishes this value as `limits.edgesTiles`. - pub tiles: u32 = 256, - /// Most edges one response delivers. - /// - /// Beyond it the rank-ordered cap truncates and `HEAD` reports `complete: false`. - /// - /// One delivered edge spends 40 bytes of columns across a 4-byte source, a 4-byte target, and its 32-byte link entity id. The default cap of `0x4000` edges bounds one response's columns at 640 KiB. - pub edges: u32 = 0x4000, -} - -const impl Default for EdgesLimits { - fn default() -> Self { - Self { .. } - } -} - -/// One assembled edges response: everything [`Atlas::encode_edges`] needs. -/// -/// The document owns its columns, so it crosses thread boundaries between assembly, hydration, and -/// encoding - the envelope was designed for hydration-last, and the split mirrors it: assembly and -/// encoding are CPU-bound, hydration awaits the store between them. -#[derive(Debug)] -pub(crate) struct EdgesDocument { - complete: bool, - /// The delivered edges in column form, ascending link-entity identity bytes. - edges: EdgeColumns, -} - -/// The in-process display lookup one detail request binds. -/// -/// Cohort first, artifact second - the register's own precedence. The captured displays answer -/// every published link and every fitted identity whose revision has captured, and a fitted -/// link with no captured legend answers from the generation's baked legend. Either way a legend -/// names its representative type as an ontology row, so the store's whole share of the trailer is -/// resolving the distinct rows' uuids to their versioned URLs in one read. -#[derive(Debug, Copy, Clone)] -struct EdgeDisplays<'atlas> { - /// The opened generation, whose legends and identity tables answer the fitted arm. - atlas: &'atlas Atlas, - /// The entry's placement cohort, whose snapshot holds the captured legends. - cohort: PlacementCohort<'atlas>, -} - -impl<'atlas> EdgeDisplays<'atlas> { - /// Returns the label and representative type uuid of the delivered link `id`. - /// - /// A link's uuid reads [`None`] when neither the generation's table nor the cohort's - /// extension names its representative row, and the trailer serves no reference for it. - /// - /// # Panics - /// - /// This panics on a delta link the cohort holds no captured legend for. Publication - /// withholds a link until its legend captures, and the delivered set admits links from the - /// cohort's own snapshot, so every delivered delta link resolves. - fn display( - &self, - id: ArchivedEntityId, - origin: EdgeOrigin, - ) -> (&'atlas Label, Option) { - let legend = self.cohort.legend_of(id).unwrap_or_else(|| match origin { - EdgeOrigin::Fitted(edge) => self - .atlas - .edge_ids - .payload_of(edge) - .expect("open validated the identity rows against the adjacency's edges"), - EdgeOrigin::Delta => unreachable!( - "publication withholds a link until its legend captures, and the delivered set \ - admits links from the cohort's own snapshot" - ), - }); - - (legend.label(), self.representative_uuid(legend)) - } - - /// Returns the uuid of `legend`'s representative row, from the generation's table or the - /// cohort's extension. - fn representative_uuid(&self, legend: &Legend) -> Option { - let row = legend.representative_ontology(); - self.atlas - .ontology_ids - .id(row) - .or_else(|| self.cohort.ontology_id_of(row)) - } -} - -impl Atlas { - /// Answers one edges request over its bound delivery view. - /// - /// `SALTILEE` envelope bytes carrying the edges whose endpoints both lie in the listed tiles' - /// delivered rows, ready to send under `application/vnd.hash.saltile-v1`. - /// - /// A request asking for the detail trailer resolves every label and type uuid in process, - /// from the generation's own payloads and the cohort's captured displays, and `store` - /// answers the one hydration order such a request places: the distinct required type uuids' - /// versioned URLs. A minimal request drops the capability unused. - /// - /// # Errors - /// - /// As [`Atlas::assemble_edges`], plus [`EdgesError::Details`] when the store half of the - /// detail trailer fails. - pub(crate) fn edges( - &self, - request: &EdgesRequest, - limits: EdgesLimits, - view: View<'_>, - store: impl EdgesStore, - ) -> Result, EdgesError> { - let document = self.assemble_edges(request, limits, &view)?; - let details = match request.detail { - EdgesDetail::Minimal => None, - EdgesDetail::Auxiliary => { - let displays = EdgeDisplays { - atlas: self, - cohort: view.cohort(), - }; - - // Every delivered link's display resolves in process, so only the distinct - // representative type uuids reach the store, in first-occurrence order over - // the delivered slots. - let mut labels = IdVec::with_capacity(document.edges.count()); - let mut required: IdVec = IdVec::new(); - let mut slots: FastHashMap = - FastHashMap::default(); - let mut type_slots: IdVec> = - IdVec::with_capacity(document.edges.count()); - for (slot, &origin) in document.edges.origins().iter_enumerated() { - let (label, uuid) = displays.display(document.edges.ids()[slot], origin); - labels.push(label); - type_slots.push( - uuid.map(|uuid| *slots.entry(uuid).or_insert_with(|| required.push(uuid))), - ); - } - - let urls = store.hydrate(&required).map_err(EdgesError::Details)?; - - let mut representative_type_urls = IdVec::with_capacity(document.edges.count()); - for &slot in &type_slots { - representative_type_urls.push(slot.and_then(|slot| urls[slot].clone())); - } - - Some(EdgeLinkDetails::new(labels, representative_type_urls)) - } - }; - - Ok(self.encode_edges(&document, details.as_ref())) - } - - /// Assembles one edges request into its owned document. - /// - /// Every rejection happens here, so encoding cannot fail. - /// - /// Delivery order is ascending link-entity identity bytes, independent of the tiles listed and - /// of truncation, so identical requests yield identical bytes under one bound serving state - /// (generation, visibility, secret, limits) - and the order is client-verifiable from the - /// `EDGE_IDS` column alone, carrying no internal-order information. Beyond `limits.edges` the - /// rank-ordered cap keeps the edges whose worse endpoint ranks best - an edge is only as - /// prominent as its less-prominent endpoint - with ties broken by identity bytes, and `HEAD` - /// reports `complete: false`. - /// - /// The bounding set is the tile route's own delivery under `view`. An operator view unions the - /// corpus schedule's cumulative prefixes at `z + span`, and a scoped view its own cascade's at - /// `z + span + k`, which is exactly the set of rows its tiles rendered. - /// - /// An edge delivers under three conditions the proof states together. The proof holds the - /// edge's own link row and both of its endpoints. The delivered row sets intersect the proof - /// before edges qualify, and the link row carries the link entity's own authorization, which - /// its endpoints do not imply. - /// - /// Version 0 serves the full unfiltered edge set. The body vocabulary admits no visibility - /// filter, so a request naming one rejects as `invalid-body` rather than receiving bytes that - /// ignore the filter without saying so. - /// - /// # Errors - /// - /// Returns [`EdgesError::Tiles`] when the request lists more tiles than `limits.tiles`, - /// [`EdgesError::Depth`] when a listed zoom exceeds the generation's deepest served tile, and - /// [`EdgesError::Grid`] when a listed coordinate lies outside its zoom's grid. The delivery - /// contract is `view`'s, checked when it bound, so no rejection here is about it. - fn assemble_edges( - &self, - request: &EdgesRequest, - limits: EdgesLimits, - view: &View<'_>, - ) -> Result { - if request.tiles.len() > limits.tiles as usize { - return Err(EdgesError::Tiles { - count: request.tiles.len(), - maximum: limits.tiles, - }); - } - - let walk = Walk::of(self, view.proof()); - let bounds = self.delivered_bounds(&walk, view, &request.tiles)?; - let set = EdgeSet::of(self, view, bounds, limits.edges as usize); - - Ok(EdgesDocument { - complete: set.complete(), - edges: EdgeColumns::of( - &self.node_codec, - view.cohort().universe(self.node_universe), - set.edges(), - ), - }) - } - - /// Encodes an assembled document. - /// - /// `SALTILEE` envelope bytes, ready to send under `application/vnd.hash.saltile-v1`. The - /// response carries the detail trailer iff the caller supplies `details`. - /// - /// The trailer interns type URLs at encode time. The table is the bytewise-sorted union of - /// every edge's representative type, and each reference keys by index into it. - /// - /// # Panics - /// - /// This panics when supplied details do not cover the document's delivered edges, which is a - /// transport bug rather than request data. - #[must_use] - fn encode_edges( - &self, - document: &EdgesDocument, - details: Option<&EdgeLinkDetails<'_>>, - ) -> Vec { - let columns = details.map(|details| { - let table = Table::new(details.representative_type_urls().iter().flatten()); - let link_type_ids: IdVec>> = details - .representative_type_urls() - .iter() - .map(|url| url.as_ref().map(|url| table.index_of(url))) - .collect(); - - (details.labels(), table, link_type_ids) - }); - - let trailer = columns - .as_ref() - .map(|(labels, table, link_type_ids)| EdgesTrailer { - type_table: table.entries(), - link_labels: labels, - link_type_ids, - }); - - EdgesResponse { - generation: self.generation.id().digest(), - variant: 0, - complete: document.complete, - edges: &document.edges, - trailer, - } - .encode() - } - - /// Collects the union of the listed tiles' delivered sets, one per serving domain. - /// - /// A tile's delivered set is mode-independent, because its cumulative delta set equals its - /// total set. The union is therefore one cumulative-schedule read per tile, deduplicated by - /// the sets themselves. A scoped view reads its own cascade's prefix through its cut, arrivals - /// included, while an operator view reads the corpus schedule's runs and its overlay's - /// arrival runs. Each is the delivery the tile route answers under that same view. - fn delivered_bounds( - &self, - walk: &Walk<'_>, - view: &View<'_>, - tiles: &[TileCoordinate], - ) -> Result { - let mut rows = DenseBitSet::new_empty(self.rows.len()); - let mut arrivals = DenseBitSet::new_empty(view.arrivals().len()); - let maximum = self.grid.max_tile_depth(); - for &coordinate in tiles { - if coordinate.z > maximum { - return Err(EdgesError::Depth { - z: coordinate.z, - maximum, - }); - } - let cell = grid::cell_of(coordinate).ok_or(EdgesError::Grid { - z: coordinate.z, - x: coordinate.x, - y: coordinate.y, - })?; - - if let Some(cut) = view.cut() { - let row_ids = self.rows.view(); - for row in cut.total(coordinate.z, cell).rows { - match row { - ViewRow::Base(position) => { - rows.insert(row_ids[position]); - } - ViewRow::Arrival(index) => { - arrivals.insert(index); - } - } - } - } else { - walk.delivered_rows_into(coordinate.z, cell, &mut rows); - let deepest = self.grid.deepest(); - for bucket in self.grid.cut_buckets(coordinate.z) { - for (_, index) in view.overlay().run(bucket, cell, deepest) { - arrivals.insert(index); - } - } - } - } - - Ok(DeliveredBounds { rows, arrivals }) - } -} diff --git a/libs/@local/graph/atlas/src/serve/error.rs b/libs/@local/graph/atlas/src/serve/error.rs deleted file mode 100644 index 5f511edc980..00000000000 --- a/libs/@local/graph/atlas/src/serve/error.rs +++ /dev/null @@ -1,475 +0,0 @@ -use core::{error::Error, fmt}; - -use crate::{ - file::{ - array::OpenArrayError, - generation::{GenerationId, OpenError}, - identity::read::OpenIdentityError, - morton::read::OpenMortonError, - postings::read::OpenPostingsError, - quad::read::OpenQuadError, - repository::IntegrityVerificationError, - sprs::read::OpenSprsError, - }, - identity::{BasePosition, ImportanceRank, NodeRowId}, - morton::Depth, - salt::{ - adjacency::InvalidAdjacencyFile, - fit::prepare::identity::InvalidIdentityFile, - postings::{artifact::InvalidPostingsFile, closure::ParentCycle}, - }, -}; - -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub enum ArrayKind { - Coordinates, - Rows, - Endpoints, - Ranks, - Positions, - RankPositions, -} - -impl fmt::Display for ArrayKind { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Coordinates => write!(fmt, "coordinates"), - Self::Rows => write!(fmt, "rows"), - Self::Endpoints => write!(fmt, "endpoints"), - Self::Ranks => write!(fmt, "ranks"), - Self::Positions => write!(fmt, "positions"), - Self::RankPositions => write!(fmt, "rank positions"), - } - } -} - -/// The identity domain an identity artifact serves, for error reporting. -/// -/// The identity artifacts share one file format. A failure therefore names which table it hit. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub enum IdentityDomain { - /// The ontology identity table, joining type rows to type uuids. - Ontology, - /// The node identity table, joining node rows to entity ids. - Node, - /// The edge identity table, joining edge rows to link-entity ids. - Edge, -} - -impl fmt::Display for IdentityDomain { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - fmt.write_str(match self { - Self::Ontology => "ontology", - Self::Node => "node", - Self::Edge => "edge", - }) - } -} - -/// Opening a generation's serving surface failed. -#[derive(Debug)] -pub enum OpenAtlasError { - /// The generation is not published in this root. - Unpublished(GenerationId), - Open(OpenError), - OpenQuad(OpenQuadError), - OpenMorton(OpenMortonError), - OpenArray { - kind: ArrayKind, - error: OpenArrayError, - }, - OpenAdjacency(OpenSprsError), - /// The adjacency file violates the incident-list contract. - Adjacency(InvalidAdjacencyFile), - OpenPostings(OpenPostingsError), - /// The postings file violates the artifact contract. - Postings(InvalidPostingsFile), - /// The postings parent graph holds a cycle. - /// - /// A cycle leaves no closure map to expand colouring requests. - Closure(ParentCycle), - /// An identity file failed to open. - OpenIdentity { - /// The identity domain the file serves. - domain: IdentityDomain, - /// The failure. - error: OpenIdentityError, - }, - /// Verifying a published file against the digest the metadata document records failed. - /// - /// The error names the file and, for a mismatch, both digests. - Corruption(IntegrityVerificationError), - /// An identity file violates the table contract. - /// - /// A key width other than the store's fails the open outright. A generation whose ids are not - /// store identities does not serve. - Identity { - /// The identity domain the file serves. - domain: IdentityDomain, - /// The violation. - error: InvalidIdentityFile, - }, - /// The recorded schedule needs more subdivisions than a 64-bit Morton key resolves. - Schedule { - /// The recorded cells-per-tile-axis exponent. - span_log2: u8, - /// The recorded deepest tile zoom. - max_tile_depth: u8, - }, - /// An artifact's element type or shape is not the serving contract's. - Shape { - /// The artifact's repository role. - kind: ArrayKind, - }, - /// The base-order columns disagree on the point count. - Columns { - /// Codes in the morton column. - codes: u64, - /// Points in the wire-coordinate column. - coordinates: u64, - /// Entries in the row column. - rows: u64, - /// Entries in the rank column. - ranks: u64, - /// Entries in the position permutation. - positions: u64, - /// Entries in the reverse rank permutation. - rank_positions: u64, - }, - /// The rank columns disagree on a sampled position. - /// - /// The rank column and its reverse invert each other, a contract the fit pipeline proves - /// when it constructs the generation. A file's recorded digest proves its bytes are the ones - /// the fit published and says nothing about whether two files agree with each other. Open - /// therefore roundtrips a bounded sample of positions through both columns and refuses a fit - /// that published a non-inverse pair. - RankInverse { - /// The sampled base position whose roundtrip failed. - position: BasePosition, - /// The position's recorded rank. - rank: ImportanceRank, - /// The position the reverse column holds at that rank, absent when the rank lies - /// outside the domain. - roundtrip: Option, - }, - /// The row columns disagree on a sampled position. - /// - /// The position column and its reverse invert each other, the same contract the rank columns - /// carry, and open spot-checks both pairs at the same sampled positions. - RowInverse { - /// The sampled base position whose roundtrip failed. - position: BasePosition, - /// The node row the position column holds at that position. - node: NodeRowId, - /// The position the reverse column holds for that node, absent when the node lies - /// outside the reverse column's domain. - roundtrip: Option, - }, - /// The adjacency's node domain contradicts the code column. - Nodes { - /// Node rows the adjacency spans. - adjacency: u64, - /// Codes in the morton column. - codes: u64, - }, - /// The adjacency's edge domain contradicts the endpoint column. - Edges { - /// Edge rows the adjacency spans. - adjacency: u64, - /// Pairs in the endpoint column. - endpoints: u64, - }, - /// The quadtree root's subtree count contradicts the code column. - Subtree { - /// The root node's subtree point count. - quad: u64, - /// Codes in the morton column. - codes: u64, - }, - /// The postings' point domain contradicts the code column. - Points { - /// Points the postings span. - postings: u64, - /// Codes in the morton column. - codes: u64, - }, - /// The postings' type domain contradicts the ontology identities. - Types { - /// Types the postings span. - postings: u64, - /// Rows in the ontology identity table. - identities: u64, - }, - /// The node identity table contradicts the code column. - Identities { - /// Rows in the node identity table. - identities: u64, - /// Codes in the morton column. - codes: u64, - }, - /// The edge identity table contradicts the adjacency's edge domain. - EdgeIdentities { - /// Rows in the edge identity table. - identities: u64, - /// Edge rows the adjacency spans. - edges: u64, - }, - /// The row universe exceeds the wire's `u32` id domain. - Universe { - /// Entries in the row column. - rows: u64, - }, - /// The edge universe exceeds the `u32` edge-row domain. - EdgeUniverse { - /// Edge rows the adjacency spans. - edges: u64, - }, -} - -impl From for OpenAtlasError { - fn from(error: IntegrityVerificationError) -> Self { - Self::Corruption(error) - } -} - -impl From for OpenAtlasError { - fn from(error: OpenError) -> Self { - Self::Open(error) - } -} - -impl From for OpenAtlasError { - fn from(error: OpenSprsError) -> Self { - Self::OpenAdjacency(error) - } -} - -impl From for OpenAtlasError { - fn from(error: InvalidAdjacencyFile) -> Self { - Self::Adjacency(error) - } -} - -impl From for OpenAtlasError { - fn from(error: OpenPostingsError) -> Self { - Self::OpenPostings(error) - } -} - -impl From for OpenAtlasError { - fn from(error: InvalidPostingsFile) -> Self { - Self::Postings(error) - } -} - -impl From for OpenAtlasError { - fn from(error: ParentCycle) -> Self { - Self::Closure(error) - } -} - -impl From for OpenAtlasError { - fn from(error: OpenMortonError) -> Self { - Self::OpenMorton(error) - } -} - -impl From for OpenAtlasError { - fn from(error: OpenQuadError) -> Self { - Self::OpenQuad(error) - } -} - -impl fmt::Display for OpenAtlasError { - #[expect( - clippy::too_many_lines, - reason = "one display arm per open refusal; the taxonomy is the length" - )] - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Unpublished(id) => { - write!(fmt, "generation {id} is not published in this root") - } - Self::Open(error) => write!(fmt, "the artifact failed to open: {error}"), - Self::OpenMorton(error) => write!(fmt, "the morton artifact failed to open: {error}"), - Self::OpenQuad(error) => write!(fmt, "the quad artifact failed to open: {error}"), - Self::OpenArray { kind, error } => { - write!(fmt, "the {kind} artifact failed to open: {error}") - } - Self::OpenAdjacency(error) => { - write!(fmt, "the adjacency artifact failed to open: {error}") - } - Self::Adjacency(invalid) => { - write!( - fmt, - "the adjacency artifact violates the incident-list contract: {invalid}" - ) - } - Self::Schedule { - span_log2, - max_tile_depth, - } => { - write!( - fmt, - "the recorded schedule needs {max_tile_depth} + {span_log2} subdivisions \ - where a 64-bit Morton key resolves {}", - Depth::MAX.get(), - ) - } - Self::Shape { kind } => write!( - fmt, - "the {kind} artifact does not hold the serving contract's shape", - ), - Self::Columns { - codes, - coordinates, - rows, - ranks, - positions, - rank_positions, - } => { - write!( - fmt, - "the base-order columns disagree on the point count: {codes} codes, \ - {coordinates} coordinates, {rows} rows, {ranks} ranks, {positions} \ - positions, {rank_positions} rank positions", - ) - } - &Self::RankInverse { - position, - rank, - roundtrip: Some(roundtrip), - } => write!( - fmt, - "the rank columns are not inverse: position {position} carries rank {rank}, which \ - the reverse column sends to position {roundtrip}", - ), - &Self::RankInverse { - position, - rank, - roundtrip: None, - } => write!( - fmt, - "the rank columns are not inverse: position {position} carries rank {rank}, which \ - lies outside the rank domain", - ), - &Self::RowInverse { - position, - node, - roundtrip: Some(roundtrip), - } => write!( - fmt, - "the row columns are not inverse: position {position} holds node {node}, which \ - the reverse column sends to position {roundtrip}", - ), - &Self::RowInverse { - position, - node, - roundtrip: None, - } => write!( - fmt, - "the row columns are not inverse: position {position} holds node {node}, which \ - lies outside the reverse column's domain", - ), - Self::Corruption(error) => fmt::Display::fmt(error, fmt), - Self::Nodes { adjacency, codes } => write!( - fmt, - "the adjacency spans {adjacency} node rows where the code column holds {codes}", - ), - Self::Edges { - adjacency, - endpoints, - } => write!( - fmt, - "the adjacency spans {adjacency} edge rows where the endpoint column holds \ - {endpoints}", - ), - Self::Subtree { quad, codes } => write!( - fmt, - "the quadtree root counts {quad} points where the code column holds {codes}", - ), - Self::OpenPostings(error) => { - write!(fmt, "the postings artifact failed to open: {error}") - } - Self::Postings(error) => { - write!(fmt, "the postings artifact violates its contract: {error}") - } - Self::Closure(error) => { - write!(fmt, "no closure map exists: {error}") - } - Self::OpenIdentity { domain, error } => { - write!( - fmt, - "the {domain} identity artifact failed to open: {error}" - ) - } - Self::Identity { domain, error } => write!( - fmt, - "the {domain} identity artifact violates the table contract: {error}", - ), - Self::Points { postings, codes } => write!( - fmt, - "the postings span {postings} points where the code column holds {codes}", - ), - Self::Types { - postings, - identities, - } => write!( - fmt, - "the postings span {postings} types where the identity table holds {identities}", - ), - Self::Identities { identities, codes } => write!( - fmt, - "the node identity table holds {identities} rows where the code column holds \ - {codes}", - ), - Self::EdgeIdentities { identities, edges } => write!( - fmt, - "the edge identity table holds {identities} rows where the adjacency spans \ - {edges} edges", - ), - Self::Universe { rows } => write!( - fmt, - "the row column holds {rows} entries where wire ids span the u32 range", - ), - Self::EdgeUniverse { edges } => write!( - fmt, - "the adjacency spans {edges} edge rows where edge ids span the u32 range", - ), - } - } -} - -impl Error for OpenAtlasError { - fn source(&self) -> Option<&(dyn Error + 'static)> { - match self { - Self::Open(error) => Some(error), - Self::OpenMorton(morton) => Some(morton), - Self::OpenQuad(quad) => Some(quad), - Self::OpenArray { kind: _, error } => Some(error), - Self::OpenAdjacency(adjacency) => Some(adjacency), - Self::Adjacency(invalid) => Some(invalid), - Self::OpenPostings(postings) => Some(postings), - Self::Postings(invalid) => Some(invalid), - Self::Closure(cycle) => Some(cycle), - Self::OpenIdentity { domain: _, error } => Some(error), - Self::Identity { domain: _, error } => Some(error), - Self::Corruption(error) => Some(error), - Self::Unpublished(_) - | Self::Schedule { .. } - | Self::Shape { .. } - | Self::Columns { .. } - | Self::RankInverse { .. } - | Self::RowInverse { .. } - | Self::Nodes { .. } - | Self::Edges { .. } - | Self::Subtree { .. } - | Self::Points { .. } - | Self::Types { .. } - | Self::Identities { .. } - | Self::EdgeIdentities { .. } - | Self::Universe { .. } - | Self::EdgeUniverse { .. } => None, - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/grid.rs b/libs/@local/graph/atlas/src/serve/grid.rs deleted file mode 100644 index 4ee30324ff9..00000000000 --- a/libs/@local/graph/atlas/src/serve/grid.rs +++ /dev/null @@ -1,138 +0,0 @@ -//! The served tile grid. -//! -//! The bucket schedule and its addressing. -//! -//! One generation serves a quadtree of tiles whose delivery follows the recorded bucket schedule. A -//! tile at zoom `z` delivers the Morton buckets at or below the cut `z + span`, and the deepest -//! zoom's cut is the catch-all bucket holding every remaining point. [`Grid`] is that schedule -//! validated against the key width (`max_tile_depth + span ≤ 32`, the subdivisions a 64-bit Morton -//! key resolves), so every depth the serve paths derive from it exists by construction. -//! -//! Addressing lives beside the schedule: [`cell_of`] maps a request's tile coordinate onto the -//! Morton grid, and [`tile_of`] inverts a point's key back to the tile owning it at a zoom. - -use super::error::OpenAtlasError; -use crate::{ - morton::{Depth, MortonCell, MortonKey}, - salt::{lod::stage::LodConfig, wire::tile::TileCoordinate}, -}; - -/// The Morton key's coordinate width: each axis index carries 32 subdivision bits. -const AXIS_BITS: u8 = 32; - -/// One generation's validated bucket schedule. -/// -/// Construction proves `max_tile_depth + span` within the key width, so the cut of every served -/// zoom and every bucket at or below the deepest grid wrap into [`Depth`] without a failure path. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(super) struct Grid { - /// Cells per tile axis of the delivery cut, as its base-2 log. - span: u8, - /// The deepest tile zoom the schedule serves. - max_tile_depth: u8, -} - -impl Grid { - /// Validates the recorded schedule against the Morton key width. - /// - /// # Errors - /// - /// Returns [`OpenAtlasError::Schedule`] when `max_tile_depth + span` exceeds the 32 - /// subdivisions a 64-bit Morton key resolves, in which case no tile grid exists to serve. - pub(super) const fn new(config: LodConfig) -> Result { - if config.deepest().is_none() { - return Err(OpenAtlasError::Schedule { - span_log2: config.span.get(), - max_tile_depth: config.max_tile_depth, - }); - } - - Ok(Self { - span: config.span.get(), - max_tile_depth: config.max_tile_depth, - }) - } - - /// Returns the deepest tile zoom the schedule serves. - pub(super) const fn max_tile_depth(self) -> u8 { - self.max_tile_depth - } - - /// Returns the cells-per-tile-axis exponent of the delivery cut. - pub(super) const fn span_log2(self) -> u8 { - self.span - } - - /// Returns the delivery cut of zoom `z`: buckets at or below it form the zoom's cumulative - /// schedule. - /// - /// # Panics - /// - /// This panics beyond the served grid. A zoom above [`max_tile_depth`](Self::max_tile_depth) is - /// a caller defect rather than request data, and request validation rejects it first. - pub(super) const fn cut(self, z: u8) -> Depth { - assert!( - z <= self.max_tile_depth, - "the grid serves zooms 0..=max_tile_depth", - ); - Depth::new(z + self.span).expect("construction validated the schedule's deepest cut") - } - - /// Returns the deepest served bucket: the deepest zoom's cut, the catch-all. - /// - /// The fit clamps every natural bucket into it, so no corpus row and no delivered arrival - /// sits beyond it. - pub(super) const fn deepest(self) -> Depth { - self.cut(self.max_tile_depth) - } - - /// Iterates the cumulative schedule of zoom `z`: every bucket at or below its cut. - /// - /// # Panics - /// - /// As [`cut`](Self::cut). - pub(super) fn cut_buckets(self, z: u8) -> impl Iterator { - let cut = self.cut(z); - (0..=cut.get()).map(|bucket| Depth::new(bucket).expect("bounded by the validated cut")) - } - - /// Returns the first zoom whose cumulative schedule delivers bucket `bucket`. - /// - /// Bucket `b` first enters the schedule at zoom `b - span`, clamped to the root for the buckets - /// the root itself spans. - pub(super) const fn first_zoom(self, bucket: Depth) -> u8 { - bucket.get().saturating_sub(self.span) - } -} - -/// Returns the Morton cell a tile coordinate addresses. -/// -/// [`None`] outside the zoom's `2^z` grid or beyond the key width. -pub(super) const fn cell_of(coordinate: TileCoordinate) -> Option { - let Some(depth) = Depth::new(coordinate.z) else { - return None; - }; - - MortonCell::new(depth, coordinate.x, coordinate.y) -} - -/// Returns the tile owning a Morton key at zoom `zoom`. -/// -/// The key's axis indices truncate to the zoom's grid. Zoom 0 is the root tile whole. -/// -/// # Panics -/// -/// This panics beyond the key width, which the schedule's zooms rule out. -pub(super) const fn tile_of(key: MortonKey, zoom: u8) -> TileCoordinate { - assert!(zoom <= AXIS_BITS, "zooms lie within the key width"); - if zoom == 0 { - return TileCoordinate { z: 0, x: 0, y: 0 }; - } - - let [x, y] = key.coordinates(); - TileCoordinate { - z: zoom, - x: x >> (AXIS_BITS - zoom), - y: y >> (AXIS_BITS - zoom), - } -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/client.rs b/libs/@local/graph/atlas/src/serve/hydrate/client.rs deleted file mode 100644 index dfecbdece53..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/client.rs +++ /dev/null @@ -1,532 +0,0 @@ -//! The store boundary. -//! -//! Live detail reads over the serving store pool. -//! -//! Each hydration resolves its identities through the store's own query compiler, so a -//! statement reads under the live temporal axes and the draft exclusion and masks properties -//! per actor, by construction. A property value leaves the store -//! masked for the requesting actor under exactly the conditions the graph's entity reads mask -//! it - the deployment configures protection and the actor is not an instance admin - and -//! [`MaskingActor`] carries that actor from the scope's policy resolution into every order. -//! Label attribution reads the store's per-edition cache and no property value, so it stands -//! outside the masking, as labels do on the graph's own read path: see the trailer contract -//! in [the module above](super). -//! -//! Each hydration borrows one connection for its own duration and returns it, and statements -//! sharing the connection pipeline, so a request's hydration waits on one round trip of the -//! store's own work. - -use alloc::sync::Arc; -use core::pin::pin; - -use error_stack::Report; -use futures::{StreamExt as _, TryStreamExt as _}; -use hash_graph_postgres_store::store::{ - AsClient, PostgresStorePool, error::StoreError, postgres::query::SelectCompiler, -}; -use hash_graph_store::{ - filter::{ - Filter, - protection::{PropertyProtectionFilter, PropertyProtectionFilterConfig}, - }, - pool::StorePool as _, - subgraph::temporal_axes::{QueryTemporalAxes, QueryTemporalAxesUnresolved}, -}; -use hashql_core::{ - collections::FastHashMap, - id::{Id as _, IdSlice, IdVec, bit_vec::DenseBitSet}, -}; -use tokio::try_join; -use tokio_postgres::GenericClient; -use type_system::{ - knowledge::entity::id::EntityId, - ontology::{ - entity_type::EntityTypeUuid, - id::{BaseUrl, OntologyTypeUuid, VersionedUrl}, - }, - principal::actor::ActorId, -}; - -use super::{ - columns::{EdgeSlot, NodeSlot, ScalarValue}, - order::{LocateLinkHydration, LocateNodeHydration}, - statements::{DetailColumns, TypeColumns, TypeUrlColumns, identity_filter}, - type_urls::TypeUrlResolver, -}; -use crate::{ - bitset::DenseBitSlice, dataset::postgres::PostgresDatasetError, postgres::id::ArchivedEntityId, -}; - -/// The resolved actor one hydration masks properties for. -/// -/// Property protection is a per-actor condition on the graph's read path, and a hydration -/// carries the actor identity the scope's policy resolution produced. -#[derive(Debug, Copy, Clone)] -pub(crate) struct MaskingActor { - /// The actor the request's admitted scope names. - pub id: ActorId, - /// Whether the actor is an instance admin, whose reads bypass property protection. - pub instance_admin: bool, -} - -impl MaskingActor { - /// Returns whether `config` masks this actor's reads. - #[must_use] - pub(crate) fn masked_by(self, config: &PropertyProtectionFilterConfig<'_>) -> bool { - !config.is_empty() && !self.instance_admin - } - - /// Returns the property protection over this actor's reads, absent when nothing masks. - /// - /// The self-access clause binds to this actor, who reads their own protected properties. - #[must_use] - pub(crate) fn protection<'config, 'rules>( - self, - config: &'config PropertyProtectionFilterConfig<'rules>, - ) -> Option> { - self.masked_by(config) - .then(|| config.to_property_protection_filter(Some(self.id))) - } -} - -/// A detail hydration failed against the store. -#[derive(Debug)] -pub(crate) enum DetailError { - /// No connection was available for the query. - Connect(Report), - /// The store rejected the query. - Query(tokio_postgres::Error), - /// The channel carrying the answer closed before it arrived. - /// - /// The party holding the store side of the order dropped it, which happens when its request - /// ends early, so no answer can reach the response either way. - Disconnected, - /// The query returned too many rows. - TooManyRows, - /// The dataset-layer read behind a delta display failed. - Dataset(PostgresDatasetError), -} - -impl From for DetailError { - fn from(value: PostgresDatasetError) -> Self { - Self::Dataset(value) - } -} - -impl From for DetailError { - fn from(value: tokio_postgres::Error) -> Self { - Self::Query(value) - } -} - -impl core::fmt::Display for DetailError { - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::Connect(report) => { - write!( - fmt, - "the detail hydration reached no store connection: {report}" - ) - } - Self::Query(error) => write!(fmt, "the detail hydration failed: {error}"), - Self::Disconnected => { - fmt.write_str("the hydration channel closed before an answer arrived") - } - Self::TooManyRows => fmt.write_str("the detail hydration returned too many rows"), - Self::Dataset(error) => write!(fmt, "the link-display read failed: {error}"), - } - } -} - -impl core::error::Error for DetailError { - fn source(&self) -> Option<&(dyn core::error::Error + 'static)> { - match self { - Self::Query(error) => Some(error), - Self::Dataset(error) => Some(error), - Self::Connect(_) | Self::Disconnected | Self::TooManyRows => None, - } - } -} - -/// Reads every requested identity's resolution flag and direct-type URLs. -/// -/// # Panics -/// -/// This panics when the store answers rows outside the request domain, when a column does not -/// decode at its assigned position, or when a stored URL does not parse as its domain type. -async fn read_types( - client: &impl GenericClient, - ids: &IdSlice, - temporal_axes: &QueryTemporalAxes, -) -> Result<(DenseBitSet, IdVec>), DetailError> { - let filter = identity_filter(ids.iter().copied().map(EntityId::from)); - - let mut compiler = SelectCompiler::new(Some(temporal_axes), false); - compiler - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - - let columns = TypeColumns::select(&mut compiler); - let (statement, parameters) = compiler.compile(); - - let rows = client.query_raw(&statement, parameters).await?; - - let lookup: FastHashMap<_, _> = ids - .iter_enumerated() - .map(|(slot, &id)| (id, slot)) - .collect(); - - let mut resolved = DenseBitSet::new_empty(ids.len()); - let mut type_urls: IdVec<_, _> = IdVec::from_elem(Vec::new(), ids.len()); - - let mut rows = pin!(rows); - while let Some(row) = rows.next().await { - let row = row?; - let slot = lookup[&columns.entity_id(&row)]; - - if resolved.insert(slot) { - type_urls[slot].extend(columns.direct_type_urls(&row)); - } - } - - Ok((resolved, type_urls)) -} - -/// Reads one identity's capped properties and their completeness. -/// -/// `None` when the store no longer serves the identity. -/// -/// # Panics -/// -/// This panics when a column does not decode at its assigned position, and when a stored key -/// does not parse as a base URL. -async fn read_detail( - client: &(impl GenericClient + Sync), - source: ArchivedEntityId, - protection: Option<&PropertyProtectionFilter<'_, '_>>, - cap: usize, - temporal_axes: &QueryTemporalAxes, -) -> Result<(Option>, bool), DetailError> { - let filter = identity_filter([source.into()]); - - let mut compiler = SelectCompiler::new(Some(temporal_axes), false); - compiler - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - - let columns = DetailColumns::select(&mut compiler, protection); - let (statement, parameters) = compiler.compile(); - - let stream = client.query_raw(&statement, parameters).await?; - let mut stream = pin!(stream); - let Some(row) = stream.try_next().await? else { - return Ok((None, false)); - }; - - if stream.try_next().await?.is_some() { - return Err(DetailError::TooManyRows); - } - - let (properties, complete) = columns.capped_properties(&row, cap); - Ok((Some(properties), complete)) -} - -/// Live detail reads over the serving store pool. -/// -/// The pool's settings carry the deployment's property protection, so a serving process masks -/// exactly the properties that process's store protects. -#[derive(Debug)] -pub(crate) struct GraphDatabaseClient { - pool: Arc, -} - -impl GraphDatabaseClient { - /// Opens the detail path over the serving store pool. - #[must_use] - pub(crate) const fn new(pool: Arc) -> Self { - Self { pool } - } - - /// Holds one connection for the duration of one hydration. - async fn connection(&self) -> Result { - self.pool - .acquire(None) - .await - .map_err(|report| DetailError::Connect(report.change_context(StoreError))) - } - - /// Answers the node half of one locate order. - /// - /// Every resolved node reads its resolution flag and direct-type URLs. The source, the first - /// delivered identity, also reads its capped scalar-valued properties and their completeness, - /// masked for `masking`'s actor. Entities the store no longer serves read `false` flags and - /// empty columns. - /// - /// # Errors - /// - /// Returns [`DetailError`] when the store rejects a query. - /// - /// # Panics - /// - /// This panics when the store answers rows outside the request domain, when a column does - /// not decode at its assigned position, or when a stored URL does not parse as its domain - /// type. - #[tracing::instrument(skip_all, fields(points = ids.len()))] - pub(crate) async fn locate_node_hydration( - &self, - ids: &IdSlice, - properties: u32, - masking: MaskingActor, - ) -> Result { - if ids.is_empty() { - return Ok(LocateNodeHydration::empty(0)); - } - - let connection = self.connection().await?; - let client = connection.as_client(); - - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - let protection = masking.protection(&self.pool.settings.filter_protection); - - let ((resolved, type_urls), (source_properties, source_properties_complete)) = try_join!( - read_types(client, ids, &temporal_axes), - read_detail( - client, - ids[NodeSlot::MIN], - protection.as_ref(), - properties as usize, - &temporal_axes, - ), - )?; - - Ok(LocateNodeHydration { - resolved, - type_urls, - source_properties, - source_properties_complete, - }) - } - - /// Answers the link half of one locate order. - /// - /// Every resolved edge reads capped direct-type URLs and capped scalar-valued properties, - /// masked for `masking`'s actor, and a completeness flag accompanies each cap. Links the - /// store no longer serves read `None` properties, empty types, and `false` flags. - /// - /// # Errors - /// - /// Returns [`DetailError`] when the store rejects the query. - /// - /// # Panics - /// - /// This panics when the store answers rows outside the request domain, when a column does - /// not decode at its assigned position, or when a stored URL does not parse as its domain - /// type. - #[tracing::instrument(skip_all, fields(edges = ids.len()))] - pub(crate) async fn locate_link_hydration( - &self, - ids: &IdSlice, - type_ids: u32, - properties: u32, - masking: MaskingActor, - ) -> Result { - if ids.is_empty() { - return Ok(LocateLinkHydration::empty(0)); - } - - let connection = self.connection().await?; - let client = connection.as_client(); - - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - - let filter = identity_filter(ids.iter().copied().map(EntityId::from)); - let protection = masking.protection(&self.pool.settings.filter_protection); - let mut compiler = SelectCompiler::new(Some(&temporal_axes), false); - compiler - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - - let columns = DetailColumns::select(&mut compiler, protection.as_ref()); - let (statement, parameters) = compiler.compile(); - - let rows = client.query_raw(&statement, parameters).await?; - - let lookup: FastHashMap<_, _> = ids - .iter_enumerated() - .map(|(slot, id)| (*id, slot)) - .collect(); - - let mut type_url_columns: IdVec<_, _> = IdVec::from_elem(Vec::new(), ids.len()); - let mut type_urls_complete = DenseBitSlice::new_empty(ids.len()); - - let mut properties_columns: IdVec<_, _> = IdVec::from_elem(None, ids.len()); - let mut properties_complete = DenseBitSlice::new_empty(ids.len()); - - let mut rows = pin!(rows); - while let Some(row) = rows.next().await { - let row = row?; - let slot = lookup[&columns.entity_id(&row)]; - - let mut type_urls = columns.direct_type_urls(&row); - type_urls_complete.set(slot, type_urls.len() <= (type_ids as usize)); - - type_urls.truncate(type_ids as usize); - type_url_columns[slot].extend(type_urls); - - let (survivors, complete) = columns.capped_properties(&row, properties as usize); - - properties_columns.insert(slot, survivors); - properties_complete.set(slot, complete); - } - - Ok(LocateLinkHydration { - type_urls: type_url_columns, - type_urls_complete, - properties: properties_columns, - properties_complete, - }) - } -} - -impl TypeUrlResolver for GraphDatabaseClient { - /// Reads each requested type's versioned URL from the store's ontology records. - /// - /// The read carries no temporal condition. A type uuid derives from the URL it names, so any - /// row that exists answers correctly whatever its archival state. A type deleted from the - /// store is absent from the answer. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position or when a stored URL - /// does not parse as its domain type. - #[tracing::instrument(skip_all, fields(types))] - async fn resolve( - &self, - types: impl IntoIterator + Send, - ) -> Result, DetailError> { - let types = types.into_iter(); - tracing::Span::current().record("types", types.len()); - - if types.is_empty() { - return Ok(Vec::new()); - } - - let connection = self.connection().await?; - let client = connection.as_client(); - - let uuids: Vec<_> = types.map(EntityTypeUuid::from).collect(); - let filter = Filter::for_entity_type_uuids(&uuids); - - let mut compiler = SelectCompiler::new(None, false); - compiler - .add_filter(&filter) - .expect("the type-uuid filter compiles against the entity-type query paths"); - - let columns = TypeUrlColumns::select(&mut compiler); - let (statement, parameters) = compiler.compile(); - - let rows = client.query_raw(&statement, parameters).await?; - - let mut pairs = Vec::with_capacity(uuids.len()); - let mut rows = pin!(rows); - while let Some(row) = rows.next().await { - let row = row?; - - pairs.push(columns.pair(&row)); - } - - Ok(pairs) - } -} - -#[cfg(test)] -mod tests { - use hash_graph_store::filter::{ - Filter, FilterExpression, Parameter, protection::PropertyProtectionFilterConfig, - }; - use type_system::{ - knowledge::Entity, - principal::actor::{ActorId, UserId}, - }; - use uuid::Uuid; - - use super::MaskingActor; - - /// The masking actor over the user `actor` names. - fn masking(actor: u128, instance_admin: bool) -> MaskingActor { - MaskingActor { - id: ActorId::User(UserId::new(Uuid::from_u128(actor))), - instance_admin, - } - } - - /// Returns whether `filter` compares against the parameter `actor` anywhere in its tree. - fn binds_actor(filter: &Filter<'_, Entity>, actor: Uuid) -> bool { - let is_actor = |expression: &FilterExpression<'_, Entity>| { - matches!( - expression, - FilterExpression::Parameter { - parameter: Parameter::Uuid(uuid), - .. - } if *uuid == actor - ) - }; - match filter { - Filter::All(filters) | Filter::Any(filters) => { - filters.iter().any(|filter| binds_actor(filter, actor)) - } - Filter::Not(filter) => binds_actor(filter, actor), - Filter::Equal(lhs, rhs) | Filter::NotEqual(lhs, rhs) => is_actor(lhs) || is_actor(rhs), - Filter::Exists { .. } - | Filter::Greater(..) - | Filter::GreaterOrEqual(..) - | Filter::Less(..) - | Filter::LessOrEqual(..) - | Filter::In(..) - | Filter::StartsWith(..) - | Filter::EndsWith(..) - | Filter::ContainsSegment(..) => false, - } - } - - /// A deployment that protects no property masks nobody. - #[test] - fn protection_empty_config() { - let config = PropertyProtectionFilterConfig::new(); - - assert!(!masking(11, false).masked_by(&config)); - assert!(masking(11, false).protection(&config).is_none()); - } - - /// An instance admin reads unmasked under a protecting deployment. - #[test] - fn protection_instance_admin() { - let config = PropertyProtectionFilterConfig::hash_default(); - - assert!(!masking(11, true).masked_by(&config)); - assert!(masking(11, true).protection(&config).is_none()); - } - - /// A plain actor reads under the deployment's protection, with the self-access clause bound - /// to that actor and to no other. - #[test] - fn protection_plain_actor() { - let config = PropertyProtectionFilterConfig::hash_default(); - - assert!(masking(11, false).masked_by(&config)); - let protection = masking(11, false) - .protection(&config) - .expect("a protecting deployment masks a plain actor"); - assert!(!protection.is_empty(), "the protection holds no rule"); - for (_property, filter) in protection.iter() { - assert!( - binds_actor(filter, Uuid::from_u128(11)), - "the rule does not compare against the reading actor: {filter:?}" - ); - assert!( - !binds_actor(filter, Uuid::from_u128(12)), - "the rule compares against another actor: {filter:?}" - ); - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/columns.rs b/libs/@local/graph/atlas/src/serve/hydrate/columns.rs deleted file mode 100644 index 9671c2d69b6..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/columns.rs +++ /dev/null @@ -1,358 +0,0 @@ -//! The hydrated data model. -//! -//! One column set per hydration read, each aligned to its delivered order: hydration writes them -//! off the store rows, assembly documents and encoders read them. An entity the store no longer -//! serves reads `null` in every column and stays outside every completeness set. - -use hashql_core::id::{IdSlice, IdVec}; -use type_system::ontology::id::{BaseUrl, VersionedUrl}; - -use crate::{ - bitset::DenseBitSlice, - dataset::auxiliary::{Icon, Label}, - identity::{BasePosition, NodeRowId}, - postgres::id::ArchivedEntityId, - serve::schedule::{ArrivalIndex, ArrivalRow, ViewRow}, -}; - -hashql_core::id::newtype! { - /// A reference to a delivered node by its slot in one response's delivered order. - /// - /// Slots are dense and zero-based over one response's delivered nodes. Every node detail - /// column aligns to this domain. A slot is valid only against the response that delivered it, - /// because two responses share no slot vocabulary. - pub(crate) struct NodeSlot(u32) -} - -hashql_core::id::newtype! { - /// A reference to a delivered edge by its slot in one response's edge order. - /// - /// Slots are dense and zero-based over one response's delivered edges. Every link detail - /// column aligns to this domain. A slot is valid only against the response that delivered it, - /// because two responses share no slot vocabulary. - pub(crate) struct EdgeSlot(u32) -} - -hashql_core::id::newtype! { - /// A reference to a required ontology type by its slot in one response's requirement order. - /// - /// Slots are dense and zero-based over the distinct type uuids one response's trailer - /// requires, in first-occurrence order over the delivered edges. The hydrated URL column - /// aligns to this domain. A slot is valid only against the response that required it, - /// because two responses share no slot vocabulary. - pub(crate) struct TypeSlot(u32) -} - -/// The node identities behind one delivered set, viewed in slot order. -/// -/// The hydration request's node subject. The view joins the delivered rows to their identities -/// on demand - a fitted row through the generation's identity column, a placed arrival through -/// the view's arrival table - and building one therefore allocates nothing. The transport that -/// must own the identities collects the iterator, which is the one copy the boundary pays. -#[derive(Debug, Copy, Clone)] -pub(crate) struct DeliveredNodes<'doc> { - /// The generation's identity column, row order. - ids: &'doc IdSlice, - /// The generation's row column, base order. - rows: &'doc IdSlice, - /// The delivered rows, slot order, each in the domain that publishes it. - delivered: &'doc IdSlice, - /// The view's arrival table, which the delivered arrival vessels address. - arrivals: &'doc IdSlice, -} - -impl<'doc> DeliveredNodes<'doc> { - /// Views one delivered set's identities in slot order. - /// - /// The order is the hydration key: every detail column the store answers aligns to it. - pub(crate) const fn new( - ids: &'doc IdSlice, - rows: &'doc IdSlice, - delivered: &'doc IdSlice, - arrivals: &'doc IdSlice, - ) -> Self { - Self { - ids, - rows, - delivered, - arrivals, - } - } - - /// Returns the delivered count the details must cover. - #[inline] - #[must_use] - #[cfg(test)] // The serve tests size expected trailers from delivered columns. - pub(crate) const fn count(&self) -> usize { - self.delivered.len() - } - - /// Iterates the delivered identities, in slot order. - /// - /// # Panics - /// - /// Iteration panics on a delivered row outside the identity column, which open's - /// cross-artifact validation rules out. - pub(crate) fn iter(&self) -> impl Iterator + 'doc { - let Self { - ids, - rows, - delivered, - arrivals, - } = *self; - - delivered.iter().map(move |&vessel| match vessel { - ViewRow::Base(position) => ids[rows[position]], - ViewRow::Arrival(index) => arrivals[index].identity, - }) - } -} - -impl IntoIterator for &DeliveredNodes<'_> { - type Item = ArchivedEntityId; - - type IntoIter = impl Iterator; - - fn into_iter(self) -> Self::IntoIter { - self.iter() - } -} - -/// Hydrated per-point tile details, aligned to the delivered order. -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct NodeDetails<'details> { - /// The display label per delivered point. - labels: Vec<&'details Label>, - /// The icon per delivered point. - icons: Vec<&'details Icon>, -} - -impl<'details> NodeDetails<'details> { - /// Assembles the columns, aligned to one delivered order. - pub(crate) const fn new(labels: Vec<&'details Label>, icons: Vec<&'details Icon>) -> Self { - Self { labels, icons } - } - - /// All-`null` details covering `count` points, the result when no id can resolve. - #[must_use] - #[cfg(test)] // The serve tests build unresolved-detail fixtures. - pub(crate) fn empty(count: usize) -> Self { - Self { - labels: vec![Label::EMPTY; count], - icons: vec![Icon::empty(); count], - } - } - - /// Views the label column, delivered order. - #[inline] - pub(crate) const fn labels(&self) -> &[&'details Label] { - &self.labels - } - - /// Views the icon column, delivered order. - #[inline] - pub(crate) const fn icons(&self) -> &[&'details Icon] { - &self.icons - } -} - -/// One scalar property value. -/// -/// A hydrated property value takes no other shape. The store filters out nested objects and arrays, -/// so they never cross the connection. -#[derive(Debug, Clone, PartialEq)] -pub(crate) enum ScalarValue { - /// A text scalar. - String(String), - /// A number the store renders integral, within `i64`. - Integer(i64), - /// Any other number. - /// - /// Store scalars are doubles on the wire. - Float(f64), - /// A boolean scalar. - Bool(bool), - /// An explicit null the entity carries. - Null, -} - -/// Hydrated per-point locate node details, aligned to the delivered order. -/// -/// Labels and direct types for every delivered node, plus properties and their completeness for -/// the source alone - neighbour detail is one locate away. -#[derive(Debug, Clone, PartialEq)] -pub(crate) struct LocateNodeDetails<'details> { - /// The display label per delivered point. - labels: IdVec, - /// The direct-type versioned URLs per delivered point, canonical order. - /// - /// Empty when the store no longer serves the entity or records no types for it. - type_urls: IdVec>, - /// The source's surviving properties, ascending by base URL. - /// - /// `None` marks a source the store no longer serves. A resolved source without scalar - /// properties reads an empty list. - source_properties: Option>, - /// Whether the source's surviving properties are the entity's whole deliverable set. - /// - /// `false` when the scalar-value filter or the cap dropped anything, and when the store no - /// longer serves the source. - source_properties_complete: bool, -} - -impl<'details> LocateNodeDetails<'details> { - /// Assembles the columns, aligned to one delivered order. - pub(crate) const fn new( - labels: IdVec, - type_urls: IdVec>, - source_properties: Option>, - source_properties_complete: bool, - ) -> Self { - Self { - labels, - type_urls, - source_properties, - source_properties_complete, - } - } - - /// Views the label column, slot order. - #[inline] - pub(crate) const fn labels(&self) -> &IdSlice { - &self.labels - } - - /// Views the direct-type URL column, slot order. - #[inline] - pub(crate) const fn type_urls(&self) -> &IdSlice> { - &self.type_urls - } - - /// Views the source's surviving properties. - /// - /// `None` marks a store-absent source. - #[inline] - pub(crate) const fn source_properties(&self) -> Option<&[(BaseUrl, ScalarValue)]> { - self.source_properties.as_deref() - } - - /// Returns whether the source's surviving properties are the entity's whole deliverable set. - #[inline] - pub(crate) const fn source_properties_complete(&self) -> bool { - self.source_properties_complete - } -} - -/// Hydrated per-link locate details, aligned to the delivered edge order. -/// -/// Every edge carries a label, direct types under a cap, properties under a cap, and both -/// completeness flags. -#[derive(Debug, PartialEq)] -pub(crate) struct LocateLinkDetails<'details> { - /// The link entity's display label per delivered edge. - labels: IdVec, - /// The link's direct-type versioned URLs per delivered edge, canonical order, capped. - /// - /// Empty when the store no longer serves the link or records no types for it. - type_urls: IdVec>, - /// The delivered edges whose type list is the link's whole direct set. - /// - /// An edge stays out when the cap truncated its list and when the store no longer serves the - /// link. - type_urls_complete: Box>, - /// The link's surviving properties per delivered edge, ascending by base URL. - /// - /// `None` marks a link the store no longer serves. - properties: IdVec>>, - /// The delivered edges whose surviving properties are the link entity's whole deliverable set. - properties_complete: Box>, -} - -impl<'details> LocateLinkDetails<'details> { - /// Assembles the columns, aligned to one delivered order. - pub(crate) const fn new( - labels: IdVec, - type_urls: IdVec>, - type_urls_complete: Box>, - properties: IdVec>>, - properties_complete: Box>, - ) -> Self { - Self { - labels, - type_urls, - type_urls_complete, - properties, - properties_complete, - } - } - - /// Views the link label column, slot order. - #[inline] - pub(crate) const fn labels(&self) -> &IdSlice { - &self.labels - } - - /// Views the capped direct-type URL column, slot order. - #[inline] - pub(crate) const fn type_urls(&self) -> &IdSlice> { - &self.type_urls - } - - /// Views the per-edge type completeness set, over the delivered edge slots. - #[inline] - pub(crate) fn type_urls_complete(&self) -> &DenseBitSlice { - &self.type_urls_complete - } - - /// Views the per-edge property column, slot order. - #[inline] - pub(crate) const fn properties( - &self, - ) -> &IdSlice>> { - &self.properties - } - - /// Views the per-edge property completeness set, over the delivered edge slots. - #[inline] - pub(crate) fn properties_complete(&self) -> &DenseBitSlice { - &self.properties_complete - } -} - -/// Hydrated per-link edges details, aligned to the delivered edge order. -/// -/// One label and one representative-type reference per edge. -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct EdgeLinkDetails<'details> { - /// The link entity's display label per delivered edge. - labels: IdVec, - /// The link's representative type's versioned URL per delivered edge. - representative_type_urls: IdVec>, -} - -impl<'details> EdgeLinkDetails<'details> { - /// Assembles the columns, aligned to one delivered order. - pub(crate) const fn new( - labels: IdVec, - representative_type_urls: IdVec>, - ) -> Self { - Self { - labels, - representative_type_urls, - } - } - - /// Views the link label column, slot order. - #[inline] - pub(crate) const fn labels(&self) -> &IdSlice { - &self.labels - } - - /// Views the representative-type URL column, slot order. - #[inline] - pub(crate) const fn representative_type_urls( - &self, - ) -> &IdSlice> { - &self.representative_type_urls - } -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/compile.rs b/libs/@local/graph/atlas/src/serve/hydrate/compile.rs deleted file mode 100644 index b870b720f18..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/compile.rs +++ /dev/null @@ -1,419 +0,0 @@ -//! Store-side visibility resolution that yields a proof of an actor's viewable rows. -//! -//! One query answers both halves of a request's scope. The actor's `ViewEntity` policies produce a -//! filter. The caller's own filter intersects with it when the request carries one. The result -//! selects the web id and entity uuid of every entity in the actor's viewable set. Each returned id -//! resolves against the generation's identity tables, and the rows that resolve form the proof's -//! two masks. Node rows go in one mask and link rows in the other. An id the generation never -//! fitted resolves once more through the resolution's placement cohort. A placed arrival -//! admits its slot into the node mask, so the proof's width follows the cohort's universe rather -//! than the generation's, and a published delta link enters the proof's admitted identity set, -//! so the same resolution authorizes the post-fit edges a response may append. -//! -//! Because the caller's filter and the policy filter meet in the same statement, a filtered request -//! and a permission-restricted request arrive at serving in the same shape, a proof -//! that admits fewer rows. The proof therefore carries the request's whole visible view. -//! -//! A caller filter also carries the store's own protection obligation. The store's entity reads -//! transform a caller's filter through [`PropertyProtectionFilterConfig`] before compiling it, so a -//! filter over a protected property cannot enumerate the entity types that configuration excludes. -//! This path compiles a caller filter as well. A proof is observable (the rows it admits are the -//! rows the responses deliver), so this path applies the same transformation under the same -//! condition, from a configuration its caller supplies. -//! -//! The proof admits exactly the rows the query returned. Permissions evaluate against the live -//! decision-time axes, so the proof reflects policy as it stands at request time. Entities the -//! store admits that the generation does not carry contribute no rows. - -use core::{error::Error, fmt, pin::pin}; - -use error_stack::Report; -use futures::StreamExt as _; -use hash_graph_authorization::policies::{ - MergePolicies, PolicyComponents, - action::ActionName, - store::{PolicyStore, PrincipalStore, error::ContextCreationError}, -}; -use hash_graph_postgres_store::store::{ - AsClient, StoreProvider, - error::StoreError, - postgres::query::{SelectCompiler, SelectCompilerError}, -}; -use hash_graph_store::{ - entity::EntityQueryPath, - filter::{ - Filter, ParameterConversionError, - protection::{PropertyProtectionFilterConfig, transform_filter}, - }, - subgraph::temporal_axes::QueryTemporalAxesUnresolved, -}; -use hash_graph_types::ontology::DataTypeLookup; -use hashql_core::collections::fast_hash_set; -use tokio_postgres::GenericClient as _; -use type_system::{ - knowledge::{Entity, entity::id::EntityUuid}, - principal::{actor::ActorId, actor_group::WebId}, -}; -use uuid::Uuid; - -use super::MaskingActor; -use crate::{ - bitset::CompressedBitSet, - offload::OffloadError, - postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId}, - serve::{Atlas, VisibilityProof, delta::PlacementCohort}, -}; - -/// Resolving an actor's visible rows against the store failed. -/// -/// Each variant names one failing stage, so a caller can separate a request it can repair from a -/// condition it cannot. [`Filter`](Self::Filter) is the one variant a caller's own input produces. -#[derive(Debug)] -pub(crate) enum ProofError { - /// No store connection was available for the resolution. - Connect(Report), - /// Assembling the actor's policy set failed. - Policies(Report), - /// The caller's filter does not compile against the entity query paths. - Filter(Report), - /// The caller's filter carries a parameter that does not match its path's type. - Convert(Report), - /// The scope's held filter document does not parse. - Document(serde_json::Error), - /// The policy filter does not compile against the entity query paths. - PolicyFilter(Report), - /// The store rejected the visibility query. - Query(tokio_postgres::Error), - /// The visibility query stopped partway through its rows. - Rows(tokio_postgres::Error), - /// The offloaded schedule-and-census computation produced no value. - ComputeView(OffloadError), -} - -impl From for ProofError { - fn from(error: OffloadError) -> Self { - Self::ComputeView(error) - } -} - -impl fmt::Display for ProofError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Connect(_) => fmt.write_str("the resolution reached no store connection"), - Self::Policies(_) => fmt.write_str("the actor's policy set could not be assembled"), - Self::Filter(_) => fmt.write_str("the request filter does not compile"), - Self::Convert(_) => { - fmt.write_str("the request filter's parameters do not match its paths") - } - Self::Document(_) => fmt.write_str("the scope's held filter document does not parse"), - Self::PolicyFilter(_) => fmt.write_str("the policy filter does not compile"), - Self::Query(_) => fmt.write_str("the store rejected the visibility query"), - Self::Rows(_) => fmt.write_str("the visibility query stopped partway through its rows"), - Self::ComputeView(_) => { - fmt.write_str("the view's schedule and census failed to compute") - } - } - } -} - -impl Error for ProofError { - fn source(&self) -> Option<&(dyn Error + 'static)> { - match self { - Self::Connect(report) => Some(report.current_context()), - Self::Policies(report) => Some(report.current_context()), - Self::Filter(report) | Self::PolicyFilter(report) => Some(report.current_context()), - Self::Convert(report) => Some(report.current_context()), - Self::Document(error) => Some(error), - Self::Query(error) | Self::Rows(error) => Some(error), - Self::ComputeView(error) => Some(error), - } - } -} - -/// Returns whether a request answers with the whole generation without asking the store. -/// -/// True for an unconstrained view, where the caller narrows nothing and the compiled policy filter -/// is the tautology. Under those conditions the query returns every entity id the store holds and -/// the resulting proof admits every row. The check reads both conditions where the code decides -/// them rather than inferring them from the actor's kind. An actor's administrative standing is a -/// statement about its policies. The compiled filter already holds the resolution of those -/// policies. -/// -/// [`Filter::for_policies`] yields an empty [`Filter::All`] on exactly one path, an unconstrained -/// permit meeting no forbid. Every other shape carries a conjunct, a disjunct, or a negation. A -/// caller filter keeps the query, since a caller that narrows its view asked for the narrowed view. -const fn admits_every_row( - filter: Option<&Filter<'_, Entity>>, - policy_filter: &Filter<'_, Entity>, -) -> bool { - filter.is_none() && matches!(policy_filter, Filter::All(conjuncts) if conjuncts.is_empty()) -} - -/// Resolves the rows visible to `actor` as a [`VisibilityProof`] over `atlas`, beside the -/// [`MaskingActor`] the same policy resolution produced. -/// -/// `filter` narrows the view the proof admits. A request without one resolves the actor's whole -/// viewable set, while a request with one resolves its intersection with that set, so the proof -/// describes the view the request asked for. The masking actor travels with the proof so the -/// scope's hydrations mask properties for the actor whose rows the proof admits. -/// -/// An unconstrained view answers without the store. Every row is visible when the actor's policies -/// compile to the tautology and the request narrows nothing. The proof is then -/// [`VisibilityProof::full_visibility`] rather than one mask bit per row of the generation. -/// -/// A row is visible when the query returned its entity id. Both masks stay separate because the -/// link rows an actor's policies admit are not a function of the node rows they admit. -/// -/// `cohort` is the arrivals snapshot this resolution reads. A returned identity the generation -/// never fitted admits its cohort slot into the node mask, so a scoped proof answers placed -/// arrivals exactly where it answers fitted rows, and the mask's width follows the cohort's -/// universe. A returned identity the cohort publishes as a delta link enters the proof's -/// admitted identity set, the same admission one query grants the other three shapes. The -/// caller binds the same snapshot beside the proof, so the slots and links the proof admits and -/// the placements a request reads come from one publication. -/// -/// Caller requirement: `atlas` is the generation the proof serves, since row ids are that -/// generation's own. Caller requirement: the generation's node and link identity tables carry -/// disjoint entity ids. An id that both tables carry resolves as a node row. -/// -/// # Errors -/// -/// Returns [`ProofError::Policies`] when assembling the actor's policy set fails, -/// [`ProofError::Convert`] when a filter parameter does not match its path's type, -/// [`ProofError::Filter`] when `filter` does not compile, [`ProofError::PolicyFilter`] when the -/// policy filter does not compile, [`ProofError::Query`] when the store rejects the statement, and -/// [`ProofError::Rows`] when the row stream fails before it ends. A failure yields no proof, so a -/// partial row stream admits no rows anywhere. -#[tracing::instrument(skip_all)] -pub(crate) async fn visibility_proof( - actor: ActorId, - filter: Option<&Filter<'_, Entity>>, - protection: &PropertyProtectionFilterConfig<'static>, - store: &S, - atlas: &Atlas, - cohort: PlacementCohort<'_>, -) -> Result<(VisibilityProof, MaskingActor), ProofError> -where - S: PrincipalStore + PolicyStore + AsClient + Sync, - for<'store> StoreProvider<'store, S>: DataTypeLookup + Sync, -{ - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - let mut compiler = SelectCompiler::new(Some(&temporal_axes), false); - - let policy_components = PolicyComponents::builder(store, Some(actor)) - .with_action(ActionName::ViewEntity, MergePolicies::Yes) - .await - .map_err(ProofError::Policies)?; - - let masking = MaskingActor { - id: actor, - instance_admin: policy_components.is_instance_admin(), - }; - - let policy_filter = Filter::::for_policies( - policy_components.extract_filter_policies(ActionName::ViewEntity), - policy_components.actor_id(), - policy_components.optimization_data(ActionName::ViewEntity), - ); - - if admits_every_row(filter, &policy_filter) { - return Ok((VisibilityProof::full_visibility(), masking)); - } - - // Convert the caller filter's parameters to the types its paths expect - a web-id text - // parameter to a UUID, a property value to its data type - exactly as the entity read path - // does before it compiles (`PostgresStore::query_entities_impl`). The compiler binds a - // converted parameter with its column's type. An unconverted text parameter against a UUID - // or typed column compiles but the store rejects the statement at execution. The conversion - // reads data types through the same `StoreProvider` the read path builds. - let converted; - let filter = match filter { - Some(filter) => { - let mut owned = filter.clone(); - owned - .convert_parameters(&StoreProvider::new(store, &policy_components)) - .await - .map_err(ProofError::Convert)?; - converted = owned; - Some(&converted) - } - None => None, - }; - - // The store's read path transforms a caller's filter whenever the deployment configures - // protection and the actor is not an instance admin. The same condition governs here, since the - // same filter reaches the same compiler. - let protected; - let filter = match filter { - Some(filter) if masking.masked_by(protection) => { - protected = - transform_filter(filter.clone(), protection, 0, policy_components.actor_id()); - Some(&protected) - } - filter => filter, - }; - - if let Some(filter) = filter { - compiler.add_filter(filter).map_err(ProofError::Filter)?; - } - compiler - .add_filter(&policy_filter) - .map_err(ProofError::PolicyFilter)?; - - let web_id_index = compiler.add_selection_path(&EntityQueryPath::WebId); - let uuid_index = compiler.add_selection_path(&EntityQueryPath::Uuid); - - let (statement, parameters) = compiler.compile(); - let stream = store - .as_client() - .query_raw(&statement, parameters) - .await - .map_err(ProofError::Query)?; - - let mut nodes = CompressedBitSet::default(); - let mut edges = CompressedBitSet::default(); - let mut links = fast_hash_set(); - // Placed arrivals the actor may view, admitted into the node mask on their cohort slots. - let mut placed = 0_u64; - // Entities the actor may view that neither the generation nor the cohort carries: staged or - // unplaced arrivals, or of a shape the corpus does not place. - let mut unplaced = 0_u64; - - let mut stream = pin!(stream); - while let Some(row) = stream.next().await { - let row = row.map_err(ProofError::Rows)?; - - let web_id: WebId = row.get(web_id_index); - let uuid: EntityUuid = row.get(uuid_index); - - let id = ArchivedEntityId { - web_id: ArchivedWebId::from(Uuid::from(web_id)), - entity_uuid: ArchivedEntityUuid::from(Uuid::from(uuid)), - }; - - if let Some(row_id) = atlas.node_ids.row_of(id) { - nodes.insert(row_id); - } else if let Some(row_id) = atlas.edge_ids.row_of(id) { - edges.insert(row_id); - } else if let Some(arrival) = cohort.node(id) { - nodes.insert(arrival.id); - placed += 1; - } else if cohort.edge(id).is_some() { - links.insert(id); - } else { - unplaced += 1; - } - } - - tracing::debug!( - nodes = nodes.count(), - edges = edges.count(), - links = links.len(), - placed, - unplaced, - "resolved the actor's visible rows" - ); - - Ok((VisibilityProof::from_masks(nodes, edges, links), masking)) -} - -#[cfg(test)] -mod tests { - use hash_graph_authorization::policies::{ - Effect, OptimizationData, resource::ResourceConstraint, - }; - use hash_graph_store::filter::Filter; - use type_system::{ - knowledge::{Entity, entity::id::EntityId}, - principal::actor_group::WebId, - }; - use uuid::Uuid; - - use super::admits_every_row; - - /// The compiled tautology plus no caller filter is the unconstrained view. - /// - /// The bug class is a short-circuit that reads a shape the policy compiler does not reserve for - /// unconstrained permits. The expectation therefore comes from - /// [`Filter::for_policies`](hash_graph_store::filter::Filter::for_policies) itself, not from a - /// hand-built filter, so a change in what that constructor emits fails here rather than serving - /// a scoped caller the operator's rows. - #[test] - fn only_an_unconstrained_permit_admits_every_row() { - let optimization = OptimizationData::default(); - - let unconstrained = - Filter::::for_policies([(Effect::Permit, None)], None, &optimization); - assert!( - admits_every_row(None, &unconstrained), - "an unconstrained permit with no caller filter admits every row: {unconstrained:?}" - ); - - // A web-scoped permit is the same actor shape with one resource constraint, and it must - // keep the query. - let web = ResourceConstraint::Web { - web_id: WebId::new(Uuid::nil()), - }; - let scoped = - Filter::::for_policies([(Effect::Permit, Some(&web))], None, &optimization); - assert!(!admits_every_row(None, &scoped)); - - // A blank forbid denies everything, and no permit at all denies everything: neither is the - // tautology, and reading either as one would invert the decision. - let forbidden = - Filter::::for_policies([(Effect::Forbid, None)], None, &optimization); - assert!(!admits_every_row(None, &forbidden)); - let silent = Filter::::for_policies([], None, &optimization); - assert!(!admits_every_row(None, &silent)); - - // An unconstrained permit met by a forbid compiles to a negation, which the store must - // still evaluate. - let partly = Filter::::for_policies( - [(Effect::Permit, None), (Effect::Forbid, Some(&web))], - None, - &optimization, - ); - assert!(!admits_every_row(None, &partly)); - - // A scoped permit met by a forbid is the one shape that compiles to a non-empty - // conjunction. It is the dangerous neighbour of the tautology: reading the constructor - // rather than the conjunction's emptiness would answer this scoped actor with the whole - // generation. - let elsewhere = ResourceConstraint::Web { - web_id: WebId::new(Uuid::from_u128(1)), - }; - let scoped_with_forbid = Filter::::for_policies( - [ - (Effect::Permit, Some(&web)), - (Effect::Forbid, Some(&elsewhere)), - ], - None, - &optimization, - ); - assert!( - matches!(&scoped_with_forbid, Filter::All(conjuncts) if !conjuncts.is_empty()), - "the fixture builds the non-empty conjunction it is here to reject: \ - {scoped_with_forbid:?}" - ); - assert!(!admits_every_row(None, &scoped_with_forbid)); - } - - /// A caller filter keeps the query even under the tautology. - /// - /// The bug class is a short-circuit that answers the filtered request with the whole - /// generation. An operator asking for a narrowed view would receive every row instead, which is - /// a wrong answer rather than a leak. - #[test] - fn caller_filter_keeps_the_query() { - let optimization = OptimizationData::default(); - let unconstrained = - Filter::::for_policies([(Effect::Permit, None)], None, &optimization); - let requested = Filter::::for_entity_by_entity_id(EntityId { - web_id: WebId::new(Uuid::nil()), - entity_uuid: type_system::knowledge::entity::id::EntityUuid::new(Uuid::nil()), - draft_id: None, - }); - - assert!(!admits_every_row(Some(&requested), &unconstrained)); - } -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/mod.rs b/libs/@local/graph/atlas/src/serve/hydrate/mod.rs deleted file mode 100644 index 601c7c2254e..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/mod.rs +++ /dev/null @@ -1,79 +0,0 @@ -//! Live store reads that hydrate detail for delivered points and edges. -//! -//! Detail hydrates properties and type references at request time from Postgres, inline in the -//! trailer. Labels never ride these reads. Edges resolves them in process - the server's captured -//! displays first, the generation's payloads otherwise - while tile and locate serve generation -//! payloads plus each placed arrival's placement capture. -//! -//! # What a trailer carries -//! -//! The guarantees compose at two altitudes. Hydration reads only post-intersection ids, so every -//! hydrated entity is one the request's proof admits. That is the guarantee about *rows*. Inside an -//! admitted row, the deployment's property protection decides which *fields* may leave the store, -//! and an entity's **deliverable set** is every property no protection withholds from the -//! requesting actor. Property values reach a trailer from the deliverable set, and the caps, -//! counts, and completeness flags below all describe that set. -//! -//! The store's protection is a per-actor condition, and the hydration queries evaluate it for the -//! requesting actor. Each order carries the actor its scope's policy resolution produced -//! ([`MaskingActor`]), and the property statements compile through the store's own query compiler -//! under the read path's masking conditions, so a trailer withholds exactly what the graph's entity -//! reads withhold from that actor. An owner reads a protected value of their own where a stranger -//! does not, and an instance admin reads unmasked. -//! -//! Labels stand outside that rule, here and on the graph's own read path. A label is a property -//! value materialized per edition. The store derives `entity_edition_cache.labels[1]` from the -//! whole properties object through the type's `labelProperty` path, with no actor in the -//! derivation, so a type whose label property the deployment protects keeps that value in its label -//! column. Fitting copies that value into the generation's identity tables, and the server's -//! captured displays carry the same statement-shared spelling for later editions. Locate reads the -//! generation payloads plus each placed arrival's placement capture, and edges reads -//! captured-display-first, while hydration determines whether an entity still resolves and which -//! live type references and properties may leave the store. The label property itself leaves the -//! store under the same masking as every other property. -//! -//! The locate and edges responses deliver type *references* instead of rendered type display. Each -//! entity's direct types read from `entity_edition_cache.versioned_urls`, and the client resolves -//! their labels and icons through its own type metadata, so one owner holds each type display -//! concern. -//! -//! Properties reach the wire as [`ScalarValue`] entries only, covering strings, numbers, booleans, -//! and explicit nulls. Nested objects and arrays never survive the store-side filter. An over-cap -//! entity drops properties reverse-lexicographically by base URL, and its label property drops -//! last, so the label survives every cap that admits at least one property. That label property is -//! the base URL whose value provides the display label, resolved through the same canonical type -//! order the label cache uses. Survivors emit ascending by name, the wire's map-key order. A number -//! reaches the wire as an integer when the store renders it integral and it fits `i64`, and as a -//! double otherwise. Each hydration also counts the entity's *whole deliverable* set, so the -//! trailer reports completeness (nothing filtered, nothing capped) per entity from that count. -//! -//! An id that resolves to no visible entity - deleted since publish, archived, drafted, or with its -//! derived edition cache not yet landed - reads `null` in every column and `false` in every -//! completeness flag, mirroring the zero-mask rule for unresolvable type ids. -//! -//! The module splits by altitude: [`columns`] is the hydrated data model the documents and encoders -//! read, [`client`] is the store boundary - the queries and the one async connection - [`order`] is -//! the sync-facing capability one locate response hydrates through, and [`select`] is the pure -//! property-selection policy. - -mod client; -mod columns; -pub(crate) mod compile; -mod order; -pub(crate) mod select; -mod statements; -mod type_urls; - -// The hydration column constructors are test-only inputs for a fixture store's all-unresolved -// answer. No production caller constructs a hydration by hand. -#[cfg(test)] -pub(crate) use self::order::{LocateLinkHydration, LocateNodeHydration}; -pub(crate) use self::{ - client::{DetailError, GraphDatabaseClient, MaskingActor}, - columns::{ - DeliveredNodes, EdgeLinkDetails, EdgeSlot, LocateLinkDetails, LocateNodeDetails, - NodeDetails, NodeSlot, ScalarValue, TypeSlot, - }, - order::{EdgesStore, LocateHydration, LocateOrder, LocateStore}, - type_urls::{CachedTypeUrlResolver, TypeUrlResolver}, -}; diff --git a/libs/@local/graph/atlas/src/serve/hydrate/order.rs b/libs/@local/graph/atlas/src/serve/hydrate/order.rs deleted file mode 100644 index c7467032ee3..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/order.rs +++ /dev/null @@ -1,153 +0,0 @@ -//! The detail routes' store orders and the capabilities that answer them. -//! -//! A detail route assembles, hydrates, and encodes inside one synchronous call, and the store is -//! the one stage of that pipeline living on the other side of an executor. An assembled document -//! places one order naming the delivered identities and the caps, and one answer carries every -//! store-derived column back, so the boundary crosses as data rather than as control flow. Labels -//! stay out of every order on purpose. A label is a generation payload or a captured display, -//! either resolved in process, so an answer carries at most the resolution flags a label lookup -//! keys on rather than the labels themselves. -//! -//! [`LocateStore`] and [`EdgesStore`] are the capability shapes. A single call consumes each -//! shape, so a response hydrates at most once, and a rejection that never reaches hydration drops -//! it unused. - -use hashql_core::id::{IdSlice, IdVec, bit_vec::DenseBitSet}; -use type_system::ontology::id::{BaseUrl, VersionedUrl}; - -use super::{DeliveredNodes, EdgeSlot, NodeSlot, ScalarValue, TypeSlot, client::DetailError}; -use crate::{ - bitset::DenseBitSlice, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, -}; - -/// One locate response's store order. -/// -/// Both identity columns travel in delivered order, which is the alignment key for every column of -/// the answer. The caps are the serving limits the response encodes under, so the store applies -/// exactly the bounds the trailer reports against. -#[derive(Debug, Copy, Clone)] -pub(crate) struct LocateOrder<'doc> { - /// The delivered node identities, source first. - pub nodes: DeliveredNodes<'doc>, - /// The delivered link-entity identities, ascending identity bytes. - pub links: &'doc IdSlice, - /// Most properties the source's map delivers. - pub properties: u32, - /// Most direct-type URLs each link delivers. - pub link_type_ids: u32, - /// Most properties each link's map delivers. - pub link_properties: u32, -} - -/// The store's answer to one [`LocateOrder`], every column in delivered order. -#[derive(Debug, PartialEq)] -pub(crate) struct LocateHydration { - /// The node half of the answer. - pub nodes: LocateNodeHydration, - /// The link half of the answer. - pub links: LocateLinkHydration, -} - -/// The store-answered node columns of one locate hydration. -#[derive(Debug, PartialEq)] -pub(crate) struct LocateNodeHydration { - /// The delivered nodes the store resolved. - /// - /// An absent slot marks an entity the store no longer serves, whose every other column reads - /// empty and whose label stays empty. - pub resolved: DenseBitSet, - /// The direct-type versioned URLs per delivered node, canonical order. - /// - /// Empty when the store no longer serves the entity or records no types for it. - pub type_urls: IdVec>, - /// The source's surviving properties, ascending by base URL. - /// - /// `None` marks a source the store no longer serves. - pub source_properties: Option>, - /// Whether the source's surviving properties are the entity's whole deliverable set. - pub source_properties_complete: bool, -} - -impl LocateNodeHydration { - /// All-unresolved columns covering `count` nodes, the answer when no id can resolve. - #[must_use] - pub(crate) fn empty(count: usize) -> Self { - Self { - resolved: DenseBitSet::new_empty(count), - type_urls: IdVec::from_elem(Vec::new(), count), - source_properties: None, - source_properties_complete: false, - } - } -} - -/// The store-answered link columns of one locate hydration. -/// -/// The properties column doubles as the resolution flag. An entry is `Some` exactly when the store -/// resolved the link, so an unresolved link reads `None` there, empty types, and a slot outside -/// both completeness sets. -#[derive(Debug, PartialEq)] -pub(crate) struct LocateLinkHydration { - /// The link's direct-type versioned URLs per delivered edge, canonical order, capped. - pub type_urls: IdVec>, - /// The delivered edges whose type list is the link's whole direct set. - pub type_urls_complete: Box>, - /// The link's surviving properties per delivered edge, ascending by base URL. - pub properties: IdVec>>, - /// The delivered edges whose surviving properties are the link entity's whole deliverable set. - pub properties_complete: Box>, -} - -impl LocateLinkHydration { - /// All-unresolved columns covering `count` edges, the answer when no id can resolve. - #[must_use] - pub(crate) fn empty(count: usize) -> Self { - Self { - type_urls: IdVec::from_elem(Vec::new(), count), - type_urls_complete: DenseBitSlice::new_empty(count), - properties: IdVec::from_elem(None, count), - properties_complete: DenseBitSlice::new_empty(count), - } - } -} - -/// The store half of one locate response. -/// -/// One call consumes the capability, so a response hydrates at most once and the type states -/// that contract instead of a runtime check. An implementation answers the order from -/// wherever its store lives. The transport bridges to an async connection, and a test answers from -/// a fixture table with no store at all. -pub(crate) trait LocateStore { - /// Answers one locate order with every store-derived column. - /// - /// # Errors - /// - /// Returns [`DetailError`] when the store rejects a query or the answer can no longer reach - /// the caller. - fn hydrate(self, order: LocateOrder<'_>) -> Result; -} - -/// The store half of one edges response's detail trailer. -/// -/// The one order an edges response places is the distinct representative type uuids its -/// delivered links require, in first-occurrence order over the delivered slots, fitted and -/// delta alike - a delivered link's label and type uuid both resolve in process, from the -/// generation's payloads or the register's captured displays, so the store's whole share is -/// resolving each uuid to its versioned URL. `None` marks a type the store no longer serves. -/// -/// One call consumes the capability, so a response hydrates at most once and the type states -/// that contract instead of a runtime check. An implementation answers from wherever its -/// store lives, and a test answers from a fixture table with no store at all. -pub(crate) trait EdgesStore { - /// Answers one edges order: each required type uuid's versioned URL, in requirement order. - /// - /// # Errors - /// - /// Returns [`DetailError`] when the store rejects a query or the answer can no longer reach - /// the caller. - fn hydrate( - self, - types: &IdSlice, - ) -> Result>, DetailError>; -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/select.rs b/libs/@local/graph/atlas/src/serve/hydrate/select.rs deleted file mode 100644 index 85481a101d0..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/select.rs +++ /dev/null @@ -1,102 +0,0 @@ -//! The property-selection policy. -//! -//! Pure functions over hydrated property sets convert the store's property objects into -//! [`ScalarValue`] entries and apply the per-entity cap, which drops the label property last. -//! -//! The cap bounds size only. The sets reaching it hold no property the store's protection -//! withholds from the requesting actor, because the queries mask before any value crosses the -//! connection ([`client`](super::client)). - -use type_system::ontology::id::BaseUrl; - -use super::columns::ScalarValue; - -/// Converts one entity's property object into its [`ScalarValue`] entries. -/// -/// A key that is not a base URL and a nested value are store-contract violations: the write path -/// admits only base-URL keys and the query aggregates a filtered object. Either one skips its -/// entry with a warning rather than failing the read. -/// -/// # Panics -/// -/// This panics when the value is not a JSON object, which the store's aggregation rules out. -pub(crate) fn scalar_properties(value: serde_json::Value) -> Vec<(BaseUrl, ScalarValue)> { - let serde_json::Value::Object(object) = value else { - panic!("the store aggregates a JSON object") - }; - - object - .into_iter() - .filter_map(|(name, value)| { - let name = match BaseUrl::new(name) { - Ok(name) => name, - Err(error) => { - tracing::warn!( - %error, - "the store should key properties by base URL, but a key does not parse \ - as one" - ); - - return None; - } - }; - - let value = match value { - serde_json::Value::String(string) => ScalarValue::String(string), - serde_json::Value::Number(number) => { - if let Some(integer) = number.as_i64() { - ScalarValue::Integer(integer) - } else if let Some(float) = number.as_f64() { - ScalarValue::Float(float) - } else { - tracing::warn!( - %number, - "query should have returned only scalar values, but included a \ - number f64 cannot carry" - ); - - return None; - } - } - serde_json::Value::Bool(bool) => ScalarValue::Bool(bool), - serde_json::Value::Null => ScalarValue::Null, - value @ (serde_json::Value::Object(_) | serde_json::Value::Array(_)) => { - tracing::warn!( - ?value, - "query should have returned only scalar values, but included a JSON \ - object or array" - ); - - return None; - } - }; - - Some((name, value)) - }) - .collect() -} - -/// Selects the surviving properties under the per-entity cap. -/// -/// The drop order is reverse-lexicographic by base URL (bytewise), the label property drops last, -/// and survivors sort ascending by name, which is the wire's map-key order. -pub(crate) fn select_properties( - mut entries: Vec<(BaseUrl, ScalarValue)>, - label_property: Option<&BaseUrl>, - cap: usize, -) -> Vec<(BaseUrl, ScalarValue)> { - entries.sort_by(|(lhs_key, _), (rhs_key, _)| lhs_key.cmp(rhs_key)); - - if entries.len() > cap && cap > 0 { - // A label beyond the cap takes the last surviving slot. It compares greater than every - // earlier survivor, so the list stays ascending through the swap. - if let Some(offset) = label_property - .and_then(|label| entries[cap..].iter().position(|(name, _)| name == label)) - { - entries.swap(cap - 1, cap + offset); - } - } - - entries.truncate(cap); - entries -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap deleted file mode 100644 index 1fceb70e13a..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__bare_detail_statement_text.snap +++ /dev/null @@ -1,17 +0,0 @@ ---- -source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs -assertion_line: 388 -expression: detail.compile().0 ---- -SELECT "entity_temporal_metadata_0_0_0"."web_id", "entity_temporal_metadata_0_0_0"."entity_uuid", "entity_edition_cache_1_1_0"."versioned_urls", "entity_edition_cache_1_1_0"."direct_types", (SELECT jsonb_object_agg("scalar_property"."key", "scalar_property"."value") -FROM jsonb_each("entity_editions_1_1_0"."properties") AS "scalar_property"("key", "value") -WHERE jsonb_typeof("scalar_property"."value") = ANY(($6::text[]))), ((SELECT count(*) -FROM jsonb_each("entity_editions_1_1_0"."properties") AS "scalar_property"("key", "value"))::int4), ("entity_edition_cache_1_1_0"."label_properties")[1] -FROM "entity_temporal_metadata" AS "entity_temporal_metadata_0_0_0" -INNER JOIN "entity_editions" AS "entity_editions_0_1_0" - ON "entity_editions_0_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -INNER JOIN "entity_edition_cache" AS "entity_edition_cache_1_1_0" - ON "entity_edition_cache_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -INNER JOIN "entity_editions" AS "entity_editions_1_1_0" - ON "entity_editions_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -WHERE ("entity_temporal_metadata_0_0_0"."draft_id" IS NULL) AND ("entity_temporal_metadata_0_0_0"."transaction_time" @> $1::TIMESTAMPTZ) AND ("entity_temporal_metadata_0_0_0"."decision_time" && $2) AND (("entity_temporal_metadata_0_0_0"."web_id" = $3) AND ("entity_temporal_metadata_0_0_0"."entity_uuid" = $4) AND ("entity_editions_0_1_0"."archived" = $5)) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap deleted file mode 100644 index 7cc15cec2bc..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__masked_detail_statement_text.snap +++ /dev/null @@ -1,16 +0,0 @@ ---- -source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs -expression: detail.compile().0 ---- -SELECT "entity_temporal_metadata_0_0_0"."web_id", "entity_temporal_metadata_0_0_0"."entity_uuid", "entity_edition_cache_1_1_0"."versioned_urls", "entity_edition_cache_1_1_0"."direct_types", (SELECT jsonb_object_agg("scalar_property"."key", "scalar_property"."value") -FROM jsonb_each(("entity_editions_1_1_0"."properties" - (CASE WHEN ("entity_temporal_metadata_0_0_0"."entity_uuid" != $7) AND ("entity_edition_cache_1_1_0"."base_urls" @> ARRAY[$8]::text[]) THEN ARRAY[$9]::text[] ELSE ARRAY[]::text[] END))) AS "scalar_property"("key", "value") -WHERE jsonb_typeof("scalar_property"."value") = ANY(($6::text[]))), ((SELECT count(*) -FROM jsonb_each(("entity_editions_1_1_0"."properties" - (CASE WHEN ("entity_temporal_metadata_0_0_0"."entity_uuid" != $7) AND ("entity_edition_cache_1_1_0"."base_urls" @> ARRAY[$8]::text[]) THEN ARRAY[$9]::text[] ELSE ARRAY[]::text[] END))) AS "scalar_property"("key", "value"))::int4), ("entity_edition_cache_1_1_0"."label_properties")[1] -FROM "entity_temporal_metadata" AS "entity_temporal_metadata_0_0_0" -INNER JOIN "entity_editions" AS "entity_editions_0_1_0" - ON "entity_editions_0_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -INNER JOIN "entity_edition_cache" AS "entity_edition_cache_1_1_0" - ON "entity_edition_cache_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -INNER JOIN "entity_editions" AS "entity_editions_1_1_0" - ON "entity_editions_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -WHERE ("entity_temporal_metadata_0_0_0"."draft_id" IS NULL) AND ("entity_temporal_metadata_0_0_0"."transaction_time" @> $1::TIMESTAMPTZ) AND ("entity_temporal_metadata_0_0_0"."decision_time" && $2) AND (("entity_temporal_metadata_0_0_0"."web_id" = $3) AND ("entity_temporal_metadata_0_0_0"."entity_uuid" = $4) AND ("entity_editions_0_1_0"."archived" = $5)) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap deleted file mode 100644 index 3e480590dd6..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__type_urls_statement_text.snap +++ /dev/null @@ -1,11 +0,0 @@ ---- -source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs -expression: type_urls.compile().0 ---- -SELECT "entity_types_1_1_0"."ontology_id", "entity_types_1_1_0"."schema"->>'$id' -FROM "ontology_temporal_metadata" AS "ontology_temporal_metadata_0_0_0" -INNER JOIN "entity_types" AS "entity_types_0_1_0" - ON "entity_types_0_1_0"."ontology_id" = "ontology_temporal_metadata_0_0_0"."ontology_id" -INNER JOIN "entity_types" AS "entity_types_1_1_0" - ON "entity_types_1_1_0"."ontology_id" = "ontology_temporal_metadata_0_0_0"."ontology_id" -WHERE "entity_types_0_1_0"."ontology_id" = ANY($1) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap b/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap deleted file mode 100644 index 62dfb46a731..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/snapshots/hash_graph_atlas__serve__hydrate__statements__tests__types_statement_text.snap +++ /dev/null @@ -1,12 +0,0 @@ ---- -source: libs/@local/graph/atlas/src/serve/hydrate/statements.rs -assertion_line: 348 -expression: types.compile().0 ---- -SELECT "entity_temporal_metadata_0_0_0"."web_id", "entity_temporal_metadata_0_0_0"."entity_uuid", "entity_edition_cache_1_1_0"."versioned_urls", "entity_edition_cache_1_1_0"."direct_types" -FROM "entity_temporal_metadata" AS "entity_temporal_metadata_0_0_0" -INNER JOIN "entity_editions" AS "entity_editions_0_1_0" - ON "entity_editions_0_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -INNER JOIN "entity_edition_cache" AS "entity_edition_cache_1_1_0" - ON "entity_edition_cache_1_1_0"."entity_edition_id" = "entity_temporal_metadata_0_0_0"."entity_edition_id" -WHERE ("entity_temporal_metadata_0_0_0"."draft_id" IS NULL) AND ("entity_temporal_metadata_0_0_0"."transaction_time" @> $1::TIMESTAMPTZ) AND ("entity_temporal_metadata_0_0_0"."decision_time" && $2) AND (("entity_temporal_metadata_0_0_0"."web_id" = $3) AND ("entity_temporal_metadata_0_0_0"."entity_uuid" = $4) AND ("entity_editions_0_1_0"."archived" = $5)) diff --git a/libs/@local/graph/atlas/src/serve/hydrate/statements.rs b/libs/@local/graph/atlas/src/serve/hydrate/statements.rs deleted file mode 100644 index 7e86b7a9187..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/statements.rs +++ /dev/null @@ -1,406 +0,0 @@ -//! The hydration statements, built through the store's query compiler. -//! -//! Every read builds through the store's own [`SelectCompiler`], so a statement reads under -//! the live temporal axes and the draft exclusion and masks properties per actor, by -//! construction. Each column set adds its selections to a caller's -//! compiler and decodes the rows the compiled statement answers, so a row position is known -//! to exactly the type that assigned it. -//! -//! # The masking contract -//! -//! A statement reads property values only through the compiler's scalar-properties -//! selection, which reads the properties column through the same column compilation every -//! entity read uses, so a configured masking reaches the delivered map and the count for -//! exactly the actor the caller names. The compiler's masking hook fires when a property -//! selection compiles, so masking configures before any selection is added, which -//! [`DetailColumns::select`] holds by taking the masking itself. Label attribution reads the -//! cache's per-edition `label_properties` column and no property value, so no masking -//! applies to it. The tests hold this module to zero hand-composed reads of the properties -//! column. - -use hash_graph_postgres_store::store::postgres::query::SelectCompiler; -use hash_graph_store::{ - entity::EntityQueryPath, - entity_type::EntityTypeQueryPath, - filter::{Filter, FilterExpression, Parameter, protection::PropertyProtectionFilter}, - subgraph::edges::SharedEdgeKind, -}; -use type_system::{ - knowledge::{ - Entity, - entity::id::{EntityId, EntityUuid}, - }, - ontology::{ - entity_type::EntityTypeWithMetadata, - id::{BaseUrl, OntologyTypeUuid, VersionedUrl}, - }, - principal::actor_group::WebId, -}; - -use super::{ - columns::ScalarValue, - select::{scalar_properties, select_properties}, -}; -use crate::postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId}; - -/// Builds the filter naming exactly the requested identities, excluding archived editions. -/// -/// Every identity is a non-draft entity id, so the membership set is a disjunction of the read -/// path's own per-entity filters. -pub(super) fn identity_filter<'params>( - ids: impl IntoIterator, -) -> Filter<'params, Entity> { - Filter::All(vec![ - Filter::Any( - ids.into_iter() - .map(Filter::for_entity_by_entity_id) - .collect(), - ), - Filter::Equal( - FilterExpression::Path { - path: EntityQueryPath::Archived, - }, - FilterExpression::Parameter { - parameter: Parameter::Boolean(false), - convert: None, - }, - ), - ]) -} - -/// The output columns of one type-URL read. -pub(super) struct TypeColumns { - /// The web half of the entity's identity. - web_id: usize, - /// The entity half of the entity's identity. - entity_uuid: usize, - /// The cached versioned-URL array, direct types first. - type_urls: usize, - /// How many leading entries of the array are direct types. - direct_types: usize, -} - -impl TypeColumns { - /// Adds the identity and type-URL selections to `compiler`. - pub(super) fn select(compiler: &mut SelectCompiler<'_, '_, Entity>) -> Self { - Self { - web_id: compiler.add_selection_path(&EntityQueryPath::WebId), - entity_uuid: compiler.add_selection_path(&EntityQueryPath::Uuid), - type_urls: compiler.add_selection_path(&EntityQueryPath::EntityTypeEdge { - edge_kind: SharedEdgeKind::IsOfType, - path: EntityTypeQueryPath::VersionedUrl, - inheritance_depth: None, - }), - direct_types: compiler.add_selection_path(&EntityQueryPath::DirectTypeCount), - } - } - - /// Reads one row's direct-type URLs: the cached array cut to its direct-type prefix. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position. - pub(super) fn direct_type_urls(&self, row: &tokio_postgres::Row) -> Vec { - let direct: i32 = row.get(self.direct_types); - let direct = usize::try_from(direct).expect("the store counts direct types non-negatively"); - - let mut urls: Vec = row.get(self.type_urls); - urls.truncate(direct); - urls - } - - /// Reads one row's identity from the identity columns. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position. - pub(super) fn entity_id(&self, row: &tokio_postgres::Row) -> ArchivedEntityId { - let web_id: WebId = row.get(self.web_id); - let entity_uuid: EntityUuid = row.get(self.entity_uuid); - - ArchivedEntityId { - web_id: ArchivedWebId::from(web_id), - entity_uuid: ArchivedEntityUuid::from(entity_uuid), - } - } -} - -/// The output columns of one type-URL resolution read. -pub(super) struct TypeUrlColumns { - /// The type's URL-derived ontology uuid. - ontology_id: usize, - /// The type's versioned URL, the schema's own `$id`. - versioned_url: usize, -} - -impl TypeUrlColumns { - /// Adds the uuid and versioned-URL selections to `compiler`. - pub(super) fn select(compiler: &mut SelectCompiler<'_, '_, EntityTypeWithMetadata>) -> Self { - Self { - ontology_id: compiler.add_selection_path(&EntityTypeQueryPath::OntologyId), - versioned_url: compiler.add_selection_path(&EntityTypeQueryPath::VersionedUrl), - } - } - - /// Reads one row's uuid-URL pair. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position or when a stored URL - /// does not parse as its domain type. - pub(super) fn pair(&self, row: &tokio_postgres::Row) -> (OntologyTypeUuid, VersionedUrl) { - let uuid: OntologyTypeUuid = row.get(self.ontology_id); - let url: VersionedUrl = row.get(self.versioned_url); - - (uuid, url) - } -} - -/// The output columns of one detail read. -pub(super) struct DetailColumns { - /// The identity and type-URL positions. - types: TypeColumns, - /// The scalar property map position. - scalars: usize, - /// The whole property-count position. - total: usize, - /// The label-attribution position. - label: usize, -} - -impl DetailColumns { - /// Configures `masking` and adds the detail selections to `compiler`. - /// - /// The masking configures first, so every property selection compiles against the masked - /// column. - pub(super) fn select<'params, 'query: 'params>( - compiler: &mut SelectCompiler<'params, 'query, Entity>, - masking: Option<&'params PropertyProtectionFilter<'params, 'query>>, - ) -> Self { - if let Some(protection) = masking { - compiler.with_property_masking(protection); - } - - let types = TypeColumns::select(compiler); - let scalars = compiler.add_selection_path(&EntityQueryPath::ScalarProperties); - let total = compiler.add_selection_path(&EntityQueryPath::PropertyCount); - let label = compiler.add_selection_path(&EntityQueryPath::FirstLabelProperty); - - Self { - types, - scalars, - total, - label, - } - } - - /// Reads one row's identity from the identity columns. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position. - pub(super) fn entity_id(&self, row: &tokio_postgres::Row) -> ArchivedEntityId { - self.types.entity_id(row) - } - - /// Reads one row's direct-type URLs: the cached array cut to its direct-type prefix. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position. - pub(super) fn direct_type_urls(&self, row: &tokio_postgres::Row) -> Vec { - self.types.direct_type_urls(row) - } - - /// Reads one row's capped properties and their completeness flag. - /// - /// Both property columns read the same masked object, so completeness attests the - /// deliverable set: the survivors are that whole set exactly when the scalar-type filter - /// dropped nothing and the cap holds everything. A property the masking withholds is in - /// neither column and moves the flag not at all. The label property drops last under the - /// cap. - /// - /// # Panics - /// - /// This panics when a column does not decode at its assigned position, and when a stored - /// key does not parse as a base URL. - pub(super) fn capped_properties( - &self, - row: &tokio_postgres::Row, - cap: usize, - ) -> (Vec<(BaseUrl, ScalarValue)>, bool) { - let scalars: Option = row.get(self.scalars); - let total: i32 = row.get(self.total); - let label: Option = row.get(self.label); - - let entries = scalars.map_or_else(Vec::new, scalar_properties); - let total = usize::try_from(total).expect("the store counts properties non-negatively"); - let complete = entries.len() == total && entries.len() <= cap; - - (select_properties(entries, label.as_ref(), cap), complete) - } -} - -#[cfg(test)] -mod tests { - use hash_graph_postgres_store::store::postgres::query::SelectCompiler; - use hash_graph_store::{ - filter::protection::PropertyProtectionFilterConfig, - subgraph::temporal_axes::QueryTemporalAxesUnresolved, - }; - use type_system::{ - knowledge::entity::id::{EntityId, EntityUuid}, - ontology::entity_type::EntityTypeUuid, - principal::{ - actor::{ActorId, UserId}, - actor_group::WebId, - }, - }; - use uuid::Uuid; - - use super::{DetailColumns, Filter, TypeColumns, TypeUrlColumns, identity_filter}; - - /// The reading actor the masked pins bind their self-access clause to. - fn reading_actor() -> ActorId { - ActorId::User(UserId::new(Uuid::from_u128(11))) - } - - /// The identity filter over one nil identity, the fixture request. - fn nil_filter() -> super::Filter<'static, super::Entity> { - identity_filter([EntityId { - web_id: WebId::new(Uuid::nil()), - entity_uuid: EntityUuid::new(Uuid::nil()), - draft_id: None, - }]) - } - - /// The detail read masks its property columns exactly when the caller passes a masking. - /// - /// The masked spelling is the subtraction inside `jsonb_each(`, which is the compiler's - /// column hook firing inside each property subquery. The count is over the masked object - /// too, because a whole-object count against a masked map would tell an actor how many - /// properties were withheld, the enumeration signal the protection exists to close. - #[test] - fn detail_masked_both_subqueries() { - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - let filter = nil_filter(); - - let config = PropertyProtectionFilterConfig::hash_default(); - let protection = config.to_property_protection_filter(Some(reading_actor())); - - let mut masked = SelectCompiler::new(Some(&temporal_axes), false); - masked - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - DetailColumns::select(&mut masked, Some(&protection)); - let (masked_sql, _) = masked.compile(); - assert_eq!( - masked_sql - .matches(r#"jsonb_each(("entity_editions_1_1_0"."properties" - (CASE"#) - .count(), - 2, - "the protected detail read does not mask both property subqueries: {masked_sql}" - ); - - let mut bare = SelectCompiler::new(Some(&temporal_axes), false); - bare.add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - DetailColumns::select(&mut bare, None); - let (bare_sql, _) = bare.compile(); - assert_eq!( - bare_sql - .matches(r#"jsonb_each("entity_editions_1_1_0"."properties")"#) - .count(), - 2, - "the unprotected detail read does not read the bare object: {bare_sql}" - ); - } - - /// The rendered type-URL read, pinned as the text the store receives. - /// - /// The pin makes any rendering change - a selection edit here, or a change in the - /// compiler upstream - a visible snapshot diff in review instead of a silent swap of what - /// runs against the store. Each compiled pin holds the one-identity request, which is the - /// shape the masking assertions read. A request naming more identities compiles its - /// membership as a row comparison over unnested arrays, so the pinned grammar belongs to - /// the one-identity request alone. - #[test] - fn types_statement_text() { - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - let filter = nil_filter(); - - let mut types = SelectCompiler::new(Some(&temporal_axes), false); - types - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - TypeColumns::select(&mut types); - - insta::assert_snapshot!(types.compile().0); - } - - /// The rendered masked detail read, pinned under the deployment's default protection for a - /// resolved actor, the form every hydration compiles. - /// - /// The CASE conditions grow per protected property without changing the pinned grammar. Both - /// property subqueries read the masked object, and the self-access clause compares the entity - /// against the reading actor's parameter. A pin without an actor masks unconditionally and - /// cannot see that clause regress. - #[test] - fn masked_detail_statement_text() { - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - let filter = nil_filter(); - - let config = PropertyProtectionFilterConfig::hash_default(); - let protection = config.to_property_protection_filter(Some(reading_actor())); - let mut detail = SelectCompiler::new(Some(&temporal_axes), false); - detail - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - DetailColumns::select(&mut detail, Some(&protection)); - - insta::assert_snapshot!(detail.compile().0); - } - - /// The rendered bare detail read, pinned without any masking configured. - /// - /// Reviewing a diff, hold it to the masking contract: both property subqueries read the - /// bare object, and the text carries no CASE subtraction. - #[test] - fn bare_detail_statement_text() { - let temporal_axes = QueryTemporalAxesUnresolved::live_only().resolve(); - let filter = nil_filter(); - - let mut detail = SelectCompiler::new(Some(&temporal_axes), false); - detail - .add_filter(&filter) - .expect("the identity filter compiles against the entity query paths"); - DetailColumns::select(&mut detail, None); - - insta::assert_snapshot!(detail.compile().0); - } - - /// The rendered type-URL resolution read, pinned as the text the store receives. - /// - /// The membership array binds as one parameter, so this text is the rendering at every - /// batch width. The read carries no temporal condition on purpose. A type uuid derives - /// from the URL it names, so any row that exists answers correctly whatever its archival - /// state, and the pin makes an upstream compiler change that reintroduced a temporal - /// predicate a visible snapshot diff. - #[test] - fn type_urls_statement_text() { - let uuids = [EntityTypeUuid::from_url( - &"https://example.com/types/entity-type/fixture/v/1" - .parse() - .expect("the fixture URL parses"), - )]; - let filter = Filter::for_entity_type_uuids(&uuids); - - let mut type_urls = SelectCompiler::new(None, false); - type_urls - .add_filter(&filter) - .expect("the type-uuid filter compiles against the entity-type query paths"); - TypeUrlColumns::select(&mut type_urls); - - insta::assert_snapshot!(type_urls.compile().0); - } -} diff --git a/libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs b/libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs deleted file mode 100644 index 99100ddfeac..00000000000 --- a/libs/@local/graph/atlas/src/serve/hydrate/type_urls.rs +++ /dev/null @@ -1,258 +0,0 @@ -//! Resolution from ontology type uuids to their versioned URLs, behind a composable cache. -//! -//! An ontology type uuid derives from the versioned URL it names, so the mapping from uuid to -//! URL is immutable. A store row cannot change its URL under the same uuid, and a resolved pair -//! stays true for as long as the process lives. That is what makes a lazily filled, -//! never-invalidated cache correct here, and why the cache is generation-independent - a -//! generation swap changes which uuids a response requires, never what a uuid resolves to. -//! -//! [`TypeUrlResolver`] is the resolution capability and [`CachedTypeUrlResolver`] the cache -//! that composes over any source of it. The source stays cache-oblivious. It answers every uuid it -//! receives, and the wrapper decides which uuids reach it. - -use alloc::sync::Arc; -use std::sync::nonpoison::RwLock; - -use hashql_core::collections::FastHashMap; -use type_system::ontology::{VersionedUrl, id::OntologyTypeUuid}; - -use super::client::DetailError; - -/// The capability to resolve ontology type uuids to their versioned URLs. -pub(crate) trait TypeUrlResolver { - /// Resolves the uuids the source knows among `types`, as uuid-URL pairs. - /// - /// A uuid the source does not know is absent from the answer rather than an error, so a - /// caller distinguishes a failed read from a type the store no longer serves. The answer - /// may repeat a uuid, because the store can hold more than one row for one type across - /// archival cycles. Every such row answers the same immutable pair, so consumers key by - /// uuid rather than count. - /// - /// # Errors - /// - /// Returns [`DetailError`] when the source rejects the read or the answer can no longer - /// reach the caller. - // A plain `async fn` reveals no auto traits to generic callers, and the transport awaits - // this future inside a spawned task - that holds because every transport caller names its - // resolver concretely, so the future's `Send` leaks from the concrete impl instead of an - // explicit `impl Future + Send` bound. - async fn resolve( - &self, - types: impl IntoIterator + Send, - ) -> Result, DetailError>; -} - -impl TypeUrlResolver for Arc -where - T: TypeUrlResolver, -{ - async fn resolve( - &self, - types: impl IntoIterator + Send, - ) -> Result, DetailError> { - T::resolve(self, types).await - } -} - -/// A resolution source behind a lazily filled cache that never invalidates. -/// -/// Every resolved pair enters the cache, and a cached uuid never reaches the inner source -/// again. An unresolved uuid stays uncached on purpose: deriving uuids from URLs means a -/// re-created type resurrects under its old uuid, so absence is re-asked on every call rather -/// than remembered. -pub(crate) struct CachedTypeUrlResolver { - /// The cache-oblivious source answering the uuids the cache does not hold. - inner: T, - /// Every pair any resolution ever answered. - known: RwLock>, -} - -impl CachedTypeUrlResolver { - /// Wraps `inner` behind an empty cache. - pub(crate) fn new(inner: T) -> Self { - Self { - inner, - known: RwLock::new(FastHashMap::default()), - } - } -} - -impl TypeUrlResolver for CachedTypeUrlResolver -where - T: TypeUrlResolver, -{ - async fn resolve( - &self, - types: impl IntoIterator + Send, - ) -> Result, DetailError> { - let types = types.into_iter(); - - let mut found = Vec::with_capacity(types.len()); - let mut misses = Vec::new(); - - { - let known = self.known.read(); - - for uuid in types { - match known.get(&uuid) { - Some(url) => found.push((uuid, url.clone())), - None => misses.push(uuid), - } - } - } - - if misses.is_empty() { - return Ok(found); - } - - let fresh = self.inner.resolve(misses).await?; - - { - let mut known = self.known.write(); - for &(uuid, ref url) in &fresh { - known.insert(uuid, url.clone()); - } - } - - found.extend(fresh); - Ok(found) - } -} - -#[cfg(test)] -mod tests { - use alloc::sync::Arc; - use core::future; - use std::sync::nonpoison::Mutex; - - use hashql_core::collections::FastHashMap; - use type_system::ontology::{VersionedUrl, id::OntologyTypeUuid}; - - use super::{CachedTypeUrlResolver, TypeUrlResolver}; - use crate::serve::hydrate::client::DetailError; - - /// A resolution source that records every uuid set that reaches it. - struct Ledger { - urls: FastHashMap, - asked: Mutex>>, - } - - impl TypeUrlResolver for Ledger { - fn resolve( - &self, - types: impl IntoIterator + Send, - ) -> impl Future, DetailError>> - { - let types = types.into_iter().collect::>(); - self.asked.lock().push(types.clone()); - - future::ready(Ok(types - .into_iter() - .filter_map(|uuid| self.urls.get(&uuid).map(|url| (uuid, url.clone()))) - .collect())) - } - } - - fn type_url(ordinal: u64) -> VersionedUrl { - format!("https://example.com/types/entity-type/fixture-{ordinal}/v/1") - .parse() - .expect("the fixture URL parses") - } - - fn ledger(ordinals: impl IntoIterator) -> (Ledger, Vec) { - let urls: FastHashMap<_, _> = ordinals - .into_iter() - .map(|ordinal| { - let url = type_url(ordinal); - (OntologyTypeUuid::from_url(&url), url) - }) - .collect(); - let uuids = urls.keys().copied().collect(); - - ( - Ledger { - urls, - asked: Mutex::new(Vec::new()), - }, - uuids, - ) - } - - #[tokio::test] - async fn cache_hit() { - let (ledger, uuids) = ledger([0]); - let cached = CachedTypeUrlResolver::new(ledger); - - let first = cached - .resolve(uuids.iter().copied()) - .await - .expect("the source is total"); - let second = cached - .resolve(uuids.iter().copied()) - .await - .expect("the source is total"); - - assert_eq!(first, second); - assert_eq!( - *cached.inner.asked.lock(), - vec![uuids], - "the second resolution answers from the cache alone" - ); - } - - #[tokio::test] - async fn unresolved_retried() { - let (ledger, _) = ledger([]); - let unknown = OntologyTypeUuid::from_url(&type_url(7)); - let cached = CachedTypeUrlResolver::new(ledger); - - let first = cached - .resolve([unknown]) - .await - .expect("an absent uuid is not an error"); - let second = cached - .resolve([unknown]) - .await - .expect("an absent uuid is not an error"); - - assert!(first.is_empty()); - assert!(second.is_empty()); - assert_eq!( - cached.inner.asked.lock().len(), - 2, - "absence is never cached, so both calls reach the source" - ); - } - - #[tokio::test] - async fn partial_hit() { - let (ledger, known) = ledger([0, 1]); - let cached = CachedTypeUrlResolver::new(ledger); - - let warm: Vec<_> = known.iter().copied().take(1).collect(); - let warmed = cached.resolve(warm).await.expect("the source is total"); - assert_eq!(warmed.len(), 1); - let answer = cached - .resolve(known.iter().copied()) - .await - .expect("the source is total"); - - assert_eq!(answer.len(), 2); - let asked = cached.inner.asked.lock(); - assert_eq!( - asked[1], - known[1..].to_vec(), - "the warmed uuid stays out of the second read" - ); - } - - #[tokio::test] - async fn arc_source() { - let (ledger, uuids) = ledger([3]); - let cached = CachedTypeUrlResolver::new(Arc::new(ledger)); - - let answer = cached.resolve(uuids).await.expect("the source is total"); - - assert_eq!(answer.len(), 1); - } -} diff --git a/libs/@local/graph/atlas/src/serve/intern.rs b/libs/@local/graph/atlas/src/serve/intern.rs deleted file mode 100644 index 0cc9f642e3e..00000000000 --- a/libs/@local/graph/atlas/src/serve/intern.rs +++ /dev/null @@ -1,283 +0,0 @@ -//! Wire intern tables. -//! -//! A trailer stores each referenced value's wire spelling once. The table renders every -//! reference the trailer makes, deduplicated and ascending bytewise, and every reference keys by -//! index into it. The per-entity ascending-name order the hydration layer produces maps to -//! ascending index order, so the wire's ordering laws hold by construction. -//! -//! The table owns the ordering resolution between a domain type and its wire spelling. A -//! [`VersionedUrl`] orders its version numerically, which disagrees with bytewise order over the -//! rendering for multi-digit versions, so the wire order runs over the renderings while the -//! value's own order only deduplicates. - -use alloc::borrow::Cow; -use core::{ - cmp::Ordering, - fmt::{self, Debug, Display}, - hash::{Hash, Hasher}, - marker::PhantomData, -}; - -use hashql_core::{ - algorithms::co_sort, - id::{Id, IdError, IdSlice, IdVec}, -}; -use type_system::ontology::id::{BaseUrl, VersionedUrl}; - -/// A value a trailer references by intern-table index. -/// -/// The contract behind [`Table`]'s deduplication: two references render equal exactly when they -/// are equal, so deduplicating by the reference's own order is deduplicating by rendering. -pub(super) trait Reference: Ord { - /// Renders the reference in its wire spelling. - fn rendering(&self) -> Cow<'_, str>; -} - -impl Reference for BaseUrl { - fn rendering(&self) -> Cow<'_, str> { - Cow::Borrowed(self.as_str()) - } -} - -impl Reference for VersionedUrl { - fn rendering(&self) -> Cow<'_, str> { - // The rendering is injective: the base URL ends with `/` by validation and the version - // is numeric, so `{base}v/{version}` parses back uniquely from its right end. - Cow::Owned(self.to_string()) - } -} - -/// One reference's index into its domain's intern table. -/// -/// The parameter is the interned domain. A type-table index and a property-table index are -/// distinct types, and one response builds exactly one table per domain, so an index cannot -/// reach the wrong table. Index order is the table's wire order: ascending bytewise over the -/// interned renderings. -pub(crate) struct TableIndex(u32, PhantomData T>); - -impl TableIndex { - /// Creates an index from a raw table position. - #[must_use] - #[inline] - pub(crate) const fn new(value: u32) -> Self { - Self(value, PhantomData) - } -} - -impl Copy for TableIndex {} - -impl Clone for TableIndex { - #[inline] - fn clone(&self) -> Self { - *self - } -} - -const impl PartialEq for TableIndex { - #[inline] - fn eq(&self, other: &Self) -> bool { - self.0 == other.0 - } -} - -const impl Eq for TableIndex {} - -const impl PartialOrd for TableIndex { - #[inline] - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -const impl Ord for TableIndex { - #[inline] - fn cmp(&self, other: &Self) -> Ordering { - self.0.cmp(&other.0) - } -} - -impl Hash for TableIndex { - fn hash(&self, state: &mut H) { - self.0.hash(state); - } -} - -impl Debug for TableIndex { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - fmt.debug_tuple("TableIndex").field(&self.0).finish() - } -} - -impl Display for TableIndex { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - Display::fmt(&self.0, fmt) - } -} - -const impl TryFrom for TableIndex { - type Error = IdError; - - fn try_from(value: u32) -> Result { - Ok(Self::new(value)) - } -} - -const impl TryFrom for TableIndex { - type Error = IdError; - - fn try_from(value: u64) -> Result { - if value <= u32::MAX as u64 { - #[expect( - clippy::cast_possible_truncation, - reason = "guarded by the range check" - )] - Ok(Self::new(value as u32)) - } else { - Err(IdError::OutOfRange { - value, - min: 0, - max: u32::MAX as u64, - }) - } - } -} - -const impl TryFrom for TableIndex { - type Error = IdError; - - fn try_from(value: usize) -> Result { - if value as u64 <= u32::MAX as u64 { - #[expect( - clippy::cast_possible_truncation, - reason = "guarded by the range check" - )] - Ok(Self::new(value as u32)) - } else { - Err(IdError::OutOfRange { - value: value as u64, - min: 0, - max: u32::MAX as u64, - }) - } - } -} - -const impl Id for TableIndex { - const MAX: Self = Self::new(u32::MAX); - const MIN: Self = Self::new(0); - - #[inline] - fn from_u32(index: u32) -> Self { - Self::new(index) - } - - #[inline] - fn from_u64(index: u64) -> Self { - assert!(index <= u32::MAX as u64, "id value must fit in `u32`"); - - #[expect( - clippy::cast_possible_truncation, - reason = "guarded by the width assertion" - )] - Self::new(index as u32) - } - - #[inline] - fn from_usize(index: usize) -> Self { - assert!( - index as u64 <= u32::MAX as u64, - "id value must fit in `u32`" - ); - - #[expect( - clippy::cast_possible_truncation, - reason = "guarded by the width assertion" - )] - Self::new(index as u32) - } - - #[inline] - fn as_u32(self) -> u32 { - self.0 - } - - #[inline] - fn as_u64(self) -> u64 { - self.0 as u64 - } - - #[inline] - fn as_usize(self) -> usize { - self.0 as usize - } - - #[inline] - fn prev(self) -> Option { - match self.0.checked_sub(1) { - Some(value) => Some(Self::new(value)), - None => None, - } - } -} - -/// One trailer's intern table. -/// -/// Construction collects every reference first, so lookups are total for the trailer that built -/// the table. Each unique reference renders once, at construction, and the entries borrow where -/// the reference already is its own spelling. -#[derive(Debug)] -pub(super) struct Table<'doc, T> { - /// The interned renderings, ascending bytewise, deduplicated. - entries: IdVec, Cow<'doc, str>>, - /// The interned references in their own ascending order, each with its wire index. - lookup: Vec<(&'doc T, TableIndex)>, -} - -impl<'doc, T: Reference + 'static> Table<'doc, T> { - /// Builds the table over every reference the trailer makes. - /// - /// # Panics - /// - /// This panics for a table above `u32::MAX` entries. - pub(super) fn new(references: impl IntoIterator) -> Self { - let mut values: Vec<&'doc T> = references.into_iter().collect(); - values.sort_unstable(); - values.dedup(); - - let (mut entries, mut lookup): (IdVec, _>, Vec<_>) = values - .into_iter() - .map(|value| (value.rendering(), (value, TableIndex::MIN))) - .collect(); - - // `str` orders byte-lexicographically, so the renderings' own order is the wire's - // bytewise order. - co_sort(entries.as_raw_mut(), &mut lookup); - for (index, (_, table_index)) in lookup.iter_mut().enumerate() { - *table_index = TableIndex::from_usize(index); - } - - lookup.sort_unstable_by_key(|&(value, _)| value); - - Self { entries, lookup } - } - - /// Returns a reference's table index. - /// - /// # Panics - /// - /// This panics for a reference the table does not intern, which construction over the whole - /// reference set rules out. - pub(super) fn index_of(&self, reference: &T) -> TableIndex { - let index = self - .lookup - .binary_search_by(|&(value, _)| value.cmp(reference)) - .expect("every reference is interned"); - - self.lookup[index].1 - } - - /// Views the interned renderings, ascending bytewise. - pub(super) const fn entries(&self) -> &IdSlice, Cow<'doc, str>> { - &self.entries - } -} diff --git a/libs/@local/graph/atlas/src/serve/locate.rs b/libs/@local/graph/atlas/src/serve/locate.rs deleted file mode 100644 index 670a3fea637..00000000000 --- a/libs/@local/graph/atlas/src/serve/locate.rs +++ /dev/null @@ -1,1015 +0,0 @@ -//! The locate endpoint. -//! -//! Source resolution, ego-graph assembly, and the request/assembly/encode surface. -//! -//! Locate answers the source's ego-graph, which is every edge incident to the source whose other -//! endpoint is visible, together with the partners those edges connect. Assembly is one adjacency -//! probe plus column gathers, with no spatial index behind it. Locate is the detail view, and the -//! trailer always accompanies the response, so serving locate requires a store connection for -//! hydration. - -use hashql_core::id::{IdSlice, IdVec}; -use type_system::ontology::id::{BaseUrl, VersionedUrl}; - -use super::{ - Atlas, ServeLimits, WireRow, - colour::Palette, - grid, - hydrate::{ - DeliveredNodes, DetailError, EdgeSlot, LocateLinkDetails, LocateNodeDetails, LocateOrder, - LocateStore, NodeSlot, ScalarValue, - }, - intern::{Table, TableIndex}, - neighbourhood::{DeltaEndpoint, EdgeColumns, EdgeOrigin, Neighbourhood, ServedEdge}, - schedule::{ArrivalIndex, ArrivalRow, ViewRow, cut::ScheduleCut}, - view::{View, ViewError}, - visibility::{ResolvedRow, VisibleRow}, -}; -use crate::{ - dataset::auxiliary::{Label, Legend}, - identity::{BasePosition, NodeRowId}, - math::Vec2, - morton::MortonKey, - postgres::id::ArchivedEntityId, - salt::{ - lod::stage::WIRE_FRAME, - wire::{ - locate::{LocateResponse, LocateTrailer, PropertyMap, PropertyValue}, - tile::TileCoordinate, - }, - }, -}; - -/// The locate endpoint's limits. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub struct LocateLimits { - /// Most ego-graph edges one response delivers. - /// - /// A larger incident set keeps the edges whose partners lie nearest the source, and HEAD - /// reports `complete: false`. Every delivered edge also costs live link hydration, so the cap - /// bounds the store round trip and not only wire bytes. - pub edges: u32 = 512, - /// Most properties the source delivers. - /// - /// An over-cap entity drops properties reverse-lexicographically by base URL with its label - /// property protected to the end, so the label survives every cap that admits at least - /// one property. - pub properties: u32 = 10, - /// Most direct types one delivered edge delivers. - /// - /// An over-cap link truncates its type list in canonical order and its completeness bit - /// reads unset. - pub link_type_ids: u32 = 5, - /// Most properties one delivered edge delivers. - /// - /// The source's drop rule per link. - pub link_properties: u32 = 10, -} - -const impl Default for LocateLimits { - fn default() -> Self { - Self { .. } - } -} - -/// The subject's identity in every domain a locate response speaks. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct SourcePoint { - /// The subject, in the domain that publishes it. - pub subject: SourceSubject, - /// The first zoom whose cumulative schedule delivers the point. - pub zoom: u8, - /// The point's tile at that zoom: the client's fly-to target. - pub cell: TileCoordinate, -} - -/// A locate subject in the domain that publishes it. -/// -/// The response speaks two row domains, exactly as a tile's delivered set does. Fitted rows -/// resolve their payloads from the generation's columns, and placed arrivals resolve theirs -/// from the view's arrival table, which holds exactly the admitted arrivals of the entry's -/// cohort. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum SourceSubject { - /// A fitted row, proven visible, with the base position behind it. - Base { - /// The node row id. - row: VisibleRow, - /// The base position behind the row. - position: BasePosition, - }, - /// A placed arrival, addressed into the view's arrival table. - Arrival(ArrivalIndex), -} - -impl Atlas { - /// Resolves a locate source. - /// - /// Upstream entity id to node row, base position, first visible zoom, and fly-to tile. - /// - /// [`None`] for everything that does not name a visible node - unparsable, draft-suffixed, - /// unknown, hidden by the proof, or an edge id - the transport's `unknown-entity` problem, - /// identical for missing and denied. An identity the generation never fitted resolves - /// through the view's arrival table instead, under the same absent answer for everything - /// the table does not hold. - pub(crate) fn resolve_source(&self, view: &View<'_>, entity_id: &str) -> Option { - let id = super::translate::parse(entity_id)?; - if let Some(row) = self.node_ids.row_of(id) { - let row = view.proof().verify(row)?; - - return self.source_point(view, row); - } - - self.arrival_source(view, id) - } - - /// Resolves a locate source named by its wire node row id. - /// - /// The identifier a rendered tile put in the client's hand. - /// - /// Ingress goes through [`Atlas::resolve`], the same keyed codec as egress, so the lookup is - /// pure arithmetic that resolves in process. [`None`] for out-of-universe values and for rows - /// the proof hides, collapsed at the resolution before any caller observes the cause. A wire - /// id allocated for a cohort slot resolves through the view's arrival table, exactly as the - /// same arrival's entity id does. - pub(crate) fn resolve_wire_source( - &self, - view: &View<'_>, - wire: WireRow, - ) -> Option { - match self.resolve(view.proof(), view.cohort(), wire)? { - ResolvedRow::Fitted(row) => self.source_point(view, row), - ResolvedRow::Arrival(identity) => self.arrival_source(view, identity), - } - } - - /// Returns the first zoom whose cumulative schedule delivers a base position under `cut`. - /// - /// [`None`] when the view's schedule does not hold the position. - /// - /// An operator view inverts the corpus cut rule off the position's fencepost segment, and a - /// scoped view inverts its own rule `z + span + k` over its own cascade, so the answer is a - /// function of the visible rows alone and carries no evidence of what the mask removed. - fn first_visible_zoom( - &self, - cut: Option>, - position: BasePosition, - ) -> Option { - // A corpus bucket is a first-occupant result over every row, hidden ones included, so a - // scoped view reading it would let a hidden row decide a visible row's zoom. - cut.map_or_else( - || Some(self.grid.first_zoom(self.morton.bucket_of(position))), - |cut| cut.first_zoom(position), - ) - } - - /// Answers a proven-visible node row's identity in every domain a locate response speaks. - /// - /// Base position, first visible zoom, and fly-to tile. - /// - /// [`None`] when `view`'s schedule holds no bucket for the row, which collapses into the - /// endpoint's `unknown-entity`: a source the view's own delivery never reaches is a source - /// this view cannot locate, and the resolution already answers missing and denied alike. - fn source_point(&self, view: &View<'_>, row: VisibleRow) -> Option { - // Both ingress paths converge here, so the withdrawal check covers the entity-keyed and - // the wire-keyed resolutions alike, and a withdrawn source is nonexistent to both. - if view - .delta() - .is_some_and(|delta| delta.withdraws_node(row.get())) - { - return None; - } - - let position = self.positions_of_row()[row.get()]; - let zoom = self.first_visible_zoom(view.cut(), position)?; - - let key = MortonKey::from_bits(self.morton.codes()[position].get()); - let cell = grid::tile_of(key, zoom); - - Some(SourcePoint { - subject: SourceSubject::Base { row, position }, - zoom, - cell, - }) - } - - /// Answers an admitted placed arrival's identity in every domain a locate response speaks. - /// - /// Arrival table index, first visible zoom, and fly-to tile. Both ingress paths converge - /// here, exactly as fitted rows converge on [`Self::source_point`], so the ingress - /// withdrawal filter covers the entity-keyed and the wire-keyed resolutions alike. - /// - /// The zoom inverts the view's own cut rule over the arrival's bucket. Under a bound cut - /// the bucket is the schedule's or the overlay's, and an operator view reads the overlay's - /// natural bucket clamped into the corpus catch-all, below which the fit lets nothing sit. - /// - /// [`None`] when the ingress capture withdraws the identity and when the view's arrival - /// table does not hold it - an arrival the scope's own resolution never admitted - both - /// collapsing into the endpoint's `unknown-entity` exactly as every fitted refusal does. - fn arrival_source(&self, view: &View<'_>, id: ArchivedEntityId) -> Option { - if view.delta().is_some_and(|delta| delta.withdraws(id)) { - return None; - } - - let arrivals = view.arrivals(); - let index = arrivals.partition_point(|row| row.identity < id); - let row = arrivals.get(index)?; - if row.identity != id { - return None; - } - - let zoom = view.cut().map_or_else( - || { - self.grid - .first_zoom(view.overlay().bucket_of(index).min(self.grid.deepest())) - }, - |cut| cut.arrival_first_zoom(index), - ); - - let [x, y] = WIRE_FRAME.quantize(row.position); - let cell = grid::tile_of(MortonKey::new(x, y), zoom); - - Some(SourcePoint { - subject: SourceSubject::Arrival(index), - zoom, - cell, - }) - } - - /// Assembles the locate ego-graph around a resolved source. - /// - /// Every edge incident to the source whose other endpoint is visible, and the partners those - /// edges connect. Fitted edges arrive through the generation's adjacency and post-fit links - /// through the entry cohort, so the graph spans both serving domains whatever domain the - /// source resolves in. - /// - /// Delivered order is the wire's pin: the source first, then the delivered edges' partners - /// ascending by wire row id. Partners derive from the post-cap edge set - a partner whose - /// every edge truncated is not delivered. Edges ride ascending by link-entity identity bytes - /// after the cap - the order is client-verifiable from the `EDGE_IDS` column alone. - /// - /// # Panics - /// - /// This panics when the view's arrival table does not hold an arrival source's identity, - /// which source resolution rules out. - pub(crate) fn locate_subgraph( - &self, - source: SourcePoint, - limits: LocateLimits, - view: &View<'_>, - ) -> LocateSubgraph { - let arrivals = view.arrivals(); - let cohort = view.cohort(); - - // The source's row in the entry universe, its wire-frame origin, and its delivered - // vessel, each in the domain that publishes it. An arrival's row is the cohort slot its - // placement took. - let (source_row, origin, source_vessel) = match source.subject { - SourceSubject::Base { row, position } => ( - row.get(), - self.positions()[position], - ViewRow::Base(position), - ), - SourceSubject::Arrival(index) => { - let row = &arrivals[index]; - let slot = cohort - .node(row.identity) - .expect("the view's arrival table holds the cohort's admitted arrivals") - .id; - - (slot, row.position, ViewRow::Arrival(index)) - } - }; - - let neighbourhood = Neighbourhood::of(self, view.proof(), view.delta()); - - // Hidden partners drop before selection: the cap selects among visible edges alone, and a - // response's cardinality is a function of the masked view. The generation's adjacency - // never names a cohort slot, so an arrival source's fitted half is empty by construction. - let mut edges: Vec<_> = match source.subject { - SourceSubject::Base { row, .. } => neighbourhood - .incident(row.get()) - .into_iter() - .map(|(edge, id)| (ServedEdge::Fitted(edge), id)) - .collect(), - SourceSubject::Arrival(_) => Vec::new(), - }; - edges.extend( - neighbourhood - .incident_links(view, source_row) - .into_iter() - .map(|(edge, id)| (ServedEdge::Delta(edge), id)), - ); - - let complete = edges.len() <= limits.edges as usize; - if !complete { - self.truncate_nearest(&mut edges, limits.edges as usize, source_row, origin, view); - } - - edges.sort_unstable_by_key(|&(_, id)| id); - - // Partners derive from the delivered edge set. Distinct rows - // carry distinct wire ids under the entry universe (the codec - // is a bijection), so adjacent dedup after the wire-keyed - // sort is exact. - let positions_of_row = self.positions_of_row(); - let universe = cohort.universe(self.node_universe); - let mut partners: Vec<_> = edges - .iter() - .flat_map(|&(edge, _)| edge.endpoints()) - .filter(|endpoint| endpoint.row() != source_row) - .map(|endpoint| match endpoint { - DeltaEndpoint::Fitted(row) => ( - self.node_codec.encode(row, universe), - ViewRow::Base(positions_of_row[row]), - ), - DeltaEndpoint::Arrival { identity, .. } => { - let index = arrival_index_of(arrivals, identity); - (arrivals[index].wire, ViewRow::Arrival(index)) - } - }) - .collect(); - partners.sort_unstable_by_key(|&(wire, _)| wire); - partners.dedup_by_key(|&mut (wire, _)| wire); - - let mut delivered = IdVec::with_capacity(partners.len() + 1); - delivered.push(source_vessel); - for &(_, vessel) in &partners { - delivered.push(vessel); - } - - LocateSubgraph { - delivered, - edges, - complete, - } - } - - /// Keeps the `cap` edges whose partners lie nearest the source. - /// - /// Ascending (squared wire-frame distance to the partner, partner first-visible zoom, - /// link-entity identity bytes): equidistant partners cede to the earlier-visible one, and - /// distinct identities make the key a total order. The key only selects - presentation order - /// stays ascending identity bytes. - /// - /// The zoom reads the view's own cut, so under a scoped view the tie-break ranks partners by - /// that view's own cascade and which authorized partners survive the cap is a function of the - /// visible rows alone. An arrival partner reads its position from the view's arrival table - /// and its zoom through the same inversion an arrival source resolves with. - fn truncate_nearest( - &self, - edges: &mut Vec<(ServedEdge, ArchivedEntityId)>, - cap: usize, - source_row: NodeRowId, - origin: Vec2, - view: &View<'_>, - ) { - if cap == 0 { - edges.clear(); - return; - } - - let positions = self.positions(); - let positions_of_row = self.positions_of_row(); - let arrivals = view.arrivals(); - let cut = view.cut(); - - let mut ranked: Vec<(NearestKey, (ServedEdge, ArchivedEntityId))> = edges - .drain(..) - .map(|(edge, id)| { - let (point, zoom) = match edge.opposite_endpoint(source_row) { - DeltaEndpoint::Fitted(row) => { - let position = positions_of_row[row]; - ( - positions[position], - // A partner the view's schedule does not hold cedes to every - // partner it does. The proof admitted each of these rows, so the - // schedule built over that proof holds them and the fallback never - // selects. - self.first_visible_zoom(cut, position).unwrap_or(u8::MAX), - ) - } - DeltaEndpoint::Arrival { identity, .. } => { - let index = arrival_index_of(arrivals, identity); - let zoom = cut.map_or_else( - || { - self.grid.first_zoom( - view.overlay().bucket_of(index).min(self.grid.deepest()), - ) - }, - |cut| cut.arrival_first_zoom(index), - ); - - (arrivals[index].position, zoom) - } - }; - let (dx, dy) = (point.x() - origin.x(), point.y() - origin.y()); - // Unfused f32 arithmetic pins the selection key, so - // independent derivations from the wire coordinates - // agree bit for bit. Squared distances are - // non-negative finite floats, whose bit patterns - // order exactly as their values do. - #[expect( - clippy::suboptimal_flops, - reason = "a fused mul_add rounds differently and reorders near-ties" - )] - let distance = (dx * dx + dy * dy).to_bits(); - - ( - NearestKey { - distance, - zoom, - identity: id, - }, - (edge, id), - ) - }) - .collect(); - // Partitioning at `cap - 1` places the cap smallest keys - a - // total order, since link identities are distinct - in the - // head. - ranked.select_nth_unstable_by_key(cap - 1, |&(key, _)| key); - ranked.truncate(cap); - edges.extend(ranked.into_iter().map(|(_, entry)| entry)); - } -} - -/// The nearest-partner truncation's sort key. -/// -/// The derived order is the selection rule, ascending squared distance to the partner, then the -/// partner's first visible zoom, then the link-entity identity bytes. -#[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord)] -struct NearestKey { - /// The squared wire-frame distance to the partner, by its bit pattern. - /// - /// Non-negative finite floats order by bits exactly as by value. - distance: u32, - /// The partner's first visible zoom, by which equidistant partners cede to the earlier-visible - /// one. - zoom: u8, - /// The link-entity identity, whose distinctness makes the key a total order. - identity: ArchivedEntityId, -} - -/// Resolves a delivered arrival endpoint's index in the view's arrival table. -/// -/// Caller requirement: `identity` resolved through this same table when its edge qualified, so -/// the lookup answers. -/// -/// # Panics -/// -/// This panics when `identity` resolves to no row of the table, which the caller requirement -/// rules out. -fn arrival_index_of( - arrivals: &IdSlice, - identity: ArchivedEntityId, -) -> ArrivalIndex { - let index = arrivals.partition_point(|row| row.identity < identity); - assert_eq!( - arrivals[index].identity, identity, - "a delivered arrival endpoint resolves in the view's arrival table", - ); - - index -} - -/// One assembled locate ego-graph. -/// -/// The delivered rows (source first, then partners ascending wire row id) and the capped edge -/// set among them. -#[derive(Debug, PartialEq, Eq)] -pub(crate) struct LocateSubgraph { - /// The delivered rows in delivered order, each in the domain that publishes it. - pub delivered: IdVec, - /// The delivered edges paired with their link-entity identities, ascending by those bytes. - pub edges: Vec<(ServedEdge, ArchivedEntityId)>, - /// Whether the response delivers every qualifying edge. `false` iff the cap truncated. - pub complete: bool, -} - -/// The source entity and the delivery knobs of one locate request. -/// -/// Exactly one of two domains names the source: `entityId` (the upstream identity a search result -/// or deep link carries) XOR `row` (the wire node row id a rendered tile put in the client's hand). -/// The fields are distinct JSON types, so the union is unambiguous, and assembly rejects both or -/// neither by name. -#[derive(Debug, Clone, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub(crate) struct LocateRequest { - /// The source entity id, in the node identity domain. - /// - /// Exactly one of this and `row` names the source. - #[serde(default)] - pub entity_id: Option, - /// The source as a wire node row id - the value a tile's `ROW_IDS` column delivered. - /// - /// Exactly one of this and `entityId` names the source. - #[serde(default)] - pub row: Option>, - /// Versioned type URLs conditioning the `TYPE_MASK` column. Absent or empty omits it. - /// - /// Also the `typeIdsComplete` reference set: the flag reads `true` exactly when these ids - /// cover the source's direct types. Entries parse at the transport boundary: a malformed URL - /// rejects the body, while a well-formed URL this generation never ingested is legal and - /// reads zero bits. - #[serde(default)] - #[schemars(with = "Vec")] - pub colored_type_ids: Vec, -} - -/// A locate request the atlas rejects, by name. -/// -/// Every variant is a named, data-carrying rejection for the transport layer to map onto its error -/// vocabulary. -#[derive(Debug)] -pub(crate) enum LocateError { - /// The source id does not name a visible node - nonexistent. - /// - /// Denied and unparsable are identical by doctrine (missing equals denied, and an id that - /// cannot name an entity is an entity that does not exist). An out-of-universe wire `row` - /// collapses here too: one body, whatever the input domain. - UnknownEntity, - /// The request does not name exactly one source. - /// - /// `entityId` and `row` are one subject in two identity domains, and the body must carry - /// exactly one of them. - Source { - /// How many of the two source fields the body carries - zero or two, never one. - carried: usize, - }, - /// The request carries more `coloredTypeIds` than the cap admits. - Types { - /// The carried id count. - count: usize, - /// The cap the manifest publishes as `limits.coloredTypeIds`. - maximum: u32, - }, - /// The delivery view did not bind. - /// - /// A binding refusal converts into this variant through [`From`], so one error union carries a - /// route's binding and assembly rejections together. [`Atlas::locate`] takes the view already - /// bound, so its own rejections are all request-shaped. - View(ViewError), - /// The store half of the response failed. - /// - /// Only [`Atlas::locate`] answers this, because that path places the hydration order itself. - Details(DetailError), -} - -impl core::fmt::Display for LocateError { - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::UnknownEntity => fmt.write_str("the entity id does not name a visible node"), - Self::Source { carried } => { - write!( - fmt, - "the request carries {carried} source fields where exactly one of entityId \ - and row names the subject" - ) - } - Self::Types { count, maximum } => { - write!( - fmt, - "the request carries {count} coloredTypeIds where the cap admits {maximum}" - ) - } - Self::View(error) => error.fmt(fmt), - Self::Details(error) => error.fmt(fmt), - } - } -} - -impl core::error::Error for LocateError {} - -impl From for LocateError { - fn from(value: ViewError) -> Self { - Self::View(value) - } -} - -/// One assembled locate response: everything [`Atlas::encode_locate`] needs. -/// -/// The document owns its columns apart from the view's arrival table, which it borrows for the -/// request's own scope. The envelope places hydration last, and the split mirrors it. Assembly -/// and encoding are CPU-bound, and hydration awaits the store between them - the hydration -/// boundary materializes the identities it sends, so the borrow never leaves the request. -#[derive(Debug)] -pub(crate) struct LocateDocument<'view> { - source: SourcePoint, - /// The delivered rows, source first, each in the domain that publishes it. - delivered: IdVec, - /// The view's arrival table, which the delivered arrival vessels address. - arrivals: &'view IdSlice, - /// The delivered edges in column form, ascending link-entity identity bytes. - edges: EdgeColumns, - complete: bool, - mask_set: Option, - /// The request's parsed palette, the `typeIdsComplete` reference set. - palette: Palette, -} - -impl Atlas { - /// Answers one locate request over its bound delivery view. - /// - /// `SALTILEL` envelope bytes carrying the source's ego-graph, ready to send under - /// `application/vnd.hash.saltile-v1`. - /// - /// Locate is the detail view, so the trailer always accompanies the response: `store` answers - /// the one hydration order this call places, and every label resolves in process - from the - /// generation's own payloads, or a placed arrival's placement capture - keyed on the - /// answer's resolution columns. - /// - /// # Errors - /// - /// As [`Atlas::assemble_locate`], plus [`LocateError::Details`] when the store half of the - /// response fails. - /// - /// # Panics - /// - /// This panics when an identity table contradicts the row columns behind the delivered set, - /// which open's cross-artifact validation rules out. - pub(crate) fn locate( - &self, - request: &LocateRequest, - limits: ServeLimits, - view: View<'_>, - store: impl LocateStore, - ) -> Result, LocateError> { - let document = self.assemble_locate(request, limits, &view)?; - - let nodes = self.locate_node_entities(&document); - - let hydration = store - .hydrate(LocateOrder { - nodes, - links: document.edges.ids(), - properties: limits.locate.properties, - link_type_ids: limits.locate.link_type_ids, - link_properties: limits.locate.link_properties, - }) - .map_err(LocateError::Details)?; - - let row_ids = self.rows.view(); - - // Unlike tile's detail pass, no captures_any hoist guards the per-row overlay reads - // here: the edge cap bounds locate's delivered set (the source plus at most the cap's - // partners), so the reads stay bounded per response. - let cohort = view.cohort(); - let node_labels = document - .delivered - .iter_enumerated() - .map(|(slot, &vessel)| -> &Label { - if !hydration.nodes.resolved.contains(slot) { - return Label::EMPTY; - } - - match vessel { - // Captured display first, generation payload second (the register's own - // precedence), so a revised fitted identity serves its freshest label. - ViewRow::Base(position) => { - let row = row_ids[position]; - let id = self - .node_ids - .id(row) - .expect("open validated the identity rows against the code column"); - - cohort.legend_of(id).map_or_else( - || { - self.node_ids - .payload_of(row) - .expect( - "open validated the identity rows against the code column", - ) - .label() - }, - Legend::label, - ) - } - // The generation holds no payload for an entity placed after the fit: the - // label is the placement's captured legend, under the same store-resolution - // hold every fitted label takes. - ViewRow::Arrival(index) => { - AsRef::::as_ref(&document.arrivals[index].legend).label() - } - } - }) - .collect(); - let node_details = LocateNodeDetails::new( - node_labels, - hydration.nodes.type_urls, - hydration.nodes.source_properties, - hydration.nodes.source_properties_complete, - ); - - let link_labels = document - .edges - .origins() - .iter() - .zip(document.edges.ids()) - .zip(&hydration.links.properties) - .map(|((&origin, &id), properties)| -> &Label { - // An unresolved link reads the empty label. - if properties.is_none() { - return Label::EMPTY; - } - - // Captured display first, generation payload second (the register's own - // precedence), so a revised fitted link serves its freshest label here exactly - // as it does on the edges trailer. - cohort - .legend_of(id) - .unwrap_or_else(|| match origin { - EdgeOrigin::Fitted(row) => self.edge_ids.payload_of(row).expect( - "open validated the identity rows against the adjacency's edges", - ), - EdgeOrigin::Delta => unreachable!( - "publication withholds a link until its legend captures, and the \ - delivered set admits links from the cohort's own snapshot" - ), - }) - .label() - }) - .collect(); - let link_details = LocateLinkDetails::new( - link_labels, - hydration.links.type_urls, - hydration.links.type_urls_complete, - hydration.links.properties, - hydration.links.properties_complete, - ); - - Ok(self.encode_locate(&document, &node_details, &link_details)) - } - - /// Assembles one locate request into its owned document. - /// - /// Every rejection happens here, so encoding cannot fail. - /// - /// The `coloredTypeIds` cap is the tile endpoint's own - one manifest key, - /// `limits.coloredTypeIds`, governs the field wherever it occurs. - /// - /// Version 0 serves the full unfiltered set. The body vocabulary admits no visibility filter, - /// so a request naming one rejects as `invalid-body` rather than receiving bytes that ignore - /// the filter. - /// - /// # Errors - /// - /// Returns [`LocateError::Source`] when the body does not name exactly one of `entityId` and - /// `row`, [`LocateError::UnknownEntity`] when the source does not resolve to a visible node, - /// and [`LocateError::Types`] when the request carries more `coloredTypeIds` than - /// `limits.tile.colored_type_ids`. The delivery contract is `view`'s, checked when it bound, - /// so no rejection here is about it. - fn assemble_locate<'view>( - &self, - request: &LocateRequest, - limits: ServeLimits, - view: &View<'view>, - ) -> Result, LocateError> { - if request.colored_type_ids.len() > limits.tile.colored_type_ids as usize { - return Err(LocateError::Types { - count: request.colored_type_ids.len(), - maximum: limits.tile.colored_type_ids, - }); - } - - // Both source forms resolve through different ingress paths yet reach the same SourcePoint - // domain, and every failure past this match is one rejection: unknown-entity. - let source = match (request.entity_id.as_deref(), request.row) { - (Some(id), None) => self.resolve_source(view, id), - (None, Some(wire)) => self.resolve_wire_source(view, wire), - (entity, row) => { - return Err(LocateError::Source { - carried: usize::from(entity.is_some()) + usize::from(row.is_some()), - }); - } - } - .ok_or(LocateError::UnknownEntity)?; - - let LocateSubgraph { - delivered, - edges, - complete, - } = self.locate_subgraph(source, limits.locate, view); - - let palette = Palette::of(&request.colored_type_ids); - let mask_set = (!palette.is_empty()).then(|| self.resolve_masks(&palette)); - - Ok(LocateDocument { - source, - delivered, - arrivals: view.arrivals(), - edges: EdgeColumns::of( - &self.node_codec, - view.cohort().universe(self.node_universe), - &edges, - ), - complete, - mask_set, - palette, - }) - } - - /// Views the entity identities behind the document's delivered nodes, in slot order. - /// - /// The node hydration request's subject. The link subject needs no counterpart: assembly - /// already materializes the delivered link identities as the document's `EDGE_IDS` column. - #[must_use] - pub(crate) fn locate_node_entities<'doc>( - &'doc self, - document: &'doc LocateDocument<'_>, - ) -> DeliveredNodes<'doc> { - DeliveredNodes::new( - self.node_ids.ids(), - self.rows.view(), - &document.delivered, - document.arrivals, - ) - } - - /// Encodes an assembled document with its hydrated details. - /// - /// `SALTILEL` envelope bytes, ready for the wire under `application/vnd.hash.saltile-v1`. - /// Locate is the detail view, so the trailer always accompanies the response and every call - /// needs hydrated details. - /// - /// The trailer interns type and property URLs at encode time. Each table is the bytewise-sorted - /// union of every reference the trailer makes, and every reference keys by index into it. The - /// per-entity ascending-name order the hydration layer produces is ascending index order, so - /// the wire laws hold by construction. The source's `HEAD` flags derive here: `typeIdsComplete` - /// tests the source's direct types against the request's `coloredTypeIds`, and - /// `propertiesComplete` echoes the hydration layer's whole-set attestation. - /// - /// # Panics - /// - /// This panics when supplied details do not cover the document's delivered nodes and edges, - /// which is a transport bug rather than request data. - #[must_use] - fn encode_locate( - &self, - document: &LocateDocument<'_>, - nodes: &LocateNodeDetails, - links: &LocateLinkDetails, - ) -> Vec { - let masks = document - .mask_set - .as_ref() - .map(|set| set.memberships(&self.postings)); - - // The source's identity always travels in HEAD, and the - // per-edge link identities are first-class columns. Both read - // in process: a fitted source from the generation-baked - // tables, an arrival from the view's own table. - let entity_id = match document.source.subject { - SourceSubject::Base { row, .. } => self - .node_ids - .id(row.get()) - .expect("open validated the identity rows against the code column"), - SourceSubject::Arrival(index) => document.arrivals[index].identity, - }; - - let type_ids_complete = covers_source_types( - nodes.source_properties().is_some(), - &nodes.type_urls()[NodeSlot::new(0)], - &document.palette, - ); - - let (type_table, type_ids, link_type_ids) = - intern_types(nodes.type_urls(), links.type_urls()); - let (property_table, mut property_maps) = - intern_properties(nodes.source_properties(), links.properties()); - let link_property_maps = property_maps.split_off(1); - let source_properties = property_maps - .pop() - .expect("the source's map is the intern order's first entry"); - - let link_properties: IdVec<_, _> = link_property_maps.iter().map(Option::as_ref).collect(); - - LocateResponse { - generation: self.generation.id().digest(), - variant: 0, - cell: document.source.cell, - complete: document.complete, - entity_id, - type_ids_complete, - properties_complete: nodes.source_properties_complete(), - delivered: &document.delivered, - arrivals: document.arrivals, - positions: self.positions(), - rows: self.wire_rows(), - masks: masks.as_deref(), - edges: &document.edges, - trailer: LocateTrailer { - type_table: type_table.entries(), - property_table: property_table.entries(), - labels: nodes.labels(), - type_ids: &type_ids, - properties: source_properties.as_ref(), - link_labels: links.labels(), - link_type_ids: &link_type_ids, - link_type_ids_complete: links.type_urls_complete(), - link_properties: &link_properties, - link_properties_complete: links.properties_complete(), - }, - } - .encode() - } -} - -/// One entity's interned property map. -/// -/// `None` marks an entity the store no longer serves. -type PropertyMapView<'doc> = Option>; - -/// Node representative-type references into one response's type table, delivered order. -type NodeTypeIds = IdVec>>; - -/// Link type lists into one response's type table, edge order. -type LinkTypeIds = IdVec>>; - -/// Returns whether a request's palette covers the source's direct types. -/// -/// The `typeIdsComplete` predicate holds when every direct type of the source names a palette -/// entry. Coverage compares ontology identities, the same identity the `TYPE_MASK` resolution -/// derives. `false` when the store no longer serves the source (`present` reads false) or records -/// no types for it, since coverage of an unreadable set carries no attestation. `false` too on a -/// palette with no resolvable entry, which covers nothing. -pub(crate) fn covers_source_types( - present: bool, - types: &[VersionedUrl], - palette: &Palette, -) -> bool { - present && !types.is_empty() && types.iter().all(|url| palette.covers(url)) -} - -/// Builds the type intern table and every type reference into it. -/// -/// The table is the bytewise-sorted, deduplicated rendering of each node's representative type -/// and each link's capped type list. Node references are the representative-type indexes -/// (`None` for a node without a recorded type). Link references keep the hydration layer's -/// canonical type order. -pub(crate) fn intern_types<'doc>( - nodes: &'doc IdSlice>, - links: &'doc IdSlice>, -) -> (Table<'doc, VersionedUrl>, NodeTypeIds, LinkTypeIds) { - let table = Table::new( - nodes - .iter() - .filter_map(|urls| urls.first()) - .chain(links.iter().flatten()), - ); - - let type_ids = nodes - .iter() - .map(|urls| urls.first().map(|url| table.index_of(url))) - .collect(); - let link_type_ids = links - .iter() - .map(|urls| urls.iter().map(|url| table.index_of(url)).collect()) - .collect(); - - (table, type_ids, link_type_ids) -} - -/// Views one hydrated value in the wire's borrowed form. -const fn wire_value(value: &ScalarValue) -> PropertyValue<'_> { - match value { - ScalarValue::String(text) => PropertyValue::Text(text.as_str()), - ScalarValue::Integer(value) => PropertyValue::Integer(*value), - ScalarValue::Float(value) => PropertyValue::Float(*value), - ScalarValue::Bool(flag) => PropertyValue::Boolean(*flag), - ScalarValue::Null => PropertyValue::Null, - } -} - -/// Builds the property intern table and the per-entity uint-index maps. -/// -/// The table is the bytewise-sorted, deduplicated union of the source's and every link's -/// surviving names; the returned maps lead with the source's, then the links' in edge order. Each -/// map keeps the hydration layer's ascending-name order, which maps to ascending indexes. -pub(crate) fn intern_properties<'doc>( - source: Option<&'doc [(BaseUrl, ScalarValue)]>, - links: &'doc IdSlice>>, -) -> (Table<'doc, BaseUrl>, Vec>) { - let sets = core::iter::once(source).chain(links.iter().map(Option::as_deref)); - - let table = Table::new( - sets.clone() - .flatten() - .flat_map(|entries| entries.iter().map(|(name, _)| name)), - ); - - let maps = sets - .map(|entry| { - let survivors = entry?; - - Some(PropertyMap::new_unchecked( - survivors - .iter() - .map(|(name, value)| (table.index_of(name), wire_value(value))) - .collect(), - )) - }) - .collect(); - - (table, maps) -} diff --git a/libs/@local/graph/atlas/src/serve/manifest.rs b/libs/@local/graph/atlas/src/serve/manifest.rs deleted file mode 100644 index 73f031aebc2..00000000000 --- a/libs/@local/graph/atlas/src/serve/manifest.rs +++ /dev/null @@ -1,232 +0,0 @@ -//! The manifest document. -//! -//! The Surface v1 bootstrap: the generation's serving contract - schedule, limits, provenance, no -//! corpus-derived aggregates - plus the one per-caller block, the resolved delivery schedule the -//! caller's authority token seals. - -use core::fmt; - -use super::{Atlas, ServeLimits, VARIANTS, VisibilityLimits, density::CutOffset}; -use crate::{file::generation::GenerationId, salt::wire::WIRE_VERSION}; - -/// The serving limits of the manifest's `limits` block. -/// -/// Each value comes from a value the server enforces, and an advertised limit never disagrees with -/// enforcement. Request-validation limits let a client validate before sending, and -/// response-shaping limits say what delivery truncates. The authority windows say when a client -/// renews its token and when a held token stops opening. -/// -/// A limit belongs here when a correct client's own behaviour depends on it, in the requests it may -/// send, the responses it must expect, or its refresh cadence, and the block carries nothing a -/// client cannot act on. The authority windows are the visibility cache's pair: the soft window is -/// the cadence at which entries refresh and clients re-fetch, and the hard window bounds a token's -/// age at open and an entry's age at answer. One configured value serves both, counted from an -/// entry's resolution for the cache and from a token's issuance for the token. The cache's entry -/// capacity governs no client behaviour and stays absent. -/// -/// The windows are safe for publication because a validity bound is discoverable by the party the -/// bound applies to. The holder of a token reads its issue time from the clear envelope, and -/// presenting the token is itself the test of whether it still opens. Publication therefore states -/// a threshold that holder could measure, and states nothing at all to a caller holding no token. -/// What would be a disclosure is a refusal that distinguishes its cause; an authority refusal names -/// none, and every cause answers one uniform refusal. -// Never built freehand outside tests: `ServeLimits::manifest_limits` is the one derivation. -#[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase")] -pub(crate) struct ManifestLimits { - /// Most `coloredTypeIds` entries one request may carry. - pub colored_type_ids: u32, - /// Most tiles one edges request may list. - pub edges_tiles: u32, - /// Most ego-graph edges one locate response delivers before the nearest-partner truncation. - pub locate_edges: u32, - /// Most properties a locate source delivers. - pub locate_properties: u32, - /// Most direct types one locate edge delivers. - pub locate_link_type_ids: u32, - /// Most properties one locate edge delivers. - pub locate_link_properties: u32, - /// Most entity ids one translate request may carry. - pub translate_entity_ids: u32, - /// The cadence at which a client re-fetches the manifest to renew its authority token, - /// seconds. - pub authority_refresh_seconds: u64, - /// The authority token's rejection bound, seconds. - pub authority_hard_seconds: u64, -} - -impl ServeLimits { - /// Derives the manifest's `limits` block from the values the server enforces. - #[must_use] - pub(crate) const fn manifest_limits(&self, visibility: VisibilityLimits) -> ManifestLimits { - ManifestLimits { - colored_type_ids: self.tile.colored_type_ids, - edges_tiles: self.edges.tiles, - locate_edges: self.locate.edges, - locate_properties: self.locate.properties, - locate_link_type_ids: self.locate.link_type_ids, - locate_link_properties: self.locate.link_properties, - translate_entity_ids: self.translate.entity_ids, - authority_refresh_seconds: visibility.soft.as_secs(), - authority_hard_seconds: visibility.hard.as_secs(), - } - } -} - -/// Everything a client needs before its first tile. -/// -/// Every block except [`Manifest::scope_schedule`] derives from serving configuration and snapshot -/// provenance alone and stays valid for the generation's lifetime; the scope block is the caller's -/// own, sealed into the authority token issued beside this document. -#[derive(Debug, Clone, serde::Serialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase")] -pub(crate) struct Manifest { - /// The generation identity, echoing the route. - pub generation: GenerationId, - /// The `SALTILE` family version the tile bytes speak. - pub wire_version: u16, - /// The variant names, in variant-index order. - pub variants: [&'static str; 1], - /// The bucket-cut schedule the tile grid follows. - /// - /// The generation's own corpus schedule, identical for every caller. - pub bucket_schedule: BucketSchedule, - /// The caller's resolved delivery schedule. - /// - /// The delivery-cut offset the accompanying authority token seals, with the cut rule it - /// yields. Restricted responses deliver scope-cascade buckets at or below `z + span + k`, and - /// this block is the decoder's input for attributing runs to buckets. The block varies per - /// caller and per session, one of the reasons the manifest response is `no-store`. - pub scope_schedule: ScopeCutSchedule, - /// The published serving limits. - pub limits: ManifestLimits, - /// The snapshot's decision-time point, ISO-8601. - /// - /// Absent for generations fitted from sources without temporal axes, such as synthetic - /// fixtures. - #[serde(skip_serializing_if = "Option::is_none")] - pub created_at: Option, -} - -/// The manifest's `bucketSchedule` block. -#[derive(Debug, Copy, Clone, serde::Serialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase")] -pub(crate) struct BucketSchedule { - /// Cells per tile axis of the delivery cut: `2^span`. - pub span: u32, - /// The rule by which zoom `z` delivers buckets at or below `z + span`. - #[schemars(with = "String")] - pub cut: BucketCut, - /// The deepest tile zoom the schedule serves. - pub max_zoom: u8, -} - -/// The manifest's `scopeSchedule` block: one caller's resolved delivery schedule. -#[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] -#[derive(Debug, Copy, Clone, serde::Serialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase")] -pub(crate) struct ScopeCutSchedule { - /// The resolved delivery-cut offset the authority token seals. - pub k: u8, - /// The rule by which zoom `z` delivers scope buckets at or below `z + span + k`. - #[schemars(with = "String")] - pub cut: BucketCut, - /// The deepest zoom at which this scope's schedule still delivers new points. - /// - /// The resolved view's deepest occupied bucket, carried through the cut rule and clamped to - /// the grid: past this zoom every tile repeats content the caller has already accumulated. - /// The value describes the fitted view at resolution. Post-fit arrivals can deepen the live - /// answer inside the feed's freshness bound, and the root tile's `minResolution` stays the - /// authority a session reads mid-flight. - pub max_zoom: u8, -} - -/// The cut rule of the manifest's `bucketSchedule.cut` key. -/// -/// Zoom `z`'s cumulative schedule delivers buckets at or below `z + span`, and the wire form is -/// that formula: the string `z+`. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct BucketCut { - /// The schedule's span exponent, the formula's addend. - span_log2: u8, -} - -impl fmt::Display for BucketCut { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(fmt, "z+{}", self.span_log2) - } -} - -impl serde::Serialize for BucketCut { - fn serialize(&self, serializer: S) -> Result - where - S: serde::Serializer, - { - serializer.collect_str(self) - } -} - -impl Atlas { - /// Assembles the generation's manifest document under the given request limits. - /// - /// `deepest_occupied` is the root delivery's deepest occupied bucket, the reading the root - /// tile's `HEAD` reports, and zero for an empty view. `scopeSchedule.maxZoom` derives from it - /// through the cut rule. - #[must_use] - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - pub(crate) fn manifest( - &self, - limits: ManifestLimits, - k: CutOffset, - deepest_occupied: u64, - ) -> Manifest { - // Timestamps serialize as ISO-8601 strings; anything else - // degrades to an absent `createdAt` rather than panicking a - // read path. - let created_at = self - .generation - .repository() - .metadata - .snapshot - .axes - .and_then(|axes| match serde_json::to_value(axes.decision_time) { - Ok(serde_json::Value::String(text)) => Some(text), - Ok(_) | Err(_) => None, - }); - - Manifest { - generation: self.generation.id(), - wire_version: WIRE_VERSION, - variants: VARIANTS, - bucket_schedule: BucketSchedule { - span: 1 << self.grid.span_log2(), - cut: BucketCut { - span_log2: self.grid.span_log2(), - }, - max_zoom: self.grid.max_tile_depth(), - }, - scope_schedule: ScopeCutSchedule { - k: k.get(), - cut: BucketCut { - span_log2: self.grid.span_log2() + k.get(), - }, - // Bucket `b` first enters at zoom `b - span - k`, and the deepest served tile - // bounds the answer: binding proves the catch-all inverts to `max_tile_depth`. - max_zoom: u8::try_from( - deepest_occupied - .saturating_sub(u64::from(self.grid.span_log2()) + u64::from(k.get())) - .min(u64::from(self.grid.max_tile_depth())), - ) - .expect("the minimum against a `u8` bound fits `u8`"), - }, - limits, - created_at, - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/mod.rs b/libs/@local/graph/atlas/src/serve/mod.rs deleted file mode 100644 index b67aa723004..00000000000 --- a/libs/@local/graph/atlas/src/serve/mod.rs +++ /dev/null @@ -1,422 +0,0 @@ -//! Opened generations answering atlas reads. -//! -//! [`Atlas::open`] maps one published generation's serving artifacts - quadtree topology, Morton -//! code column, wire coordinates, the base-order row column, the incident-edge adjacency with its -//! endpoint column, and the rank and position permutations - and validates each format plus their -//! cross-artifact agreement once, so every read after that is mmap gathers and wire encoding. -//! [`Atlas::tile`] answers one tile request with `SALTILET` envelope bytes and [`Atlas::edges`] one -//! edges request with `SALTILEE` bytes, both ready to send under `application/vnd.hash.saltile-v1`. -//! The manifest document of the Surface v1 bootstrap is [`Atlas::manifest`]. -//! -//! An [`Atlas`] is immutable after open and `Send + Sync`, so hold it in an -//! [`Arc`] across requests for the process lifetime of the generation. Reads are -//! synchronous and CPU-bound (the columns live in mapped memory), so an async transport schedules -//! them on a compute pool - rayon plus `catch_unwind` - never inline on its runtime threads. -//! -//! The route and body vocabulary and the response bytes form pinned public contracts. The one -//! deferral rejects instead of serving wrong bytes: no request body admits a visibility `filter` -//! member, so a request naming one rejects as `invalid-body`. -//! -//! Every assembly path takes a [`VisibilityProof`], the server-held statement of which node rows -//! and which link rows the bound scope may see. Responses compute over the masked view. Delivered -//! sets intersect the node mask, an edge delivers only when the proof admits the edge's link row -//! and both endpoints, and row ingress factors through [`Atlas::resolve`], where decode failure, -//! out-of-universe values, and mask misses collapse to one `None`, so forbidden and nonexistent -//! answer identical bytes. -//! -//! Every surface answers under any proof, the link-bearing ones included. A proof carries a mask -//! per identity domain, so the authorization of a link row is a statement the proof holds rather -//! than something its endpoints imply. Refusals are per row, and an unproven row is absent, so a -//! scope that may see nothing receives a well-formed response that delivers nothing. -//! -//! # Architecture -//! -//! Each domain concept lives in exactly one module. The foundation: `open` is the open pass - map, -//! validate, derive, construct; `column` holds the element-typed column views that validation -//! produces; `grid` is the bucket schedule and its addressing; `secret`, the wire secret; and -//! `error`, the open-failure taxonomy. -//! -//! The domain: -//! -//! - `visibility` carries the proof and its resolution -//! - `codec` the keyed row-id permutation -//! - `density` the public band that resolves one scope's delivery cut -//! - `walk` the schedule-driven point delivery - full-visibility range assembly, the masked -//! delivery chain, and the census -//! - `neighbourhood` the adjacency edge sets and their caps -//! - `colour` the type-colouring resolution -//! - `intern` the wire intern tables -//! - `authorization` the sealed authority tokens -//! - `hydrate` the live store reads behind detail trailers -//! -//! The read surfaces compose those: `tile`, `edges`, `locate`, and `translate` each hold one -//! endpoint's request vocabulary, rejection taxonomy, and assembly, and `manifest` the bootstrap -//! document. Below the [`Atlas`] facade nothing reaches into the whole value: the domain types -//! borrow exactly the columns they read. - -use alloc::sync::Arc; -use std::sync::OnceLock; - -use hash_graph_temporal_versioning::{Timestamp, TransactionTime}; -use hashql_core::id::{Id as _, IdSlice, IdVec}; - -pub use self::{ - cache::scope::VisibilityLimits, delta::staging::EmbeddingWorkflow, locate::LocateLimits, - tile::TileLimits, translate::TranslateLimits, -}; -pub(crate) use self::{ - codec::{Universe, WireRow}, - delta::{ - DeltaCell, DeltaEpoch, DeltaRegister, DeltaSnapshot, PlacementCohort, PlacementError, - consumer::{DeltaConsumer, DeltaPolling}, - staging::StagingArm, - }, - density::{CutOffset, DensityBand, DensityPolicy, ViewOccupancy}, - edges::{EdgesError, EdgesLimits, EdgesRequest}, - error::OpenAtlasError, - hydrate::GraphDatabaseClient, - intern::TableIndex, - locate::{LocateError, LocateRequest}, - manifest::Manifest, - open::OpenOptions, - secret::WireSecret, - tile::{TileError, TileQuery, TileRequest}, - translate::{TranslateError, TranslateRequest, TranslateResponse}, - view::{View, ViewError}, - visibility::VisibilityProof, - walk::ViewCensus, -}; -use self::{grid::Grid, schedule::ScopeSchedule}; -use crate::{ - device::PhysicalDevice, - file::{ - generation::{Generation, GenerationId}, - morton::read::MortonFile, - quad::read::QuadFile, - }, - identity::{BasePosition, Column, EdgeRowId, ImportanceRank, NodeRowId, OntologyRowId}, - math::{Bounds2, Log2, Vec2}, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - salt::{ - adjacency::AdjacencyArchive, - fit::prepare::identity::IdentityTableArchive, - postings::{artifact::PostingsArchive, closure::ClosureMap}, - }, -}; - -pub(crate) mod cache; -mod codec; -mod colour; -pub(crate) mod delta; -mod density; -mod edges; -mod error; -mod grid; -pub(crate) mod hydrate; -mod intern; -mod locate; -mod manifest; -pub(crate) mod neighbourhood; -mod open; -pub(crate) mod schedule; -mod secret; -mod tile; -mod translate; -mod view; -pub(crate) mod visibility; -mod walk; - -pub(crate) mod authorization; -#[cfg(test)] -mod tests; - -/// The variant names one generation serves, in variant-index order. -/// -/// Surface v1 serves exactly `plain`. Routes and manifests take variant names and indices from this -/// constant. -pub(crate) const VARIANTS: [&str; 1] = ["plain"]; - -/// The serving controls in one configurable value: request-validation limits and response-shaping -/// limits. -/// -/// The transport constructs one - flags and environment over the defaults - and the handlers -/// enforce it. Every published manifest limit derives from an enforced value through -/// [`ServeLimits::manifest_limits`], so advertisement and enforcement cannot disagree, and the -/// manifest publishes only some of the controls. The per-endpoint limits types document their -/// defaults, and none of those defaults is a wire constant. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct ServeLimits { - /// The tile endpoint's limits. - pub tile: TileLimits = TileLimits::default(), - /// The edges endpoint's limits. - pub edges: EdgesLimits = EdgesLimits::default(), - /// The locate endpoint's limits. - pub locate: LocateLimits = LocateLimits::default(), - /// The translate endpoint's limits. - pub translate: TranslateLimits = TranslateLimits::default(), -} - -const impl Default for ServeLimits { - fn default() -> Self { - Self { .. } - } -} - -/// One opened generation, ready to answer reads. -/// -/// Opening maps every serving artifact and validates each format plus their cross-artifact -/// agreement once. The value is immutable after that and shared across requests. -/// -/// # Examples -/// -/// The request types are crate-internal, so the example below stands in for a compiled one. -/// -/// ```ignore -/// use std::sync::Arc; -/// -/// use crate::{ -/// integrity::SecretHexBytes, -/// serve::{ -/// CutOffset, GenerationRoot, OpenOptions, TileCoordinate, TileLimits, TileQuery, -/// TileRequest, VisibilityProof, WireSecret, -/// }, -/// }; -/// -/// let root = GenerationRoot::new("/var/atlas/generations")?; -/// let id = root.current()?.expect("a generation is active"); -/// let secret = WireSecret::from( -/// "6ad599a5c17e1fc4d7e2988bd4f3e0367f3c4a35d6dae135f9a1e0efc775ce55" -/// .parse::>()?, -/// ); -/// let atlas = Arc::new(Atlas::open( -/// &root, -/// id, -/// OpenOptions { -/// wire_secret: secret, -/// }, -/// )?); -/// -/// // Authority over the whole corpus, stated at the call site. -/// let proof = VisibilityProof::full_visibility(); -/// let bytes = atlas.tile( -/// &TileRequest { -/// coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, -/// query: TileQuery::default(), -/// }, -/// TileLimits::default(), -/// &proof, -/// CutOffset::ZERO, -/// )?; -/// ``` -#[derive(Debug)] -pub(crate) struct Atlas { - generation: Generation, - /// The server secret this generation opened under. - /// - /// Every wire-facing derivation keys from it, both the row-id codec's round keys at open and - /// the authority token key at router construction. Held for the generation's lifetime, which - /// is the retention its type documents. - wire_secret: WireSecret, - /// The validated bucket schedule and its addressing. - grid: Grid, - quad: QuadFile, - morton: MortonFile, - /// The wire-coordinate column in base order. - points: Column, - /// The row column in base order: the node universe's permutation. - rows: Column, - adjacency: AdjacencyArchive, - /// The endpoint column mapping each edge row to `[source, target]`. - endpoints: Column, - /// The rank column in base order. - ranks: Column, - /// The position permutation in row order. - positions_of_row: Column, - /// The reverse rank permutation, mapping each rank to its base position. - /// - /// The fit pipeline constructs it as the rank column's inverse, so a traversal in rank order - /// visits every base position exactly once, and the scoped schedule's gather reads it - /// instead of sorting the view by rank. The open pass spot-checks the pairing at a bounded - /// sample of roundtrips rather than proving the full inversion. - positions_of_rank: Column, - postings: PostingsArchive, - closure: ClosureMap, - /// The ontology identity table, joining type uuids to ontology rows. - /// - /// Present by construction: a generation whose ids are not store identities fails the open, as - /// do the node and edge tables below. - ontology_ids: IdentityTableArchive, - /// The node identity table, joining node rows to entity identities. - node_ids: IdentityTableArchive, - /// The edge identity table, joining edge rows to link-entity identities. - edge_ids: IdentityTableArchive, - /// The node universe's wire row-id codec, derived at open. - /// - /// The one wire-id domain, since edges cross the wire as link-entity identities. - node_codec: codec::RowCodec, - /// The generation's base row universe, the validated row column's bound. - /// - /// The bound before any delta, which the slot allocator starts past and a delta snapshot - /// widens. A request answering with no snapshot reads this value at every encode and decode. - node_universe: Universe, - /// The wire row-id column in base order. - /// - /// The row column mapped through the node codec once at open, so position-driven gathers - /// (tiles, locate) pay nothing per request. - wire_rows: IdVec>, - /// The tight wire-frame extent of the full point set. - /// - /// Absent iff the generation holds no points. Derived from the world frame: normalization - /// anchors each non-degenerate axis's extremes onto the frame edges and collapses degenerate - /// axes to the centre, so the extent follows without scanning the column. - bounds: Option, - /// The cascade every saturated scope serves, built on first use. - /// - /// A scope schedule is a function of the visible node rows alone, so every scope whose node - /// mask admits the whole corpus builds the same cascade. One copy per generation serves them - /// all. - saturated: OnceLock>, -} - -/// The validated column views. -impl Atlas { - /// Returns the generation identity: the `HEAD` echo, the route echo. - #[inline] - #[must_use] - pub(crate) const fn generation(&self) -> GenerationId { - self.generation.id() - } - - /// Returns the transaction-time point the generation's dataset observed, or [`None`] for a - /// source without temporal axes. - /// - /// A replay of the entity feed from this point covers every store change the fitted - /// artifacts cannot know about. - #[must_use] - pub(crate) fn fitted_at(&self) -> Option> { - self.generation - .repository() - .metadata - .snapshot - .axes - .map(|axes| axes.transaction_time) - } - - /// Views the server secret this generation opened under. - pub(crate) const fn wire_secret(&self) -> &WireSecret { - &self.wire_secret - } - - /// Returns the generation's base row universe, the bound before any delta. - #[must_use] - pub(crate) const fn node_universe(&self) -> Universe { - self.node_universe - } - - /// Returns the generation's ontology row universe, the tabulated types' bound. - #[must_use] - pub(crate) fn ontology_universe(&self) -> Universe { - Universe::new(OntologyRowId::from_u64(self.ontology_ids.len())) - } - - /// Returns the generation's edge row universe, the fitted edges' bound before any delta. - #[must_use] - pub(crate) fn edge_universe(&self) -> Universe { - Universe::new(EdgeRowId::from_u64(self.edge_ids.len())) - } - - /// Opens the generation's publish path for placing arrivals online. - /// - /// Returns `Ok(None)` for a generation that placed rows by landmark baseline: it promises no - /// publish path, and its arrivals stage until a refit. - /// - /// # Errors - /// - /// Returns [`delta::PlacementError`] when the generation stages a projector checkpoint whose - /// publish path does not reopen and certify. Each refusal logs its own line at the site. - pub(crate) fn arrival_placer( - &self, - device: PhysicalDevice, - ) -> Result, delta::PlacementError> { - delta::Placer::open(&self.generation, device) - } - - /// Configures the delivery-cut policy over this generation's schedule, aiming for `band`. - /// - /// [`None`] for the schedules no offset deepens - a terminal root, or a schedule already at the - /// key width - where every scope serves the recorded cut, [`CutOffset::ZERO`]. - pub(crate) fn density_policy(&self, band: DensityBand) -> Option { - DensityPolicy::new( - band, - Log2::new(self.grid.span_log2()).expect("the validated schedule's span fits the key"), - self.grid.max_tile_depth(), - ) - .ok() - } - - /// Views the wire-coordinate column in base order. - fn positions(&self) -> &IdSlice { - self.points.view() - } - - /// Views the row column in base order. - fn row_ids(&self) -> &IdSlice { - self.rows.view() - } - - /// Views the wire row-id column in base order. - const fn wire_rows(&self) -> &IdSlice> { - self.wire_rows.as_slice() - } - - /// Views the endpoint column: edge row to `[source, target]`. - fn endpoint_pairs(&self) -> &IdSlice { - self.endpoints.view() - } - - /// Views the position permutation in row order. - fn positions_of_row(&self) -> &IdSlice { - self.positions_of_row.view() - } - - /// Returns the cascade a saturated scope serves, building it on first use. - /// - /// Concurrent first callers wait for a single construction rather than duplicating it. The - /// build gathers under the full-visibility proof, which admits exactly the fitted rows a - /// saturated node mask admits, and under the empty cohort, because the memo outlives any - /// one entry's arrivals. - fn saturated_scope_schedule(&self) -> &Arc { - self.saturated.get_or_init(|| { - Arc::new(ScopeSchedule::of( - self, - &VisibilityProof::full_visibility(), - delta::PlacementCohort::EMPTY, - )) - }) - } - - /// Returns the saturated cascade only when some resolution already built it. - /// - /// The cache's weigher asks whether an entry's schedule is the memo, and a sharer took its - /// `Arc` from the memo itself, so an unbuilt memo already answers no. Recognition therefore - /// never builds: forcing the full-corpus construction here would bill the first small - /// scope's resolution for the whole corpus. [`Self::saturated_scope_schedule`] stays the - /// building accessor. - fn saturated_scope_schedule_if_built(&self) -> Option<&Arc> { - self.saturated.get() - } -} - -impl delta::IdentityTables for Atlas { - fn node_row_of(&self, id: ArchivedEntityId) -> Option { - self.node_ids.row_of(id) - } - - fn edge_row_of(&self, id: ArchivedEntityId) -> Option { - self.edge_ids.row_of(id) - } - - fn ontology_row_of(&self, id: ArchivedOntologyTypeUuid) -> Option { - self.ontology_ids.row_of(id) - } -} diff --git a/libs/@local/graph/atlas/src/serve/neighbourhood.rs b/libs/@local/graph/atlas/src/serve/neighbourhood.rs deleted file mode 100644 index b3720f794b1..00000000000 --- a/libs/@local/graph/atlas/src/serve/neighbourhood.rs +++ /dev/null @@ -1,846 +0,0 @@ -//! Edge sets over delivered or resolved rows. -//! -//! Serving delivers two edge-set shapes. The ego graph is every edge at a source whose other -//! endpoint is visible: [`Neighbourhood::incident`] walks its fitted half and -//! [`Neighbourhood::incident_links`] folds the entry cohort's post-fit half. [`EdgeSet`] answers -//! the edges response's delivered set, which folds the delivered rows' induced fitted edges and -//! the entry cohort's admitted post-fit links into one capped, identity-ordered selection. The -//! fitted gathers walk the adjacency's outgoing runs and collect each qualifying edge exactly -//! once, because an edge occupies exactly one outgoing slot and a self-loop's one endpoint is -//! both its source and its target. -//! -//! An edge delivers only when the proof admits the edge's link row and both of its endpoints, and -//! when the ingress withdrawal snapshot withdraws none of the three. Both shapes reach their -//! fitted candidates through [`Neighbourhood::edge`], which answers [`None`] for an edge the proof -//! withholds or the snapshot withdraws, so every collected edge carries its delivery proof in the -//! type and neither walk states either rule a second time. A delta link has no generation row, and -//! it qualifies through the proof's identity set, the capture's retention, and delivery of both -//! endpoints. - -use alloc::collections::BinaryHeap; -use core::cmp::Ordering; - -use hashql_core::{ - collections::fast_hash_map, - id::{IdSlice, IdVec, bit_vec::DenseBitSet}, -}; - -use super::{ - Atlas, WireRow, - codec::{RowCodec, Universe}, - delta::DeltaSnapshot, - hydrate::EdgeSlot, - schedule::ArrivalIndex, - view::View, - visibility::{VisibilityProof, VisibleEdge}, -}; -use crate::{ - identity::{BasePosition, EdgeRowId, ImportanceRank, NodeRowId}, - postgres::id::ArchivedEntityId, - salt::{adjacency::AdjacencyArchive, fit::prepare::identity::IdentityTableArchive}, -}; - -/// The wire columns' row ids for one qualifying edge during assembly. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct DeliveredEdge { - /// The edge row id, proven deliverable under the proof that collected this value. - pub row: VisibleEdge, - /// The source node row id. - pub source: NodeRowId, - /// The target node row id. - pub target: NodeRowId, -} - -impl DeliveredEdge { - /// Returns the endpoint opposite `row`. - /// - /// A self-loop's partner is `row` itself. - pub(super) fn partner_of(self, row: NodeRowId) -> NodeRowId { - if self.source == row { - self.target - } else { - self.source - } - } -} - -/// One delta-edge endpoint, resolved into the domain that delivers it. -/// -/// A fitted endpoint is a generation node row in the response's delivered set, and an arrival -/// endpoint a delivered placed arrival. The arrival keeps its identity in the vessel because -/// arrival ranking orders by identity bytes where fitted ranking reads the importance column. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum DeltaEndpoint { - /// A delivered generation row. - Fitted(NodeRowId), - /// A delivered placed arrival, on its cohort slot. - Arrival { - /// The slot the endpoint column encodes. - slot: NodeRowId, - /// The arrival's identity, its place in the arrival rank order. - identity: ArchivedEntityId, - }, -} - -impl DeltaEndpoint { - /// Returns the row id the endpoint column encodes. - pub(super) const fn row(self) -> NodeRowId { - match self { - Self::Fitted(row) => row, - Self::Arrival { slot, .. } => slot, - } - } -} - -/// One delta edge's resolved endpoints during assembly. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct DeltaEdge { - /// The left attachment's delivered endpoint. - pub source: DeltaEndpoint, - /// The right attachment's delivered endpoint. - pub target: DeltaEndpoint, -} - -/// One qualifying edge during assembly, from either serving domain. -/// -/// A fitted edge is a generation row that the structural walks collect with its delivery proof -/// in the vessel, and a delta edge a cohort-published post-fit link that the proof's identity -/// set admits, its endpoints resolved against the response's own delivered sets. The wire -/// columns hold both shapes in one delivery order. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum ServedEdge { - /// A generation edge row, proven deliverable. - Fitted(DeliveredEdge), - /// A cohort-published post-fit link between two delivered endpoints. - Delta(DeltaEdge), -} - -impl ServedEdge { - /// Returns the node row the `EDGE_SOURCES` column encodes. - const fn source_row(self) -> NodeRowId { - match self { - Self::Fitted(edge) => edge.source, - Self::Delta(edge) => edge.source.row(), - } - } - - /// Returns the node row the `EDGE_TARGETS` column encodes. - const fn target_row(self) -> NodeRowId { - match self { - Self::Fitted(edge) => edge.target, - Self::Delta(edge) => edge.target.row(), - } - } - - /// Returns the hydration key behind the edge. - const fn origin(self) -> EdgeOrigin { - match self { - Self::Fitted(edge) => EdgeOrigin::Fitted(edge.row.get()), - Self::Delta(_) => EdgeOrigin::Delta, - } - } - - /// Views a delivered edge's endpoints in the vocabulary both serving domains share. - pub(crate) const fn endpoints(self) -> [DeltaEndpoint; 2] { - match self { - Self::Fitted(edge) => [ - DeltaEndpoint::Fitted(edge.source), - DeltaEndpoint::Fitted(edge.target), - ], - Self::Delta(edge) => [edge.source, edge.target], - } - } - - /// Returns the endpoint opposite the source, in the vocabulary both serving domains share. - /// - /// A self-loop's partner is the source itself, in either domain. - pub(crate) fn opposite_endpoint(self, source: NodeRowId) -> DeltaEndpoint { - match self { - Self::Fitted(edge) => DeltaEndpoint::Fitted(edge.partner_of(source)), - Self::Delta(edge) => { - if edge.source.row() == source { - edge.target - } else { - edge.source - } - } - } - } -} - -/// One endpoint's truncation rank, over the fitted-plus-delta union. -/// -/// The derived order places every fitted rank before every arrival rank, because a placed -/// arrival ranks past every generation row, and it orders arrivals by identity bytes, the -/// arrival order's own law. Larger values are less prominent in both arms. -#[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord)] -enum EndpointRank { - /// A generation row's importance rank. - Fitted(ImportanceRank), - /// A placed arrival's identity. - Arrival(ArchivedEntityId), -} - -/// The provenance behind one delivered edge: the hydration key. -/// -/// Both origins resolve their display in process, captured display first. A fitted edge falls -/// back to the generation payload its row addresses while a delta edge holds no generation row -/// and always answers from its captured display keyed by the identity the `EDGE_IDS` column -/// already carries. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum EdgeOrigin { - /// A generation edge row. - Fitted(EdgeRowId), - /// A cohort-published delta link. - Delta, -} - -/// The delivered edges in column form, every column in lockstep. -/// -/// One constructor writes every column in one pass over the delivered edge set, so the columns -/// cover the same edges by construction and delivery order is the input's own. The endpoint -/// columns speak wire row ids. The identity column is the wire's `EDGE_IDS`, and the origin -/// column is the hydration key behind it. -#[derive(Debug)] -pub(crate) struct EdgeColumns { - /// The `EDGE_SOURCES` column: node row ids in wire form, edge order. - sources: IdVec>, - /// The `EDGE_TARGETS` column: node row ids in wire form, edge order. - targets: IdVec>, - /// The `EDGE_IDS` column: link-entity identities, edge order. - ids: IdVec, - /// The provenance behind `ids`, edge order: the hydration key. - origins: IdVec, -} - -impl EdgeColumns { - /// Gathers the delivered edges into their column form. - /// - /// `universe` is the caller's accepted row bound, read at every endpoint encode. - /// - /// Caller requirement: `universe` admits every endpoint row, so a view serving cohort slots - /// passes its entry universe rather than the generation's fitted bound. A fitted row - /// encodes to the same bytes under either bound. - pub(super) fn of( - codec: &RowCodec, - universe: Universe, - edges: &[(ServedEdge, ArchivedEntityId)], - ) -> Self { - let mut columns = Self { - sources: IdVec::with_capacity(edges.len()), - targets: IdVec::with_capacity(edges.len()), - ids: IdVec::with_capacity(edges.len()), - origins: IdVec::with_capacity(edges.len()), - }; - - for &(edge, id) in edges { - columns - .sources - .push(codec.encode(edge.source_row(), universe)); - columns - .targets - .push(codec.encode(edge.target_row(), universe)); - columns.origins.push(edge.origin()); - columns.ids.push(id); - } - - columns - } - - /// Returns the delivered edge count. - pub(crate) const fn count(&self) -> usize { - self.ids.len() - } - - /// Views the `EDGE_SOURCES` column, edge order. - pub(crate) const fn sources(&self) -> &IdSlice> { - &self.sources - } - - /// Views the `EDGE_TARGETS` column, edge order. - pub(crate) const fn targets(&self) -> &IdSlice> { - &self.targets - } - - /// Views the `EDGE_IDS` column, edge order. - pub(crate) const fn ids(&self) -> &IdSlice { - &self.ids - } - - /// Views the per-edge provenance, edge order. - pub(crate) const fn origins(&self) -> &IdSlice { - &self.origins - } -} -#[cfg(test)] // The serve tests pin expected edge columns literally. -impl EdgeColumns { - /// Builds columns from one literal wire-form triple per edge: source, target, identity. - /// - /// The triple shape keeps the columns in lockstep at the call site, and the internal row - /// column numbers the edges in order, because a pinned response never hydrates. - pub(crate) fn pinned(edges: impl IntoIterator) -> Self { - use hashql_core::id::Id as _; - - let mut columns = Self { - sources: IdVec::new(), - targets: IdVec::new(), - ids: IdVec::new(), - origins: IdVec::new(), - }; - for (source, target, id) in edges { - let row = u32::try_from(columns.origins.len()).expect("pinned fixtures are small"); - columns.sources.push(WireRow::pinned(source)); - columns.targets.push(WireRow::pinned(target)); - columns.ids.push(id); - columns - .origins - .push(EdgeOrigin::Fitted(EdgeRowId::from_u32(row))); - } - - columns - } -} - -/// One generation's edge columns under one visibility proof. -/// -/// The value borrows the opened generation, so construction is free per request and the edge-set -/// constructions take no column parameters. -#[derive(Debug, Copy, Clone)] -pub(super) struct Neighbourhood<'atlas> { - adjacency: &'atlas AdjacencyArchive, - endpoints: &'atlas IdSlice, - edge_ids: &'atlas IdentityTableArchive, - ranks: &'atlas IdSlice, - positions_of_row: &'atlas IdSlice, - node_universe: Universe, - proof: &'atlas VisibilityProof, - /// The request's ingress withdrawal snapshot, absent before the first publication. - delta: Option<&'atlas DeltaSnapshot>, -} - -impl<'atlas> Neighbourhood<'atlas> { - /// Binds the generation's edge columns to `proof`, with `delta` as the request's ingress - /// withdrawal capture. - pub(super) fn of( - atlas: &'atlas Atlas, - proof: &'atlas VisibilityProof, - delta: Option<&'atlas DeltaSnapshot>, - ) -> Self { - Self { - adjacency: &atlas.adjacency, - endpoints: atlas.endpoints.view(), - edge_ids: &atlas.edge_ids, - ranks: atlas.ranks.view(), - positions_of_row: atlas.positions_of_row.view(), - node_universe: atlas.node_universe, - proof, - delta, - } - } - - /// Collects the source's incident edges the proof delivers, paired with their link-entity - /// identities, in no particular order. - /// - /// A self-loop occupies one slot in each direction's run. The walk skips its incoming - /// appearance as the duplicate, so it collects every incident edge exactly once. An edge the - /// proof withholds, for a hidden partner or for a hidden link row, drops before any selection, - /// so a response's cardinality is a function of the masked view. - /// - /// # Panics - /// - /// This panics when `source` lies outside the adjacency's node domain, which resolution rules - /// out. - pub(super) fn incident( - &self, - node: NodeRowId, - ) -> impl IntoIterator { - let outgoing = self - .adjacency - .outgoing(node) - .expect("resolved sources lie inside the adjacency's node domain"); - let incoming = self - .adjacency - .incoming(node) - .expect("resolved sources lie inside the adjacency's node domain"); - - let incident = outgoing.iter().chain( - incoming - .iter() - .filter(move |&edge| self.endpoints[edge][0] != node), - ); - - incident.filter_map(move |edge| { - let delivered = self.edge(edge)?; - Some((delivered, self.edge_identity(delivered.row))) - }) - } - - /// Collects the cohort links incident to `source` that the request delivers, paired with - /// their link-entity identities, in no particular order. - /// - /// The ego graph's post-fit half, under the same partner rule as the fitted walk: the other - /// endpoint must be visible, never delivered-in-tiles. A link qualifies when the proof's - /// identity set admits it and the ingress capture does not withdraw it, with each endpoint - /// filtered under the same capture. Fitted endpoints serve while the proof admits their - /// rows, and an arrival endpoint while the view's arrival table holds its identity, which - /// carries the proof's slot admission and the cohort's retention together. - pub(super) fn incident_links( - &self, - view: &View<'_>, - source: NodeRowId, - ) -> Vec<(DeltaEdge, ArchivedEntityId)> { - let cohort = view.cohort(); - let ingress = self.delta; - - let endpoint = |row: NodeRowId| { - if self.node_universe.contains(row) { - (self.proof.contains(row) - && !ingress.is_some_and(|delta| delta.withdraws_node(row))) - .then_some(DeltaEndpoint::Fitted(row)) - } else { - let (identity, _) = cohort.node_at(row)?; - if ingress.is_some_and(|delta| delta.withdraws(identity)) { - return None; - } - - let arrivals = view.arrivals(); - let index = arrivals.partition_point(|entry| entry.identity < identity); - (arrivals.get(index)?.identity == identity).then_some(DeltaEndpoint::Arrival { - slot: row, - identity, - }) - } - }; - - cohort - .edges() - .filter(|&(_, link)| link.source == source || link.target == source) - .filter_map(|(identity, link)| { - if !self.proof.admits_delta_link(identity) - || ingress.is_some_and(|delta| delta.withdraws(identity)) - { - return None; - } - let (source, target) = (endpoint(link.source)?, endpoint(link.target)?); - - Some((DeltaEdge { source, target }, identity)) - }) - .collect() - } - - /// Offers every induced fitted edge over `delivered` to the cap. - /// - /// The walk reads each delivered row's outgoing run and offers each edge whose other endpoint - /// is delivered and whose delivery [`Neighbourhood::edge`] proves. The set membership test - /// answers the induced-subgraph question - which rows this response draws edges between - and - /// never stands in for the delivery rule, which the proof answers as the walk reads each - /// candidate. - /// - /// Caller requirement: `delivered` is already intersected with the visibility proof. - fn offer_induced(&self, delivered: &DenseBitSet, cap: &mut RankCap) { - for row in delivered { - let row_rank = self.rank_of_row(row); - // One delivered endpoint suffices for exclusion: every edge at this row keys at or - // past the row's own rank. The skip waits for a recorded truncation, per the cap's - // caller requirement. - if cap.truncated && cap.excludes(EndpointRank::Fitted(row_rank)) { - continue; - } - - let outgoing = self - .adjacency - .outgoing(row) - .expect("delivered rows lie inside the adjacency's node domain"); - - for edge in outgoing.iter() { - let [_, target] = self.endpoints[edge]; - if !delivered.contains(target) { - continue; - } - - let worse = EndpointRank::Fitted(row_rank.max(self.rank_of_row(target))); - if cap.truncated && cap.excludes(worse) { - continue; - } - let Some(delivered_edge) = self.edge(edge) else { - continue; - }; - - cap.offer( - (worse, self.edge_identity(delivered_edge.row)), - ServedEdge::Fitted(delivered_edge), - ); - } - } - } - - /// Offers every admitted cohort link between delivered endpoints to the cap. - /// - /// A delta link serves when the proof's identity set admits it, the ingress capture does not - /// withdraw it, and the response's delivered sets hold both endpoints. Publication hands the - /// endpoints as rows. A row inside the fitted universe qualifies through the delivered rows - /// after subtraction, and an allocated row qualifies through the delivered arrivals whose - /// identities the capture keeps. Endpoint withdrawal therefore kills an edge through - /// the same delivered-set rule that kills fitted edges, one domain over. An endpoint the - /// view cannot deliver refuses the edge whole, so a link attaching an undelivered entity - /// serves nothing. - fn offer_delta_links(&self, view: &View<'_>, bounds: &DeliveredBounds, cap: &mut RankCap) { - let cohort = view.cohort(); - let ingress = self.delta; - - // A selection the fitted edges already filled admits no arrival edge: `EndpointRank` - // orders every `Arrival` below every `Fitted`. - let arrivals = (!cap.truncated).then(|| { - let table = view.arrivals(); - let mut map = fast_hash_map(); - for index in &bounds.arrivals { - let row = &table[index]; - if ingress.is_some_and(|delta| delta.withdraws(row.identity)) { - continue; - } - let Some(node) = cohort.node(row.identity) else { - continue; - }; - - map.insert(node.id, row.identity); - } - - map - }); - - let endpoint = |row: NodeRowId| { - if self.node_universe.contains(row) { - bounds - .rows - .contains(row) - .then_some(DeltaEndpoint::Fitted(row)) - } else { - arrivals - .as_ref() - .and_then(|map| map.get(&row)) - .map(|&identity| DeltaEndpoint::Arrival { - slot: row, - identity, - }) - } - }; - - for (identity, link) in cohort.edges() { - if !self.proof.admits_delta_link(identity) { - continue; - } - if ingress.is_some_and(|delta| delta.withdraws(identity)) { - continue; - } - let (Some(source), Some(target)) = (endpoint(link.source), endpoint(link.target)) - else { - continue; - }; - - cap.offer( - ( - self.endpoint_rank(source).max(self.endpoint_rank(target)), - identity, - ), - ServedEdge::Delta(DeltaEdge { source, target }), - ); - } - } - - /// Returns an edge row's link-entity identity. - /// - /// Generation-baked. The `EDGE_IDS` columns deliver exactly these identity bytes. - /// - /// # Panics - /// - /// This panics when the identity table contradicts the adjacency's edge domain, which open's - /// cross-artifact validation rules out. - pub(super) fn edge_identity(&self, row: VisibleEdge) -> ArchivedEntityId { - self.edge_ids - .id(row.get()) - .expect("open validated the identity rows against the adjacency's edges") - } - - /// Returns a delta-edge endpoint's rank in the union order. - const fn endpoint_rank(&self, endpoint: DeltaEndpoint) -> EndpointRank { - match endpoint { - DeltaEndpoint::Fitted(row) => EndpointRank::Fitted(self.rank_of_row(row)), - DeltaEndpoint::Arrival { identity, .. } => EndpointRank::Arrival(identity), - } - } - - /// Returns a node row's importance rank through the position permutation. - const fn rank_of_row(&self, row: NodeRowId) -> ImportanceRank { - let position = self.positions_of_row[row]; - self.ranks[position] - } - - /// Reads edge `row`'s wire-column ids off the endpoint column, when the request delivers it. - /// - /// [`None`] when the proof withholds the link row or either endpoint, and when the ingress - /// snapshot withdraws any of the three. A withdrawn link is a tombstone whose endpoints - /// survive, and a withdrawn endpoint kills every edge at it on the next request. The type - /// admits no value for an edge no response may name, which is why both walks filter by - /// constructing. - fn edge(&self, row: EdgeRowId) -> Option { - let [source, target] = self.endpoints[row]; - - if let Some(delta) = self.delta - && (delta.withdraws_edge(row) - || delta.withdraws_node(source) - || delta.withdraws_node(target)) - { - return None; - } - - Some(DeliveredEdge { - row: self.proof.verify_edge(row, source, target)?, - source, - target, - }) - } -} - -/// The listed tiles' delivered sets, one per serving domain. -/// -/// The rows are the fitted deliveries, as the tile route renders them. The arrivals are the -/// delivered placed arrivals as view arrival-table indices, the identity-domain half a delta -/// edge's endpoints resolve against. -#[derive(Debug)] -pub(super) struct DeliveredBounds { - /// The delivered fitted rows. - pub rows: DenseBitSet, - /// The delivered placed arrivals, as arrival-table indices. - pub arrivals: DenseBitSet, -} - -/// The edges response's delivered edge set: the fitted-plus-delta union, selected and ordered. -/// -/// One constructor folds both serving domains. The delivered rows' induced fitted edges and the -/// entry cohort's admitted post-fit links each qualify through their own domain's delivery rule, -/// compete at one rank-ordered cap, and leave in ascending link-entity identity order, so the -/// fitted-delta distinction keeps one home and every later consumer takes the folded set whole. -#[derive(Debug)] -pub(super) struct EdgeSet { - /// Whether every qualifying edge is in the set. - complete: bool, - /// The selected edges, ascending link-entity identity bytes. - edges: Vec<(ServedEdge, ArchivedEntityId)>, -} - -impl EdgeSet { - /// Folds both serving domains' qualifying edges into one selected, identity-ordered set. - /// - /// `bounds` arrives as the tile route read it, and the delivery rules apply here: the - /// delivered rows intersect the proof and drop the ingress capture's withdrawn rows, so the - /// fitted bounding set is exactly what the tiles rendered. The fitted domain then offers - /// every adjacency edge between delivered rows that the proof delivers, and the delta domain - /// offers every cohort link the proof's identity set admits, the capture retains, and the - /// delivered sets resolve at both endpoints. - /// - /// Every offer competes under one rank-ordered cap on its worse endpoint's rank, ties on - /// identity bytes - an edge is only as prominent as its less-prominent endpoint. The kept - /// set equals the full union sorted by that key and truncated to `cap`, and - /// [`EdgeSet::complete`] reports whether the cap truncated a qualifying edge. - pub(super) fn of( - atlas: &Atlas, - view: &View<'_>, - mut bounds: DeliveredBounds, - cap: usize, - ) -> Self { - let neighbourhood = Neighbourhood::of(atlas, view.proof(), view.delta()); - - // Both branches of the union already gather visible rows alone - a scope cascade holds - // only what its proof admitted, and the corpus walk answers only an operator view. The - // intersection is what discharges the fitted walk's caller requirement rather than a - // second derivation of it, and it is the guard if either branch ever widens. - neighbourhood.proof.intersect(&mut bounds.rows); - - // The bounding set is what tiles rendered, and tiles subtract at admission, so the - // withdrawn rows leave here too. `Neighbourhood::edge` states the edge rule itself. - if let Some(delta) = neighbourhood.delta { - for row in delta.withdrawn_node_rows() { - bounds.rows.remove(row); - } - } - - let mut cap = RankCap::new(cap); - neighbourhood.offer_induced(&bounds.rows, &mut cap); - neighbourhood.offer_delta_links(view, &bounds, &mut cap); - cap.into_set() - } - - /// Whether every qualifying edge is in the set. - pub(super) const fn complete(&self) -> bool { - self.complete - } - - /// Views the selected edges, ascending link-entity identity bytes. - pub(super) const fn edges(&self) -> &[(ServedEdge, ArchivedEntityId)] { - &self.edges - } -} - -/// The rank-ordered cap holding the `cap` best candidates offered so far, with every truncation -/// recorded. -/// -/// A candidate falls on arrival when its key loses to the kept worst, and a kept candidate -/// falls by eviction when a better arrival fills the selection. Either way the truncation is -/// recorded, because a response that silently skipped a qualifying edge must never report -/// itself complete. -#[derive(Debug)] -struct RankCap { - /// The number of candidates the selection keeps. - cap: usize, - /// The kept candidates in a max-heap on the key, whose root is the eviction candidate. - kept: BinaryHeap, - /// Whether any qualifying candidate fell, on arrival or by eviction. - truncated: bool, -} - -impl RankCap { - /// An empty selection admitting `cap` candidates. - const fn new(cap: usize) -> Self { - Self { - cap, - kept: BinaryHeap::new(), - truncated: false, - } - } - - /// Offers one qualifying candidate, recording the truncation when the selection is full. - /// - /// A full selection truncates either way: the offer falls when its key loses to the kept - /// worst, and the kept worst falls by eviction when the offer wins. Under a zero cap every - /// offer truncates. Equal keys cannot arrive, because distinct link identities make the key a - /// total order. - fn offer(&mut self, key: (EndpointRank, ArchivedEntityId), edge: ServedEdge) { - if self.kept.len() < self.cap { - self.kept.push(Candidate { key, edge }); - return; - } - - self.truncated = true; - if let Some(mut worst) = self.kept.peek_mut() - && key < worst.key - { - *worst = Candidate { key, edge }; - } - } - - /// Whether `rank` alone already loses the selection, before the candidate is priced. - /// - /// True when the selection is full and `rank` strictly loses to the kept worst's rank. A - /// rank equal to the worst's never suffices, because the identity tie-break can still admit - /// the candidate. - /// - /// Caller requirement: skip a candidate on this answer only after a truncation is recorded, - /// because completeness counts qualifying candidates alone and a skipped candidate is never - /// checked for qualification. - fn excludes(&self, rank: EndpointRank) -> bool { - self.kept.len() == self.cap && self.kept.peek().is_none_or(|worst| rank > worst.key.0) - } - - /// Closes the selection into the delivered set, sorted into delivery order. - fn into_set(self) -> EdgeSet { - let mut edges: Vec<(ServedEdge, ArchivedEntityId)> = self - .kept - .into_iter() - .map(|candidate| (candidate.edge, candidate.key.1)) - .collect(); - // Truncation ties and the delivery sort both compare identity bytes, so nothing the - // response exposes orders by internal id. - edges.sort_unstable_by_key(|&(_, id)| id); - - EdgeSet { - complete: !self.truncated, - edges, - } - } -} - -/// One offered candidate, carrying its selection key and the edge it delivers. -#[derive(Debug)] -struct Candidate { - /// The worse endpoint's rank, with ties on link-entity identity bytes. - key: (EndpointRank, ArchivedEntityId), - /// The edge the key selects. - edge: ServedEdge, -} - -// The comparisons read the key alone, so the heap's order ignores the carried edge, and the -// distinct link identities inside the key make the order total. -impl PartialEq for Candidate { - fn eq(&self, other: &Self) -> bool { - self.key == other.key - } -} - -impl Eq for Candidate {} - -impl PartialOrd for Candidate { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -impl Ord for Candidate { - fn cmp(&self, other: &Self) -> Ordering { - self.key.cmp(&other.key) - } -} - -#[cfg(test)] -mod tests { - use hashql_core::id::Id as _; - - use super::{DeltaEdge, DeltaEndpoint, EndpointRank, RankCap, ServedEdge}; - use crate::{ - identity::{ImportanceRank, NodeRowId}, - postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedWebId}, - }; - - /// The identity-table key `seed` spells. - fn identity(seed: u8) -> ArchivedEntityId { - ArchivedEntityId { - web_id: ArchivedWebId::from_bytes([seed; 16]), - entity_uuid: ArchivedEntityUuid::from_bytes([seed; 16]), - } - } - - /// A candidate edge whose fate the selection key alone decides. - fn edge() -> ServedEdge { - ServedEdge::Delta(DeltaEdge { - source: DeltaEndpoint::Fitted(NodeRowId::from_u32(0)), - target: DeltaEndpoint::Fitted(NodeRowId::from_u32(1)), - }) - } - - /// A full selection never excludes on rank equality, and the identity tie-break the - /// strictness protects can then evict the kept worst. - #[test] - fn a_rank_tie_never_excludes_before_the_identity_prices_it() { - let tied = EndpointRank::Fitted(ImportanceRank::from_u32(7)); - let better = EndpointRank::Fitted(ImportanceRank::from_u32(3)); - let worse = EndpointRank::Fitted(ImportanceRank::from_u32(8)); - - let mut cap = RankCap::new(1); - cap.offer((tied, identity(9)), edge()); - - assert!( - !cap.excludes(tied), - "an equal rank prices the identity tie-break", - ); - assert!(!cap.excludes(better)); - assert!(cap.excludes(worse)); - - cap.offer((tied, identity(1)), edge()); - let set = cap.into_set(); - assert_eq!( - set.edges().iter().map(|&(_, id)| id).collect::>(), - vec![identity(1)], - "the equal-ranked better identity evicts the kept worst", - ); - assert!(!set.complete(), "the eviction records the truncation"); - } -} diff --git a/libs/@local/graph/atlas/src/serve/open.rs b/libs/@local/graph/atlas/src/serve/open.rs deleted file mode 100644 index ef8208e5817..00000000000 --- a/libs/@local/graph/atlas/src/serve/open.rs +++ /dev/null @@ -1,498 +0,0 @@ -//! The open pass. -//! -//! Mapping a generation's serving artifacts and validating each format plus their cross-artifact -//! agreement once. Every read after it therefore trusts its views. The pass is one linear -//! derivation. It verifies each published file against the digest the metadata document records, -//! maps and types each artifact and proves the artifacts agree on their shared domains. It then -//! derives the serving state (the schedule, the wire codec and its encoded row column, the frame -//! extent) and constructs the [`Atlas`] whole - no half-initialized value exists at any point. - -use std::sync::OnceLock; - -use hashql_core::id::Id; -use zerocopy::{FromBytes, KnownLayout}; - -use super::{ - Atlas, - codec::{NODE_LABEL, RowCodec, Universe}, - error::{ArrayKind, IdentityDomain, OpenAtlasError}, - grid::Grid, - secret::WireSecret, -}; -use crate::{ - file::{ - array::{ArrayFile, ColumnScalar}, - generation::{Generation, GenerationId, GenerationRoot, OpenError}, - identity::{Key, Row, read::IdentityFile}, - morton::read::MortonFile, - postings::read::PostingsFile, - quad::read::QuadFile, - repository::{Artifact, Binding}, - sprs::read::SprsFile, - }, - identity::{BasePosition, Column, EdgeRowId, ImportanceRank, NodeRowId, OntologyRowId}, - math::{Bounds2, Vec2}, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - salt::{ - adjacency::AdjacencyArchive, - fit::prepare::identity::IdentityTableArchive, - postings::{artifact::PostingsArchive, closure::ClosureMap}, - }, -}; - -/// The options one serving open takes. -/// -/// Configuration travels as a struct, never constants or bare parameters. The struct has no -/// default. Every open names its secret explicitly. A deployment therefore serves only under key -/// material it configured. -#[derive(Debug, Clone)] -pub(crate) struct OpenOptions { - /// The server secret behind the wire row-id codec. - /// - /// The keyed permutation derives from it per generation at open. - /// - /// Operator contract, unenforced by any binding: the secret must not change for a generation - /// that has ever served. Nothing fingerprints the secret. Reopening the same generation under - /// a different value therefore re-keys every wire id while client cache identity - /// (authorization context, generation, route, canonical query) stays constant. A secret - /// change therefore requires a generation rotation and application-cache invalidation. - pub wire_secret: WireSecret, -} - -/// One generation's serving artifacts, each mapped and validated against its own format. -/// -/// The columns share one base order and the archives share the domains that order indexes. A -/// value of this type therefore holds artifacts that are individually well-formed and not yet known -/// to agree with one another. [`Artifacts::agree`] is that second proof, and the serving state -/// derives only after it holds. -#[derive(Debug)] -struct Artifacts { - quad: QuadFile, - morton: MortonFile, - /// The wire-coordinate column in base order. - points: Column, - /// The row column in base order: the node universe's permutation. - rows: Column, - adjacency: AdjacencyArchive, - /// The endpoint column mapping each edge row to `[source, target]`. - endpoints: Column, - /// The rank column in base order. - ranks: Column, - /// The position permutation in row order. - positions_of_row: Column, - /// The reverse rank permutation, mapping each rank to its base position. - positions_of_rank: Column, - postings: PostingsArchive, - /// The ontology identity table, joining type uuids to ontology rows. - ontology_ids: IdentityTableArchive, - /// The node identity table, joining node rows to entity identities. - node_ids: IdentityTableArchive, - /// The edge identity table, joining edge rows to link-entity identities. - edge_ids: IdentityTableArchive, -} - -impl Artifacts { - /// Maps every serving artifact of `generation` and validates each format once. - /// - /// # Errors - /// - /// Returns a per-artifact variant when an artifact fails its format's validation, and - /// [`OpenAtlasError::Shape`] when an artifact holds the wrong element type or shape. - fn open(generation: &Generation) -> Result { - let files = &generation.repository().files; - - let quad = QuadFile::open(files.quad.file().verify(generation)?)?; - let morton = MortonFile::open(files.morton.file().verify(generation)?)?; - let points: Column = - open_column(generation, &files.wire_coordinates, ArrayKind::Coordinates)?; - let rows: Column = - open_column(generation, &files.row_of_position, ArrayKind::Rows)?; - let endpoints: Column = - open_column(generation, &files.edge_endpoints, ArrayKind::Endpoints)?; - let ranks: Column = - open_column(generation, &files.rank_of_position, ArrayKind::Ranks)?; - let positions_of_row: Column = - open_column(generation, &files.position_of_row, ArrayKind::Positions)?; - let positions_of_rank: Column = open_column( - generation, - &files.position_of_rank, - ArrayKind::RankPositions, - )?; - let adjacency = - AdjacencyArchive::new(SprsFile::open(files.adjacency.file().verify(generation)?)?)?; - let postings = PostingsArchive::new(PostingsFile::open( - files.postings.file().verify(generation)?, - )?)?; - let ontology_ids = open_identities( - generation, - &files.ontology_identities, - IdentityDomain::Ontology, - )?; - let node_ids = open_identities(generation, &files.node_identities, IdentityDomain::Node)?; - let edge_ids = open_identities(generation, &files.edge_identities, IdentityDomain::Edge)?; - - Ok(Self { - quad, - morton, - points, - rows, - adjacency, - endpoints, - ranks, - positions_of_row, - positions_of_rank, - postings, - ontology_ids, - node_ids, - edge_ids, - }) - } - - /// Checks the artifacts agree on every domain they share. - /// - /// The morton column's code count is the point domain, the adjacency spans the node and edge - /// domains, and the identity tables join to both. The pass checks every shared count once. - /// The read paths therefore index across artifacts without re-validating. The rank and row - /// pairings get a deterministic bounded sample of roundtrips rather than a full inversion - /// proof ([`roundtrip`](Self::roundtrip)). A mispairing outside the sample stays the fit-time - /// contract's to exclude. - /// - /// # Errors - /// - /// Returns [`OpenAtlasError::Columns`] when the point-domain columns disagree on a count, - /// and [`OpenAtlasError::RankInverse`] or [`OpenAtlasError::RowInverse`] when a rank or row - /// column disagrees with its reverse on a sampled position. - /// Returns [`OpenAtlasError::Nodes`], [`OpenAtlasError::Edges`], - /// [`OpenAtlasError::Subtree`], [`OpenAtlasError::Points`], [`OpenAtlasError::Types`], - /// [`OpenAtlasError::Identities`] or [`OpenAtlasError::EdgeIdentities`] when two artifacts - /// disagree on a count, and [`OpenAtlasError::EdgeUniverse`] when the edge rows exceed the - /// `u32` edge-row domain. - fn agree(&self) -> Result<(), OpenAtlasError> { - let codes = self.morton.count(); - - if self.points.len() as u64 != codes - || self.rows.len() as u64 != codes - || self.ranks.len() as u64 != codes - || self.positions_of_row.len() as u64 != codes - || self.positions_of_rank.len() as u64 != codes - { - return Err(OpenAtlasError::Columns { - codes, - coordinates: self.points.len() as u64, - rows: self.rows.len() as u64, - ranks: self.ranks.len() as u64, - positions: self.positions_of_row.len() as u64, - rank_positions: self.positions_of_rank.len() as u64, - }); - } - - self.roundtrip(codes)?; - - if self.adjacency.rows() != codes { - return Err(OpenAtlasError::Nodes { - adjacency: self.adjacency.rows(), - codes, - }); - } - - if self.endpoints.len() as u64 != self.adjacency.edges() { - return Err(OpenAtlasError::Edges { - adjacency: self.adjacency.edges(), - endpoints: self.endpoints.len() as u64, - }); - } - - if let Some(root_node) = self.quad.nodes().first() - && u64::from(root_node.points()) != codes - { - return Err(OpenAtlasError::Subtree { - quad: u64::from(root_node.points()), - codes, - }); - } - - if self.postings.points() != codes { - return Err(OpenAtlasError::Points { - postings: self.postings.points(), - codes, - }); - } - - if self.ontology_ids.len() != self.postings.types() { - return Err(OpenAtlasError::Types { - postings: self.postings.types(), - identities: self.ontology_ids.len(), - }); - } - - if self.node_ids.len() != codes { - return Err(OpenAtlasError::Identities { - identities: self.node_ids.len(), - codes, - }); - } - - if self.edge_ids.len() != self.adjacency.edges() { - return Err(OpenAtlasError::EdgeIdentities { - identities: self.edge_ids.len(), - edges: self.adjacency.edges(), - }); - } - - if u32::try_from(self.adjacency.edges()).is_err() { - return Err(OpenAtlasError::EdgeUniverse { - edges: self.adjacency.edges(), - }); - } - - Ok(()) - } - - /// Roundtrips a bounded sample of positions through the rank and row columns and their - /// reverses. - /// - /// The rank column and the position column each invert their reverse by the fit pipeline's own - /// construction. A file's recorded digest proves its bytes are the ones the fit published and - /// says nothing about whether two files agree with each other. A full inversion proof would - /// read both columns of each pair a second time at open. A bounded sample of roundtrips - /// spot-checks each pairing instead, and a fit that published a non-inverse pair almost surely - /// fails a sampled roundtrip. The sample is evenly spread over the `codes` positions with the - /// first and last included. The verdict is deterministic for a given generation, and a single - /// wrong entry outside the sample stays the fit-time contract's to exclude. - /// - /// # Errors - /// - /// Returns [`OpenAtlasError::RankInverse`] when the reverse rank column does not send a - /// sampled position's rank back to that position, and [`OpenAtlasError::RowInverse`] when the - /// reverse row column does not send the position's node back to it. - fn roundtrip(&self, codes: u64) -> Result<(), OpenAtlasError> { - const SAMPLES: u64 = 64; - - if codes == 0 || u32::try_from(codes - 1).is_err() { - return Ok(()); - } - - let ranks = self.ranks.view(); - let positions_of_rank = self.positions_of_rank.view(); - let positions_of_row = self.positions_of_row.view(); - let rows = self.rows.view(); - - let samples = SAMPLES.min(codes); - for index in 0..samples { - #[expect( - clippy::integer_division, - clippy::integer_division_remainder_used, - reason = "an evenly spaced sample point is the floor of its proportional position" - )] - let at = if samples == 1 { - 0 - } else { - index * (codes - 1) / (samples - 1) - }; - let position = BasePosition::from_u64(at); - - let rank = ranks[position]; - let roundtrip = positions_of_rank.get(rank).copied(); - if roundtrip != Some(position) { - return Err(OpenAtlasError::RankInverse { - position, - rank, - roundtrip, - }); - } - - let node = rows[position]; - let roundtrip = positions_of_row.get(node).copied(); - if roundtrip != Some(position) { - return Err(OpenAtlasError::RowInverse { - position, - node, - roundtrip, - }); - } - } - - Ok(()) - } -} - -impl Atlas { - /// Opens generation `id` from `root` and maps every serving artifact. - /// - /// Validates each format once. - /// - /// # Errors - /// - /// Returns [`OpenAtlasError::Unpublished`] when the generation is not published in this root, - /// and [`OpenAtlasError::Open`] when its identity or metadata document fails to read. - /// - /// Mapping the artifacts returns the open variant of the artifact that failed - /// ([`OpenAtlasError::OpenQuad`], [`OpenAtlasError::OpenMorton`], - /// [`OpenAtlasError::OpenArray`], [`OpenAtlasError::OpenAdjacency`], - /// [`OpenAtlasError::OpenPostings`], [`OpenAtlasError::OpenIdentity`]), the contract variant - /// when a mapped artifact violates its own format contract ([`OpenAtlasError::Adjacency`], - /// [`OpenAtlasError::Postings`], [`OpenAtlasError::Identity`]), and - /// [`OpenAtlasError::Shape`] when an artifact holds the wrong element type or shape. - /// - /// [`OpenAtlasError::Schedule`] follows when the recorded schedule exceeds the Morton key - /// width, which leaves no tile grid to serve. - /// - /// The agreement pass over the mapped artifacts returns [`OpenAtlasError::Columns`], - /// [`OpenAtlasError::Nodes`], [`OpenAtlasError::Edges`], [`OpenAtlasError::Subtree`], - /// [`OpenAtlasError::Points`], [`OpenAtlasError::Types`], [`OpenAtlasError::Identities`] or - /// [`OpenAtlasError::EdgeIdentities`] when two artifacts disagree on a count they share, - /// [`OpenAtlasError::RankInverse`] when the rank columns disagree on a sampled position, and - /// [`OpenAtlasError::EdgeUniverse`] when the edge rows exceed the `u32` edge-row domain. - /// - /// Deriving the type closure over the agreed artifacts returns [`OpenAtlasError::Closure`] - /// when the parent graph holds a cycle. - /// - /// Deriving the wire codec returns [`OpenAtlasError::Universe`] when the row count exceeds the - /// wire's `u32` id domain. - #[tracing::instrument(skip_all)] - pub(crate) fn open( - root: &GenerationRoot, - id: GenerationId, - OpenOptions { wire_secret }: OpenOptions, - ) -> Result { - let generation = root.open(id).map_err(|error| match error { - OpenError::Unpublished(id) => OpenAtlasError::Unpublished(id), - error @ (OpenError::Identity { .. } | OpenError::Document(_) | OpenError::Io(_)) => { - OpenAtlasError::Open(error) - } - })?; - - let artifacts = Artifacts::open(&generation)?; - let grid = Grid::new(generation.repository().metadata.reproducibility.config.lod)?; - artifacts.agree()?; - - let Artifacts { - quad, - morton, - points, - rows, - adjacency, - endpoints, - ranks, - positions_of_row, - positions_of_rank, - postings, - ontology_ids, - node_ids, - edge_ids, - } = artifacts; - - // The agreement proof matched the identity rows to the postings' type domain. Every - // displayed row therefore seeds the icon memo in domain. - let closure = ClosureMap::new(&postings, ontology_ids.displayed_rows())?; - - // The row column is the node universe's permutation. Its validated length is therefore the - // base bound. Edges cross the wire as link-entity identities and need no codec. - let node_universe = Universe::new(NodeRowId::from_u32(u32::try_from(rows.len()).map_err( - |_error| OpenAtlasError::Universe { - rows: rows.len() as u64, - }, - )?)); - let node_codec = RowCodec::derive(&wire_secret, id, NODE_LABEL); - - // The wire column maps the validated row column once. Every position-driven gather - // therefore reads permuted ids for free. - let wire_rows = rows - .view() - .iter() - .map(|&row| node_codec.encode(row, node_universe)) - .collect(); - - let world = generation.repository().metadata.evidence.lod.world; - let bounds = (morton.count() > 0).then(|| frame_extent(world)); - - Ok(Self { - generation, - wire_secret, - grid, - quad, - morton, - points, - rows, - adjacency, - endpoints, - ranks, - positions_of_row, - positions_of_rank, - postings, - closure, - ontology_ids, - node_ids, - edge_ids, - node_codec, - node_universe, - wire_rows, - bounds, - saturated: OnceLock::new(), - }) - } -} - -/// Opens one array artifact as its serving role's typed column. -/// -/// Every failure names the role: the open error and the shape error alike carry `kind`. -fn open_column( - generation: &Generation, - binding: &Binding, - kind: ArrayKind, -) -> Result, OpenAtlasError> -where - A: Artifact, - I: Id, - T: ColumnScalar + FromBytes + KnownLayout, -{ - let array = ArrayFile::open(binding.file().verify(generation)?) - .map_err(|error| OpenAtlasError::OpenArray { kind, error })?; - - Column::new(array).ok_or(OpenAtlasError::Shape { kind }) -} - -/// Opens and validates one identity artifact, binding its failures to the domain it serves. -/// -/// Every failure is loud - a key kind other than the store's included. -fn open_identities( - generation: &Generation, - binding: &Binding, - domain: IdentityDomain, -) -> Result, OpenAtlasError> -where - A: Artifact, - I: Key, - R: Row, -{ - let identities = IdentityFile::open(binding.file().verify(generation)?) - .map_err(|error| OpenAtlasError::OpenIdentity { domain, error })?; - IdentityTableArchive::new(identities) - .map_err(|error| OpenAtlasError::Identity { domain, error }) -} - -/// Derives the tight wire-frame extent of the full point set from the world frame. -/// -/// Normalization maps each axis's world minimum and maximum onto the frame edges - values real -/// points attain - and collapses a zero-extent axis to the frame centre. The extent is therefore -/// exact without scanning the coordinate column. -fn frame_extent(world: Bounds2) -> Bounds2 { - #[expect( - clippy::float_cmp, - reason = "a zero-extent axis collapses to the centre by exact equality: the normalization \ - contract, not a tolerance check" - )] - let axis = |minimum: f32, maximum: f32| { - if minimum == maximum { - (0.0, 0.0) - } else { - (-1.0, 1.0) - } - }; - - let (min_x, max_x) = axis(world.min().x(), world.max().x()); - let (min_y, max_y) = axis(world.min().y(), world.max().y()); - - Bounds2::new(Vec2::new(min_x, min_y), Vec2::new(max_x, max_y)) - .expect("the frame extent corners are finite and ordered") -} diff --git a/libs/@local/graph/atlas/src/serve/schedule/cut.rs b/libs/@local/graph/atlas/src/serve/schedule/cut.rs deleted file mode 100644 index cb003c5c968..00000000000 --- a/libs/@local/graph/atlas/src/serve/schedule/cut.rs +++ /dev/null @@ -1,385 +0,0 @@ -//! One scope schedule bound to one resolved cut offset. -//! -//! [`ScheduleCut`] is the delivery vocabulary of a restricted response: zoom `z` cuts at -//! `d(z) = z + span + k`, with the deepest bucket as the catch-all. Every delivery query - runs, -//! child bitmasks, first zooms, the root's counts - answers from the cascade's buckets alone. -//! Binding lives here too, because the cut owns the refusal of an offset past the key width. - -use core::{cmp::Ordering, ops::Range}; - -use hashql_core::id::{Id as _, IdSlice}; - -use super::{ - ArrivalIndex, ArrivalOverlay, ArrivalRow, BucketPost, ScheduleWidthError, ScopeSchedule, - ViewRow, -}; -use crate::{ - identity::BasePosition, - morton::{Depth, MortonCell, MortonKey}, - serve::{density::CutOffset, grid::Grid}, -}; - -/// One scope schedule read at one resolved cut offset. -/// -/// The delivery vocabulary of a restricted response: zoom `z` cuts at `d(z) = z + span + k` with -/// the deepest bucket as the catch-all, and every query answers from the cascade's buckets and -/// the view's arrival overlay together. A scope that folded its arrivals into its own cascade -/// binds the empty overlay, which every merge reads as absent. -#[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] -#[derive(Debug, Copy, Clone)] -pub(crate) struct ScheduleCut<'schedule> { - schedule: &'schedule ScopeSchedule, - /// The view's arrival overlay, merged into every delivery query. - overlay: &'schedule ArrivalOverlay, - /// The generation's span exponent. - span: u8, - /// The resolved cut offset. - k: CutOffset, - /// The deepest scope bucket: `max_tile_depth + span + k`, the catch-all. - deepest: Depth, -} - -impl<'schedule> ScheduleCut<'schedule> { - /// Binds one resolved cut offset over `schedule` and `overlay`. - /// - /// The bound cut serves `grid`'s zooms at `d(z) = z + span + k`, with the deepest bucket - /// `max_tile_depth + span + k` as the catch-all. The overlay's buckets clamp into that same - /// catch-all, so the merged delivery obeys one bucket domain. - /// - /// Caller requirement: a schedule holding arrivals of its own pairs with the empty overlay, - /// and a non-empty overlay pairs with a schedule holding none. The view's - /// [`ViewRow::Arrival`] vessels then address exactly one table. - /// - /// # Errors - /// - /// Returns [`ScheduleWidthError`] when that deepest bucket lies past the key width. Binding - /// refuses the offset rather than clamping it, because a sealed offset resolves against this - /// same generation's schedule, so an out-of-domain value is a defect to surface. - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - pub(super) fn bind( - schedule: &'schedule ScopeSchedule, - overlay: &'schedule ArrivalOverlay, - grid: Grid, - k: CutOffset, - ) -> Result { - debug_assert!( - overlay.is_empty() || schedule.arrivals().is_empty(), - "a view's arrivals live in its schedule or its overlay, never both", - ); - - let width_error = ScheduleWidthError { - max_tile_depth: grid.max_tile_depth(), - span: grid.span_log2(), - k, - }; - let deepest = grid - .max_tile_depth() - .checked_add(grid.span_log2()) - .and_then(|depth| depth.checked_add(k.get())) - .and_then(Depth::new) - .ok_or(width_error)?; - - Ok(Self { - schedule, - overlay, - span: grid.span_log2(), - k, - deepest, - }) - } - - /// Returns the deepest scope bucket: the catch-all. - #[cfg(test)] // The schedule tests read the catch-all bound for the replay oracle. - pub(crate) const fn deepest(&self) -> Depth { - self.deepest - } - - /// Views the arrival table the view's [`ViewRow::Arrival`] vessels address. - /// - /// The overlay's table when the overlay holds the view's arrivals, and the schedule's own - /// otherwise. Binding checked that exactly one of the two holds any. - pub(crate) const fn arrivals(&self) -> &'schedule IdSlice { - if self.overlay.is_empty() { - self.schedule.arrivals() - } else { - self.overlay.arrivals() - } - } -} - -impl ScheduleCut<'_> { - /// Returns the delivery cut of zoom `z`: buckets at or below it form the zoom's cumulative - /// schedule. - /// - /// # Panics - /// - /// This panics beyond the served grid. A zoom above the generation's deepest tile is a caller - /// defect rather than request data, because request validation rejects it first. - pub(crate) fn cut_of(&self, z: u8) -> Depth { - let cut = z + self.span + self.k.get(); - assert!( - cut <= self.deepest.get(), - "the schedule serves zooms 0..=max_tile_depth", - ); - Depth::new(cut).expect("binding validated the deepest cut against the key width") - } - - /// Feeds one bucket's delivered rows inside `cell` to `deliver`, ascending by `(key, rank)`, - /// and returns the run length. - /// - /// Buckets above the catch-all read one slot range each. The catch-all gathers its cell's - /// rows from every bucket at or beyond the cut - each already `(key, rank)`-sorted inside - /// the column - and restores that order across them for exactly the delivered rows. A - /// bucket past the catch-all holds nothing by construction. - /// - /// The overlay's arrivals of the same bucket merge in on their keys, after any fitted row - /// sharing a key, because every fitted row outranks every arrival. - fn run(&self, bucket: Depth, cell: MortonCell, deliver: &mut impl FnMut(ViewRow)) -> u32 { - let arrivals = self.overlay.run(bucket, cell, self.deepest); - - let count = match bucket.cmp(&self.deepest) { - Ordering::Less => { - let slots = self.schedule.bucket_slots(bucket); - let bounds = ScopeSchedule::cell_bounds(slots, |slot| slot.row.key, cell); - Self::merge( - slots[bounds] - .iter() - .map(|slot| (slot.row.key, slot.row.vessel)), - &arrivals, - deliver, - ) - } - Ordering::Equal => { - let mut rows = Vec::new(); - for bucket in (self.deepest.get()..=Depth::MAX.get()).filter_map(Depth::new) { - let slots = self.schedule.bucket_slots(bucket); - let bounds = ScopeSchedule::cell_bounds(slots, |slot| slot.row.key, cell); - rows.extend(slots[bounds].iter().map(|slot| slot.row)); - } - - rows.sort_unstable_by_key(|row| (row.key, row.rank)); - Self::merge( - rows.into_iter().map(|row| (row.key, row.vessel)), - &arrivals, - deliver, - ) - } - Ordering::Greater => 0, - }; - - u32::try_from(count).expect("run lengths lie within the u32 universe") - } - - /// Feeds the fitted rows and one bucket's arrivals to `deliver` in `(key, rank)` order, - /// returning the merged count. - /// - /// Both inputs ascend by key, and an arrival delivers after every fitted row at its key, - /// because every fitted row outranks every arrival. - fn merge( - fitted: impl Iterator, - arrivals: &[(MortonKey, ArrivalIndex)], - deliver: &mut impl FnMut(ViewRow), - ) -> usize { - let mut count = 0_usize; - let mut pending = arrivals.iter().peekable(); - for (key, vessel) in fitted { - while let Some(&(_, arrival)) = pending.next_if(|&&(held, _)| held < key) { - deliver(ViewRow::Arrival(arrival)); - count += 1; - } - deliver(vessel); - count += 1; - } - for &(_, arrival) in pending { - deliver(ViewRow::Arrival(arrival)); - count += 1; - } - - count - } - - /// Counts one bucket's rows inside `cell` without delivering them. - fn run_count(&self, bucket: Depth, cell: MortonCell) -> usize { - let arrivals = self.overlay.run_count(bucket, cell, self.deepest); - - arrivals - + match bucket.cmp(&self.deepest) { - Ordering::Less => { - let slots = self.schedule.bucket_slots(bucket); - ScopeSchedule::cell_bounds(slots, |slot| slot.row.key, cell).len() - } - Ordering::Equal => (self.deepest.get()..=Depth::MAX.get()) - .filter_map(Depth::new) - .map(|bucket| { - let slots = self.schedule.bucket_slots(bucket); - ScopeSchedule::cell_bounds(slots, |slot| slot.row.key, cell).len() - }) - .sum(), - Ordering::Greater => 0, - } - } - - /// Returns whether `cell` holds a view row in any bucket of `buckets`. - fn occupied(&self, buckets: Range, cell: MortonCell) -> bool { - buckets - .filter_map(Depth::new) - .any(|bucket| self.run_count(bucket, cell) > 0) - } - - /// Counts the view's rows delivered by the root's cumulative schedule. - /// - /// The root's visible count covers rows whose scope bucket lies at or below `d(0)`, - /// arrivals included. - pub(crate) fn root_delivered(&self) -> u64 { - let cut = self.cut_of(0); - let fitted = if cut >= self.deepest { - self.schedule.slots.len() as u64 - } else { - u64::from(self.schedule.posts[BucketPost::closing(cut)].as_u32()) - }; - - fitted + self.overlay.delivered_through(cut, self.deepest) - } - - /// Returns the deepest occupied scope bucket, zero for an empty view. - pub(crate) fn min_resolution(&self) -> u64 { - let fitted = self - .schedule - .deepest_occupied() - .map_or(0, |bucket| u64::from(bucket.min(self.deepest).get())); - let arrivals = self - .overlay - .min_resolution(self.deepest) - .map_or(0, |bucket| u64::from(bucket.get())); - - fitted.max(arrivals) - } - - /// Reads the occupied-child bitmask of `cell` at zoom `z`. - /// - /// Bit `i` is one exactly when Morton child `i` holds a view row the cumulative schedule - /// through `d(z)` has yet to deliver - a row whose scope bucket exceeds the cut. The deepest - /// zoom's cut is the catch-all, below which nothing exists, so its bitmask is zero. - pub(crate) fn children(&self, z: u8, cell: MortonCell) -> u8 { - let cut = self.cut_of(z); - if cut >= self.deepest { - return 0; - } - - let Some(children) = cell.children() else { - return 0; - }; - - let mut bits = 0_u8; - for (index, child) in children.into_iter().enumerate() { - let occupied = self.occupied((cut.get() + 1)..(self.deepest.get() + 1), child) - || self.overlay.occupied_past(cut, child); - bits |= u8::from(occupied) << index; - } - - bits - } - - /// Returns the first zoom whose cumulative schedule delivers `position`, [`None`] when the - /// position is not in the view. - /// - /// [`Self::cut_of`] inverted. Bucket `b` first enters at zoom `b - span - k`, clamped to the - /// root for the buckets the root itself spans. The catch-all inverts to the deepest served - /// zoom, because binding proved `deepest = max_tile_depth + span + k`. Every row of the view - /// therefore has a delivering zoom on the served grid. - /// - /// The scope counterpart of [`Grid::first_zoom`], which answers the same question for an - /// operator view off the corpus fenceposts. - pub(crate) fn first_zoom(&self, position: BasePosition) -> Option { - let bucket = self.bucket_of(position)?; - - // Binding validated `max_tile_depth + span + k` into the key width, so the subtrahend is - // itself a depth and the difference is a served zoom. - Some(bucket.get().saturating_sub(self.span + self.k.get())) - } - - /// Returns the first zoom whose cumulative schedule delivers an arrival. - /// - /// [`Self::first_zoom`]'s arrival counterpart, over the same inversion. The arrival's bucket - /// (the overlay's when the overlay holds the view's arrivals, the schedule's own otherwise) - /// clamps into the catch-all and inverts through the cut rule. Every arrival carries a - /// bucket, so every arrival has a delivering zoom on the served grid. - pub(crate) fn arrival_first_zoom(&self, index: ArrivalIndex) -> u8 { - let natural = if self.overlay.is_empty() { - self.schedule.arrival_bucket(index) - } else { - self.overlay.bucket_of(index) - }; - - natural - .min(self.deepest) - .get() - .saturating_sub(self.span + self.k.get()) - } - - /// Returns a position's scope bucket, [`None`] when the position is not in the view. - /// - /// The natural bucket clamped into the catch-all. - pub(crate) fn bucket_of(&self, position: BasePosition) -> Option { - let index = self - .schedule - .by_position - .binary_search_by_key(&position, |entry| entry.position) - .ok()?; - - Some(self.schedule.by_position[index].bucket.min(self.deepest)) - } - - /// Assembles zoom `z`'s delta delivery inside `cell`. - /// - /// The root delivers its whole cumulative schedule, buckets `0..=d(0)`. Every deeper zoom - /// delivers exactly its own cut bucket `d(z)`, one run. Runs keep their positional slot when - /// empty, so accumulation down an ancestry reproduces the total response as a set. - pub(crate) fn delta(&self, z: u8, cell: MortonCell) -> ScopeDelivery { - let cut = self.cut_of(z); - let first = if z == 0 { Depth::MIN } else { cut }; - - self.gather(first, cut, cell) - } - - /// Assembles zoom `z`'s total delivery inside `cell`: buckets `0..=d(z)`. - pub(crate) fn total(&self, z: u8, cell: MortonCell) -> ScopeDelivery { - self.gather(Depth::MIN, self.cut_of(z), cell) - } - - /// Gathers the contiguous bucket interval `first..=last` inside `cell`. - fn gather(&self, first: Depth, last: Depth, cell: MortonCell) -> ScopeDelivery { - let mut rows = Vec::new(); - let mut runs = Vec::with_capacity(usize::from(last.get() - first.get()) + 1); - - for bucket in (first.get()..=last.get()).filter_map(Depth::new) { - let count = self.run(bucket, cell, &mut |row| { - rows.push(row); - }); - runs.push(count); - } - - ScopeDelivery { - rows, - first_bucket: first.get(), - runs, - } - } -} - -/// The gathered rows and the wire head's run vocabulary of one scope delivery. -#[derive(Debug)] -pub(crate) struct ScopeDelivery { - /// The delivered rows, bucket-major, ascending by `(key, rank)` within a bucket. - pub rows: Vec, - /// The first bucket the runs describe. - pub first_bucket: u8, - /// Per-bucket delivered counts, bucket-major from `first_bucket`. - pub runs: Vec, -} diff --git a/libs/@local/graph/atlas/src/serve/schedule/mod.rs b/libs/@local/graph/atlas/src/serve/schedule/mod.rs deleted file mode 100644 index 8165d77ad8f..00000000000 --- a/libs/@local/graph/atlas/src/serve/schedule/mod.rs +++ /dev/null @@ -1,806 +0,0 @@ -//! Delivery buckets built over exactly the visible view. -//! -//! A restricted response delivers from a schedule of its own, a first-occupant cascade over the -//! visible rows alone, under the corpus rank restricted to them, read at the view's resolved cut -//! offset `k`. Every schedule-derived output - delivered rows, per-bucket runs, the child bitmask, -//! the root's visible count and resolution - is a function of the visible rows, their pinned keys -//! and ranks, and public policy. A hidden row contributes to none of them, so a scope's responses -//! carry no evidence of what its mask removed. -//! -//! [`ScopeSchedule::of`] builds the cascade once per scope, and [`ScopeSchedule::cut`] binds one -//! resolved offset and answers the delivery queries. Construction computes the cascade at the -//! natural depth, and every offset shares it, because depths at or above `d` decide the -//! first-occupant scan at depth `d`. A deeper catch-all therefore never changes which row claims a -//! shallower cell, and a row's bucket at deepest cut `D` is `min(natural, D)`. One slot column in -//! `(bucket, key, rank)` order therefore serves every admissible `k`. Buckets above the cut read -//! as slot ranges, and the catch-all reads the buckets at or beyond the cut as per-bucket -//! segments, restoring `(key, rank)` order across exactly the rows a cell delivers. -//! -//! A view reading a shared fitted schedule - the corpus artifacts, or the saturated memo - takes -//! its admitted arrivals as an [`ArrivalOverlay`] beside it instead, a second bucket column under -//! the same law, merged into every delivery query at read time. - -use alloc::sync::Arc; -use core::{cmp::Ordering, ops::Range}; - -use hashql_core::{ - heap::CollectIn as _, - id::{Id as _, IdArray, IdSlice}, -}; - -use self::cut::ScheduleCut; -use super::grid::Grid; -use crate::{ - allocator::{MemoryUsage, MemoryUsageAllocator}, - dataset::auxiliary::OwnedLegend, - identity::{BasePosition, ImportanceRank, NodeRowId}, - math::Vec2, - morton::{Depth, MortonCell, MortonKey}, - postgres::id::ArchivedEntityId, - salt::lod::{cascade, stage::WIRE_FRAME}, - serve::{ - Atlas, VisibilityProof, WireRow, - codec::Universe, - delta::{DeltaNode, PlacementCohort}, - density::CutOffset, - visibility::ProofKind, - }, -}; - -pub(crate) mod cut; -#[cfg(test)] -pub(crate) mod tests; - -hashql_core::id::newtype! { - /// A reference to a visible row by its slot in one scope schedule's natural order. - /// - /// Slots are dense and zero-based over one view's visible rows. The order is bucket-major at - /// the natural depth, ascending by `(key, rank)` inside a bucket. A slot is valid only against - /// the schedule that assigned it, because two views, or one view under two proofs, share no - /// slot vocabulary. - pub struct ScopeSlot(u32) -} - -hashql_core::id::newtype! { - /// A reference to one cohort arrival by its slot in a schedule's arrival table. - /// - /// Indices are dense and zero-based over one schedule's arrivals, ascending by identity. An - /// index is valid only against the schedule that assigned it, exactly as a [`ScopeSlot`] is. - pub struct ArrivalIndex(u32) -} - -/// One row of a visible view, in the domain that publishes it. -/// -/// A schedule orders rows from two domains under one `(key, rank)` law: fitted rows address the -/// generation's columns by base position, while placed arrivals address the schedule's own -/// arrival table. Every delivery consumer matches on the domain, because a base row resolves its -/// payloads from the columns and an arrival from its captured placement. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum ViewRow { - /// A generation row, addressed by its slot in the base order. - Base(BasePosition), - /// A cohort arrival, addressed into the schedule's arrival table. - Arrival(ArrivalIndex), -} - -/// One cohort arrival of a schedule, resolved whole at construction. -/// -/// Everything an arrival-bearing response reads. The identity keys the ingress withdrawal -/// filter, and the projected coordinate feeds the `POSITIONS` column. Its wire id feeds the -/// `ROW_IDS` column, pre-encoded under the entry universe, which admits every cohort slot by -/// construction, and the legend feeds the detail trailer. -#[derive(Debug, Clone)] -pub(crate) struct ArrivalRow { - /// The arrival's identity, the ingress withdrawal filter's key. - pub identity: ArchivedEntityId, - /// The recorded wire coordinate. - pub position: Vec2, - /// The arrival's wire id, encoded under the entry universe at construction. - pub wire: WireRow, - /// The legend captured at placement. - pub legend: OwnedLegend, -} - -impl ArrivalRow { - /// Resolves one placed arrival into its delivery row and quantized key. - /// - /// The key quantizes the recorded coordinate on the wire frame - every published placement - /// lies inside it, because an out-of-frame projection never places - and the wire id - /// encodes under `universe`, the entry's own, which admits every cohort row by - /// construction. - fn of( - atlas: &Atlas, - universe: Universe, - identity: ArchivedEntityId, - arrival: &DeltaNode, - ) -> (MortonKey, Self) { - let [x, y] = WIRE_FRAME.quantize(arrival.position); - - ( - MortonKey::new(x, y), - Self { - identity, - position: arrival.position, - wire: atlas.node_codec.encode(arrival.id, universe), - legend: arrival.legend.clone(), - }, - ) - } -} - -/// One arrival interleaved into a range-shaped delivery. -/// -/// The operator fast paths deliver contiguous base-position ranges, and a splice names where one -/// arrival sits among them: `at` is the arrival's index in the final merged order, counting rows -/// of both kinds. A consumer walks the ranges and emits the named arrival whenever its output -/// index reaches a splice, so the interleave costs one comparison per row and no gather. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct Splice { - /// The arrival's index in the final delivered order. - pub at: u32, - /// The spliced arrival, addressed into the view's arrival table. - pub arrival: ArrivalIndex, -} - -/// Returns the cohort arrivals `proof` admits, ascending by identity. -/// -/// The filter is slot membership, the same admission the mask builder widened the proof for. An -/// operator proof admits every slot, so a corpus view takes the whole cohort. The identity sort -/// fixes arrival order whatever order the cohort's map iterates. -fn admitted_arrivals<'scope>( - proof: &VisibilityProof, - cohort: PlacementCohort<'scope>, -) -> Vec<(ArchivedEntityId, &'scope DeltaNode)> { - let mut admitted: Vec<_> = cohort - .nodes() - .filter(|&(_, arrival)| proof.contains(arrival.id)) - .collect(); - admitted.sort_unstable_by_key(|&(identity, _)| identity); - - admitted -} - -hashql_core::id::newtype! { - /// A fencepost of the slot column: the boundary preceding one natural bucket. - /// - /// Bucket `b` spans the slots between its opening post and its closing post - the opening - /// post of bucket `b + 1`. The final post trails the deepest bucket and marks the column's - /// length. The fencepost column stores each post's slot, so a post is an index into - /// [`ScopeSchedule::posts`] and its value there is a [`ScopeSlot`]. - struct BucketPost(u32) -} - -impl BucketPost { - /// The post opening `bucket`: its slot is the first at `bucket`'s depth or deeper. - fn opening(bucket: Depth) -> Self { - Self::from_u32(u32::from(bucket.get())) - } - - /// The post closing `bucket`: the opening post of the next-deeper bucket. - fn closing(bucket: Depth) -> Self { - Self::from_u32(u32::from(bucket.get()) + 1) - } -} - -/// The delivery schedule one view's responses read. -/// -/// An operator proof serves the generation's own corpus schedule, and a scoped proof serves the -/// cascade built over exactly its visible rows. The constructor derives the variant from the -/// proof, so production has one build site and the pairing law in one place. -/// -/// Both variants carry the view's arrival overlay beside the fitted schedule, because the fitted -/// side - the corpus artifacts, or the saturated memo - belongs to the generation while the -/// admitted arrivals are the entry's own. A scope that builds its own cascade folds its arrivals -/// into it instead, and its overlay is empty by construction: exactly one of the two holds the -/// view's arrivals. -/// -/// Caller requirement: as with the census, a schedule travels with the proof it derives from. -/// Assembly refuses a proof paired with the other variant's schedule. -#[derive(Debug)] -pub(crate) enum ViewSchedule { - /// The generation's corpus schedule, where every zoom keeps its recorded cut, with the - /// view's arrival overlay beside it. - Corpus(ArrivalOverlay), - /// The view's own cascade, shared by every request of its scope, with the view's arrival - /// overlay beside it. - Scope(Arc, ArrivalOverlay), -} - -impl ViewSchedule { - /// Derives the schedule variant `proof` serves under. - /// - /// An operator proof reads the corpus artifacts. Any scoped proof - saturated or empty - /// included - serves a cascade, because the serving contract follows the scope declaration - /// rather than the visible cardinality. A scope whose node mask admits the whole corpus - /// reads the generation's shared saturated cascade instead of building one. A cascade is a - /// function of the visible node rows alone, so every saturated scope builds identical - /// buckets and the sharing changes which allocation answers, never which contract. The - /// sharing test is exact, so a mask even one row short of the corpus - a scope whose cohort - /// withdrew a single fitted row included - builds and retains its own full-corpus-sized - /// cascade instead. While any snapshot withdraws a fitted row, every corpus-admitting scope - /// resolves onto that arm - a cost on resolution latency and entry weight rather than on - /// served bytes, and one an unarchive can end, because a later publication whose withdrawn - /// set is empty folds nothing and the memo answers again. - /// - /// The variants that read a shared fitted schedule - the corpus artifacts and the saturated - /// memo - take their admitted arrivals as an overlay, whose buckets are exact there because - /// the visible fitted rows are the whole corpus in both. A scope that builds its own cascade - /// folds its arrivals into the build and takes the empty overlay. - #[must_use] - pub(crate) fn of(atlas: &Atlas, proof: &VisibilityProof, cohort: PlacementCohort<'_>) -> Self { - match proof.kind() { - ProofKind::Corpus => Self::Corpus(ArrivalOverlay::of(atlas, proof, cohort)), - ProofKind::Scope if proof.nodes_saturated_below(atlas.morton.count()) => Self::Scope( - Arc::clone(atlas.saturated_scope_schedule()), - ArrivalOverlay::of(atlas, proof, cohort), - ), - ProofKind::Scope => Self::Scope( - Arc::new(ScopeSchedule::of(atlas, proof, cohort)), - ArrivalOverlay::empty(), - ), - } - } -} - -/// One visible row of the cascade's input. -/// -/// The vessel addresses the row in its own domain, and the key and rank are the row's pinned -/// layout values: quantized coordinate and corpus rank for a fitted row, quantized placement -/// and a rank past every fitted rank for an arrival. These three are the whole -/// vocabulary the schedule reads. -#[derive(Debug, Copy, Clone)] -struct ScopeRow { - /// The row, in the domain that publishes it. - vessel: ViewRow, - /// The row's Morton key, quantized from the delivered coordinate column. - key: MortonKey, - /// The row's rank within the view: dense, ascending in corpus rank order, arrivals after - /// every fitted row in identity order. - rank: ImportanceRank, -} - -/// A visible row with its natural cascade bucket. -#[derive(Debug, Copy, Clone)] -struct SlottedRow { - /// The row's first-occupant bucket at the natural depth. - bucket: Depth, - /// The row itself. - row: ScopeRow, -} - -/// One entry of the position lookup: a visible row's natural bucket, keyed by position. -#[derive(Debug, Copy, Clone)] -struct PositionBucket { - /// The row's slot in the generation's base order. - position: BasePosition, - /// The row's first-occupant bucket at the natural depth. - bucket: Depth, -} - -/// A resolved cut past the key width. -/// -/// The deepest scope bucket is `max_tile_depth + span + k` and a complete Morton key resolves 32 -/// subdivisions, so an offset that lands beyond them has no grid to deliver on. The binding refuses -/// that offset whole, and nothing clamps it or substitutes another schedule. -#[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct ScheduleWidthError { - /// The generation's deepest served tile zoom. - pub max_tile_depth: u8, - /// The generation's span exponent. - pub span: u8, - /// The refused offset. - pub k: CutOffset, -} - -impl core::fmt::Display for ScheduleWidthError { - fn fmt(&self, fmt: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - write!( - fmt, - "delivery-cut offset {} puts the deepest scope bucket {} + {} + {} past the {} \ - subdivisions a Morton key resolves", - self.k.get(), - self.max_tile_depth, - self.span, - self.k.get(), - Depth::MAX.get(), - ) - } -} - -impl core::error::Error for ScheduleWidthError {} - -/// Returns the deepest grid `key` shares with any corpus row, [`None`] for an empty corpus. -/// -/// The code column sorts within each bucket segment, and a sorted-order neighbour attains the -/// deepest shared grid over a sorted key set, so each segment answers with one binary search and -/// its two neighbouring keys. -fn deepest_corpus_shared(atlas: &Atlas, key: MortonKey) -> Option { - let codes = atlas.morton.codes(); - - let mut deepest: Option = None; - for segment in atlas.morton.fenceposts().segments() { - let slice = &codes[segment]; - let at = slice.partition_point(|code| code.get() < key.to_bits()); - - for neighbour in [at.checked_sub(1), (at < slice.len()).then_some(at)] - .into_iter() - .flatten() - { - let shared = key.shared_depth(MortonKey::from_bits(slice[neighbour].get())); - deepest = Some(deepest.map_or(shared, |held| held.max(shared))); - } - } - - deepest -} - -/// The scope cascade of one visible view, computed once and read at any admissible cut offset. -/// -/// Construction assigns every visible row the shallowest grid depth at which it is its cell's -/// first representative in rank order - the same first-occupant law behind the corpus schedule - -/// and orders the slots bucket-major, ascending by key inside a bucket, rank breaking exact-key -/// ties. [`Self::cut`] binds a resolved offset over the result. The value is immutable after -/// construction and holds no per-offset state, so one schedule serves every request of its scope -/// concurrently. -/// -/// A schedule resolves once per scope beside the proof it builds from, and the requests under that -/// scope share it. -/// -/// Caller requirement: a schedule travels with the proof it builds over. Delivery reads it as the -/// view's own cascade without re-deriving it, so a schedule paired with a foreign proof serves the -/// wrong view's rows. -#[derive(Debug)] -pub(crate) struct ScopeSchedule { - /// Every visible row in natural order. - slots: Box, MemoryUsageAllocator>, - /// Each fencepost's slot: bucket `b` spans its opening post's slot to its closing post's. - posts: IdArray, - /// The natural buckets ascending by position: the row-to-bucket lookup, binary-searched. - /// - /// The lookup covers the base domain alone, because its callers resolve identity-domain - /// ingress against the generation's columns. - by_position: Box<[PositionBucket], MemoryUsageAllocator>, - /// The cohort arrivals the schedule's [`ViewRow::Arrival`] vessels address, ascending by - /// identity. - arrivals: Box, MemoryUsageAllocator>, - /// Each arrival's bucket under the slot column's own assignment: the arrival-to-bucket - /// lookup, the arrival counterpart of the position lookup. - arrival_buckets: Box, MemoryUsageAllocator>, - memory_usage: MemoryUsage, -} - -impl ScopeSchedule { - /// The natural bucket domain: depths `0..=32` of the complete Morton key. - const BUCKETS: usize = Depth::MAX.get() as usize + 1; - - /// Builds the cascade over the visible view `proof` admits on `atlas`. - /// - /// The gather traverses the generation's reverse rank column - the rank column's inverse by - /// the fit pipeline's construction, spot-checked at open - in rank order and keeps the - /// positions whose rows `proof` admits. The view's - /// rows therefore arrive rank-ascending, and each row's local rank is its arrival ordinal: - /// dense and pairwise distinct by construction. [`Self::over`] assigns the buckets. - /// - /// The cohort's placed arrivals the proof admits join the same pass. Each takes the key its - /// projected coordinate quantizes to on the wire frame - every published placement lies inside - /// it, because an out-of-frame projection never places - and a rank past every fitted row's, - /// ascending in identity order, so every generation row outranks every arrival and arrival - /// order is stable whatever order the cohort's map iterates. The wire id pre-encodes under - /// the entry universe, which admits every cohort slot by construction. - pub(crate) fn of(atlas: &Atlas, proof: &VisibilityProof, cohort: PlacementCohort<'_>) -> Self { - let alloc = MemoryUsageAllocator::global(); - - let row_ids = atlas.rows.view(); - - let visible = usize::try_from(proof.visible_below(atlas.morton.count())) - .expect("a visible row count fits usize"); - let mut rows = Vec::with_capacity(visible); - for &position in atlas.positions_of_rank.view() { - if proof.contains(row_ids[position]) { - rows.push(ScopeRow { - vessel: ViewRow::Base(position), - key: atlas.morton.code(position), - rank: ImportanceRank::from_usize(rows.len()), - }); - } - } - - let universe = cohort.universe(atlas.node_universe()); - let mut arrivals = Vec::new_in(alloc); - for (identity, arrival) in admitted_arrivals(proof, cohort) { - let (key, row) = ArrivalRow::of(atlas, universe, identity, arrival); - rows.push(ScopeRow { - vessel: ViewRow::Arrival(ArrivalIndex::from_usize(arrivals.len())), - key, - rank: ImportanceRank::from_usize(rows.len()), - }); - arrivals.push(row); - } - - Self::over(rows, arrivals.into_boxed_slice()) - } - - /// Builds the empty schedule: nothing delivers, and no cell holds a row at any depth. - #[cfg(test)] // The cache and serve tests build empty scoped views. - pub(crate) fn empty() -> Self { - Self::over(Vec::new(), Box::new_in([], MemoryUsageAllocator::global())) - } - - /// Returns the schedule's retained heap in bytes: its own allocator's live count. - /// - /// Every retained column allocates through one counting allocator, and nothing accrues - /// after construction - the catch-all reads straight off the column, so no offset ever - /// materializes per-offset state. The count covers the columns' requested layouts alone. - /// The heap an arrival's captured display payload owns stays outside it. - pub(crate) fn heap_bytes(&self) -> u64 { - self.memory_usage.get() as u64 - } - - /// Builds the cascade over exactly the given rows. - /// - /// [`cascade::separation_buckets`] assigns each row its natural bucket - the assignment - /// [`cascade::buckets`], the function behind the corpus schedule at fit time, computes at - /// [`Depth::MAX`] - in one pass over the rows' `(key, rank)` order. Rows co-located at the - /// complete key width never claim a cell and take the deepest bucket, exactly as the corpus - /// catch-all takes them. An empty view builds an empty schedule, which delivers nothing and - /// occupies no cell at any depth. - /// - /// Caller requirement: the rows' ranks are pairwise distinct, and every [`ViewRow::Arrival`] - /// vessel addresses `arrivals`. [`Self::of`] guarantees both by enumeration, and a fixture - /// caller owes the same properties. - fn over(mut rows: Vec, arrivals: Box<[ArrivalRow], MemoryUsageAllocator>) -> Self { - let alloc = Box::allocator(&arrivals).clone(); - let memory_usage = alloc.memory_usage(); - - rows.sort_unstable_by_key(|row| (row.key, row.rank)); - let buckets = cascade::separation_buckets(&rows, |row| row.key, |row| row.rank); - - // Slot order: bucket-major, ascending key within a bucket, rank breaking exact-key - // ties. One sorted column is the whole schedule. - let mut slots: Vec<_, _> = rows - .into_iter() - .zip(&*buckets) - .map(|(row, &bucket)| SlottedRow { bucket, row }) - .collect_in(alloc.clone()); - slots.sort_unstable_by_key(|slot| (slot.bucket, slot.row.key, slot.row.rank)); - - // Counting sort's tally, then the running total: a post's slot is the number of rows - // in the buckets it closes off, so post 0 opens bucket 0 at slot 0 and the final post - // carries the column's length. - let mut counts = IdArray::::from_elem(0); - for slot in &slots { - counts[BucketPost::closing(slot.bucket)] += 1; - } - - let mut placed = 0_u32; - let posts = counts.map(|count| { - placed += count; - ScopeSlot::from_u32(placed) - }); - - let mut by_position = Vec::new_in(alloc.clone()); - let mut arrival_buckets = alloc::vec::from_elem_in(Depth::MIN, arrivals.len(), alloc); - for slot in &slots { - match slot.row.vessel { - ViewRow::Base(position) => by_position.push(PositionBucket { - position, - bucket: slot.bucket, - }), - ViewRow::Arrival(index) => arrival_buckets[index.as_usize()] = slot.bucket, - } - } - by_position.sort_unstable_by_key(|entry| entry.position); - - Self { - slots: IdSlice::from_boxed_slice(slots.into_boxed_slice()), - posts, - by_position: by_position.into_boxed_slice(), - arrivals: IdSlice::from_boxed_slice(arrivals), - arrival_buckets: IdSlice::from_boxed_slice(arrival_buckets.into_boxed_slice()), - memory_usage, - } - } - - /// Binds one resolved cut offset over the cascade and the view's arrival overlay. - /// - /// The bound cut serves `grid`'s zooms at `d(z) = z + span + k`, with the deepest bucket - /// `max_tile_depth + span + k` as the catch-all, and merges `overlay` into every delivery - /// query. A schedule that folded its arrivals binds the empty overlay. - /// - /// # Errors - /// - /// Returns [`ScheduleWidthError`] when that deepest bucket lies past the key width. Binding - /// refuses the offset rather than clamping it, because a sealed offset resolves against this - /// same generation's schedule, so an out-of-domain value is a defect to surface. - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - pub(super) fn cut<'schedule>( - &'schedule self, - overlay: &'schedule ArrivalOverlay, - grid: Grid, - k: CutOffset, - ) -> Result, ScheduleWidthError> { - ScheduleCut::bind(self, overlay, grid, k) - } - - /// Views the cohort arrivals the schedule's [`ViewRow::Arrival`] vessels address. - pub(crate) const fn arrivals(&self) -> &IdSlice { - &self.arrivals - } - - /// Returns the deepest occupied cascade bucket, [`None`] for an empty view. - /// - /// The fencepost pair of a bucket holding a slot differs, so the answer is the deepest - /// bucket whose posts part. Arrivals folded into the build count; an overlay's sit outside - /// the posts and outside this answer. - pub(crate) fn deepest_occupied(&self) -> Option { - Depth::all().rev().find(|&bucket| { - self.posts[BucketPost::closing(bucket)] > self.posts[BucketPost::opening(bucket)] - }) - } - - /// Returns an arrival's bucket under the slot column's own assignment. - fn arrival_bucket(&self, index: ArrivalIndex) -> Depth { - self.arrival_buckets[index] - } - - /// Returns one natural bucket's slots. - fn bucket_slots(&self, bucket: Depth) -> &[SlottedRow] { - let range = Range { - start: self.posts[BucketPost::opening(bucket)], - end: self.posts[BucketPost::closing(bucket)], - }; - - &self.slots[range] - } - - /// Returns the bounds of `cell`'s keys within one key-sorted slice. - fn cell_bounds( - items: &[T], - key: impl Fn(&T) -> MortonKey, - cell: MortonCell, - ) -> Range { - let min = cell.min_key(); - let max = cell.max_key(); - let start = items.partition_point(|item| key(item) < min); - let end = start + items[start..].partition_point(|item| key(item) <= max); - - start..end - } -} - -/// An admitted arrival with its natural bucket and quantized key. -#[derive(Debug, Copy, Clone)] -struct OverlaySlot { - /// The arrival's natural bucket against the corpus and its earlier cohort peers. - bucket: Depth, - /// The arrival's Morton key, quantized from the projected coordinate. - key: MortonKey, - /// The arrival addressed, ascending in identity order across the overlay. - arrival: ArrivalIndex, -} - -/// The admitted arrivals of a view whose fitted schedule is corpus-wide. -/// -/// The corpus artifacts and the saturated memo both order exactly the whole fitted corpus, and -/// both outlive any one entry, so the entry's own arrivals ride beside them rather than inside. -/// Each -/// arrival takes its natural first-occupant bucket under the same separation law the cascades -/// use: one past the deepest grid it shares with any better-ranked row, where every fitted row -/// outranks every arrival and earlier cohort identities outrank later ones. That assignment is -/// exact precisely because the visible fitted rows are the whole corpus, so the nearest -/// better-ranked key is a corpus-column search rather than a per-scope gather. -/// -/// The slots sort in `(bucket, key, rank)` order, the cascades' own delivery order, and every -/// schedule-derived output - runs, gathers, the child bitmask, the root's counts - reads the -/// overlay as a second bucket column merged at query time. The consumers clamp buckets into -/// their own catch-all, so one overlay serves the corpus contract and any saturated cut offset. -#[derive(Debug)] -pub(crate) struct ArrivalOverlay { - /// The overlay entries in `(bucket, key, rank)` order, buckets at their natural depth. - slots: Box<[OverlaySlot], MemoryUsageAllocator>, - /// The arrival table the slots and the delivered vessels address, ascending by identity. - arrivals: Box, MemoryUsageAllocator>, - /// Each arrival's natural bucket in table order: the arrival-to-bucket lookup, before any - /// catch-all clamp. - buckets: Box, MemoryUsageAllocator>, - memory_usage: MemoryUsage, -} - -impl ArrivalOverlay { - /// Builds the overlay of no arrivals, which every query reads as absent. - pub(crate) fn empty() -> Self { - let alloc = MemoryUsageAllocator::global(); - let memory_usage = alloc.memory_usage(); - - Self { - slots: Box::new_in([], alloc.clone()), - arrivals: IdSlice::from_boxed_slice(Box::new_in([], alloc.clone())), - buckets: IdSlice::from_boxed_slice(Box::new_in([], alloc)), - memory_usage, - } - } - - /// Builds the overlay of the arrivals `proof` admits from `cohort` on `atlas`. - /// - /// Each admitted arrival takes its natural bucket by the separation law: one past the - /// deepest grid it shares with any corpus row or any earlier cohort arrival, saturating at - /// [`Depth::MAX`] for a full-key co-location. A cascade assigns exactly the same bucket. The - /// corpus side of that search binary-searches each bucket segment of the code column, whose - /// keys ascend within a segment, and a sorted-order neighbour attains the deepest shared - /// grid over a sorted key set. - /// - /// Caller requirement: the visible fitted rows are the whole corpus - an operator proof, or - /// a scope with a saturated node mask. A narrower scope's arrival buckets depend on its own - /// visible keys, and its cascade build folds the arrivals instead. - pub(crate) fn of(atlas: &Atlas, proof: &VisibilityProof, cohort: PlacementCohort<'_>) -> Self { - let universe = cohort.universe(atlas.node_universe()); - let admitted = admitted_arrivals(proof, cohort); - - let alloc = MemoryUsageAllocator::global(); - let memory_usage = alloc.memory_usage(); - - let mut slots = Vec::with_capacity_in(admitted.len(), alloc.clone()); - let mut arrivals = Vec::with_capacity_in(admitted.len(), alloc.clone()); - let mut buckets = Vec::with_capacity_in(admitted.len(), alloc); - - let mut earlier: Vec = Vec::with_capacity(admitted.len()); - for (identity, arrival) in admitted { - let (key, row) = ArrivalRow::of(atlas, universe, identity, arrival); - - let mut shared = deepest_corpus_shared(atlas, key); - let at = earlier.partition_point(|&held| held < key); - for neighbour in [at.checked_sub(1), (at < earlier.len()).then_some(at)] - .into_iter() - .flatten() - { - let depth = key.shared_depth(earlier[neighbour]); - shared = Some(shared.map_or(depth, |held| held.max(depth))); - } - earlier.insert(at, key); - - let bucket = shared.map_or(Depth::MIN, |depth| depth.saturating_add(1)); - slots.push(OverlaySlot { - bucket, - key, - arrival: ArrivalIndex::from_usize(arrivals.len()), - }); - arrivals.push(row); - buckets.push(bucket); - } - - slots.sort_unstable_by_key(|slot| (slot.bucket, slot.key, slot.arrival)); - - Self { - slots: slots.into_boxed_slice(), - arrivals: IdSlice::from_boxed_slice(arrivals.into_boxed_slice()), - buckets: IdSlice::from_boxed_slice(buckets.into_boxed_slice()), - memory_usage, - } - } - - /// Returns whether the overlay holds no arrival. - pub(crate) const fn is_empty(&self) -> bool { - self.slots.is_empty() - } - - /// Views the arrival table the overlay's delivered vessels address. - pub(crate) const fn arrivals(&self) -> &IdSlice { - &self.arrivals - } - - /// Returns an arrival's natural bucket, before any catch-all clamp. - /// - /// A consumer clamps the answer into its own catch-all, exactly as the delivery queries - /// clamp the slots. - pub(crate) fn bucket_of(&self, index: ArrivalIndex) -> Depth { - self.buckets[index] - } - - /// Returns the overlay's retained heap in bytes: its own allocator's live count. - /// - /// The slots, the arrival table, and the bucket lookup allocate through one counting - /// allocator. The count covers their requested layouts alone. The heap an arrival's - /// captured display payload owns stays outside it. - pub(crate) fn heap_bytes(&self) -> u64 { - self.memory_usage.get() as u64 - } - - /// Returns the slot range of natural bucket `bucket`. - fn bucket_range(&self, bucket: Depth) -> Range { - let start = self.slots.partition_point(|slot| slot.bucket < bucket); - let end = start + self.slots[start..].partition_point(|slot| slot.bucket == bucket); - - start..end - } - - /// Returns bucket `bucket`'s delivered arrivals inside `cell` under the catch-all `deepest`, - /// ascending by `(key, rank)`. - /// - /// Buckets above the catch-all read their own entries. The catch-all gathers every entry at - /// or beyond it, restoring `(key, rank)` order across them, exactly as a cascade's deepest - /// bucket takes its co-located rows. A bucket past the catch-all holds nothing. - pub(super) fn run( - &self, - bucket: Depth, - cell: MortonCell, - deepest: Depth, - ) -> Vec<(MortonKey, ArrivalIndex)> { - match bucket.cmp(&deepest) { - Ordering::Less => { - let slots = &self.slots[self.bucket_range(bucket)]; - let bounds = ScopeSchedule::cell_bounds(slots, |slot| slot.key, cell); - - slots[bounds] - .iter() - .map(|slot| (slot.key, slot.arrival)) - .collect() - } - Ordering::Equal => { - let tail = self.slots.partition_point(|slot| slot.bucket < deepest); - let slots = &self.slots[tail..]; - let mut gathered: Vec<(MortonKey, ArrivalIndex)> = slots - .iter() - .filter(|slot| cell.contains(slot.key)) - .map(|slot| (slot.key, slot.arrival)) - .collect(); - gathered.sort_unstable(); - - gathered - } - Ordering::Greater => Vec::new(), - } - } - - /// Counts bucket `bucket`'s arrivals inside `cell` under the catch-all `deepest`. - fn run_count(&self, bucket: Depth, cell: MortonCell, deepest: Depth) -> usize { - match bucket.cmp(&deepest) { - Ordering::Less => { - let slots = &self.slots[self.bucket_range(bucket)]; - ScopeSchedule::cell_bounds(slots, |slot| slot.key, cell).len() - } - Ordering::Equal => { - let tail = self.slots.partition_point(|slot| slot.bucket < deepest); - self.slots[tail..] - .iter() - .filter(|slot| cell.contains(slot.key)) - .count() - } - Ordering::Greater => 0, - } - } - - /// Counts the arrivals the cumulative schedule through `cut` delivers under `deepest`. - pub(super) fn delivered_through(&self, cut: Depth, deepest: Depth) -> u64 { - if cut >= deepest { - self.slots.len() as u64 - } else { - self.slots.partition_point(|slot| slot.bucket <= cut) as u64 - } - } - - /// Returns the deepest occupied overlay bucket under the catch-all `deepest`, [`None`] for - /// an empty overlay. - pub(super) fn min_resolution(&self, deepest: Depth) -> Option { - self.slots.last().map(|slot| slot.bucket.min(deepest)) - } - - /// Returns whether `cell` holds an arrival the cumulative schedule through `cut` has yet to - /// deliver. - /// - /// Every natural bucket past `cut` qualifies, because the catch-all clamp keeps a deep - /// entry inside the served bucket domain. The caller's own early return covers a cut at the - /// catch-all, below which nothing exists. - pub(super) fn occupied_past(&self, cut: Depth, cell: MortonCell) -> bool { - let tail = self.slots.partition_point(|slot| slot.bucket <= cut); - self.slots[tail..] - .iter() - .any(|slot| cell.contains(slot.key)) - } -} diff --git a/libs/@local/graph/atlas/src/serve/schedule/tests.rs b/libs/@local/graph/atlas/src/serve/schedule/tests.rs deleted file mode 100644 index 972371fb81d..00000000000 --- a/libs/@local/graph/atlas/src/serve/schedule/tests.rs +++ /dev/null @@ -1,779 +0,0 @@ -//! Schedule-internal tests: the cascade law at its own vocabulary. -//! -//! These tests construct [`ScopeRow`]s and fixture schedules directly, so they live as the -//! schedule's own child. The route-level delivery battery, written against the documented -//! contract instead of these internals, stays in `serve::tests::schedule`. - -#![expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] - -use hashql_core::id::Id as _; - -use super::{ - ArrivalIndex, ArrivalOverlay, ArrivalRow, ScopeRow, ScopeSchedule, ViewRow, ViewSchedule, -}; -use crate::{ - allocator::MemoryUsageAllocator, - identity::{BasePosition, ImportanceRank}, - morton::{Depth, MortonCell, MortonKey}, - postgres::id::ArchivedEntityId, - serve::{ - VisibilityProof, - delta::PlacementCohort, - density::CutOffset, - grid::Grid, - tests::{FIXTURE_LOD, mask_hiding, publish}, - }, -}; - -/// The cascade puts co-located rows in the catch-all and everything reads back by position. -/// -/// The hand derivation uses four rows on a `span = 1, max_tile_depth = 1` grid, whose deepest -/// bucket is `2 + k`. Rows a and b share the complete key, so the better-ranked a claims depth 0 -/// and b never claims a cell. Row c claims depth 1 - its first depth apart from a - and d claims -/// depth 2 under `k = 0`'s catch-all... at which the law stops distinguishing it from b. -#[test] -fn hand_cascade_pins_the_first_occupant_law() { - // On the unit grid, a and b share the north-west cell, c takes the north-east, and d shares a's - // depth-1 cell while occupying its own depth-2 cell, because a's x top bits read 00 and d's - // read 01. - let a = MortonKey::new(0x1000_0000, 0x1000_0000); - let b = MortonKey::new(0x1000_0000, 0x1000_0000); - let c = MortonKey::new(0xC000_0000, 0x2000_0000); - let d = MortonKey::new(0x4000_0000, 0x1000_0000); - let rows = vec![ - ScopeRow { - vessel: ViewRow::Base(BasePosition::new(0)), - key: a, - rank: ImportanceRank::from_u32(0), - }, - ScopeRow { - vessel: ViewRow::Base(BasePosition::new(1)), - key: b, - rank: ImportanceRank::from_u32(3), - }, - ScopeRow { - vessel: ViewRow::Base(BasePosition::new(2)), - key: c, - rank: ImportanceRank::from_u32(1), - }, - ScopeRow { - vessel: ViewRow::Base(BasePosition::new(3)), - key: d, - rank: ImportanceRank::from_u32(2), - }, - ]; - - let schedule = ScopeSchedule::over(rows, Box::new_in([], MemoryUsageAllocator::global())); - let grid = crate::serve::grid::Grid::new(crate::salt::lod::stage::LodConfig { - span: crate::math::Log2::new(1).expect("1 lies below the shift width"), - max_tile_depth: 1, - }) - .expect("the hand grid is valid"); - - let overlay = ArrivalOverlay::empty(); - let cut = schedule - .cut(&overlay, grid, CutOffset::ZERO) - .expect("k = 0 lies on the key width"); - let root = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - - // d(0) = 1: a claims depth 0 and c claims depth 1 (a's depth-1 cell differs from c's). - // Hand-derivation: at depth 0 all four share the root cell, rank 0 (a) claims it; at depth 1 - // a and d share cell (0,0) - taken cells block b and d - while c's cell (1,0) is free and - // c claims it; at depth 2 d's cell parts from a's and d claims it, which k = 0 clamps into - // the catch-all beside b. - let delta = cut.delta(0, root); - assert_eq!( - delta.rows, - [0, 2].map(|n| ViewRow::Base(BasePosition::from_u32(n))), - "the root delivers a then c", - ); - assert_eq!(delta.runs, vec![1, 1], "one first-occupant per depth"); - assert_eq!(delta.first_bucket, 0); - - // The terminal total delivers everything; b and d sit in the catch-all in Morton order: - // b's key interleaves x-bit 28 and y-bit 28 (bits 57 and 56), d's interleaves x-bit 30 and - // y-bit 28 (bits 61 and 57), so b < d and the tail reads b then d. - let total = cut.total(1, root); - assert_eq!( - total.rows, - [0, 2, 1, 3].map(|n| ViewRow::Base(BasePosition::from_u32(n))), - "the catch-all takes b and d" - ); - assert_eq!(total.runs, vec![1, 1, 2]); - - assert_eq!(cut.root_delivered(), 2); - assert_eq!(cut.min_resolution(), 2, "the catch-all is occupied"); - assert_eq!( - cut.bucket_of(BasePosition::new(1)), - Some(Depth::new(2).expect("2 is a depth")) - ); - assert_eq!( - cut.bucket_of(BasePosition::new(4)), - None, - "position 4 is not in the view" - ); - - // k = 1 deepens the catch-all to 3: d keeps its natural depth-2 bucket and only b stays - // terminal, so bucket-major order now reads d before b. - let deeper = schedule - .cut(&overlay, grid, CutOffset::new(1)) - .expect("k = 1 lies on the key width"); - let total = deeper.total(1, root); - assert_eq!( - total.rows, - [0, 2, 3, 1].map(|n| ViewRow::Base(BasePosition::from_u32(n))) - ); - assert_eq!( - total.runs, - vec![1, 1, 1, 1], - "d parts from the catch-all at k = 1" - ); -} - -/// Arrivals extend the hand cascade under the same first-occupant law. -/// -/// One arrival shares a's complete key with the worst rank, so it takes the catch-all -/// exactly as a co-located fitted row would, ordered after its better-ranked cell-mates. The -/// other arrival occupies a cell of its own and claims it at the first depth apart from a, -/// exactly as a fitted first occupant would. -#[test] -fn hand_cascade_places_arrivals_by_the_same_law() { - let fitted = MortonKey::new(0x1000_0000, 0x1000_0000); - let colocated = fitted; - let apart = MortonKey::new(0xC000_0000, 0x2000_0000); - - let arrival_row = |index: u32| ArrivalRow { - identity: ArchivedEntityId { - web_id: uuid::Uuid::from_u128(0xAB).into(), - entity_uuid: uuid::Uuid::from_u128(u128::from(index) + 1).into(), - }, - position: crate::math::Vec2::new(0.25, -0.5), - wire: crate::serve::WireRow::pinned(index), - legend: crate::dataset::auxiliary::OwnedLegend::new( - crate::identity::OntologyRowId::new(0), - crate::dataset::auxiliary::Label::new("arrival"), - ), - }; - - let rows = vec![ - ScopeRow { - vessel: ViewRow::Base(BasePosition::new(0)), - key: fitted, - rank: ImportanceRank::from_u32(0), - }, - ScopeRow { - vessel: ViewRow::Arrival(ArrivalIndex::from_u32(0)), - key: colocated, - rank: ImportanceRank::from_u32(1), - }, - ScopeRow { - vessel: ViewRow::Arrival(ArrivalIndex::from_u32(1)), - key: apart, - rank: ImportanceRank::from_u32(2), - }, - ]; - let schedule = ScopeSchedule::over( - rows, - Box::new_in( - [arrival_row(0), arrival_row(1)], - MemoryUsageAllocator::global(), - ), - ); - - let grid = crate::serve::grid::Grid::new(crate::salt::lod::stage::LodConfig { - span: crate::math::Log2::new(1).expect("1 lies below the shift width"), - max_tile_depth: 1, - }) - .expect("the hand grid is valid"); - let overlay = ArrivalOverlay::empty(); - let cut = schedule - .cut(&overlay, grid, CutOffset::ZERO) - .expect("k = 0 lies on the key width"); - let root = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - - // d(0) = 1: the fitted row claims depth 0, the apart arrival claims its own depth-1 cell, - // and the co-located arrival never claims a cell, so the root delivers exactly two rows. - let delta = cut.delta(0, root); - assert_eq!( - delta.rows, - vec![ - ViewRow::Base(BasePosition::new(0)), - ViewRow::Arrival(ArrivalIndex::from_u32(1)), - ], - "the apart arrival is a first occupant like any other row" - ); - - // The terminal total adds the co-located arrival in the catch-all, after its cell-mate. - let total = cut.total(1, root); - assert_eq!( - total.rows, - vec![ - ViewRow::Base(BasePosition::new(0)), - ViewRow::Arrival(ArrivalIndex::from_u32(1)), - ViewRow::Arrival(ArrivalIndex::from_u32(0)), - ], - "the co-located arrival takes the catch-all" - ); - assert_eq!(total.runs, vec![1, 1, 1]); - - // The position lookup answers the base domain, and the arrival table answers by index. - assert_eq!(cut.bucket_of(BasePosition::new(0)), Some(Depth::MIN)); - assert_eq!(cut.arrivals().len(), 2); - assert_eq!( - AsRef::::as_ref( - &cut.arrivals()[ArrivalIndex::from_u32(0)].legend - ) - .label(), - crate::dataset::auxiliary::Label::new("arrival") - ); -} - -/// An empty view builds an empty schedule: nothing delivers, nothing descends. -#[test] -fn empty_view_delivers_nothing() { - let schedule = ScopeSchedule::over(Vec::new(), Box::new_in([], MemoryUsageAllocator::global())); - let grid = crate::serve::grid::Grid::new(crate::salt::lod::stage::LodConfig { - span: crate::math::Log2::new(1).expect("1 lies below the shift width"), - max_tile_depth: 1, - }) - .expect("the hand grid is valid"); - let overlay = ArrivalOverlay::empty(); - let cut = schedule - .cut(&overlay, grid, CutOffset::ZERO) - .expect("k = 0 lies on the key width"); - let root = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - - let delta = cut.delta(0, root); - assert!(delta.rows.is_empty()); - assert_eq!(delta.runs, vec![0, 0], "empty runs keep their slots"); - assert_eq!(cut.root_delivered(), 0); - assert_eq!(cut.min_resolution(), 0); - assert_eq!(cut.children(0, root), 0); -} - -/// The reverse-rank gather (`ScopeSchedule::of`) equals a naive forward gather - position order, -/// sorted by the corpus rank column, re-indexed dense - for EVERY node mask over the fixture. -/// -/// The production gather traverses `position_of_rank` and never reads the forward rank column, so -/// this equality witnesses the loaded reverse against the column it claims to invert, over -/// saturated, empty, singleton, one-hidden, prefix, suffix, and random views. -#[tokio::test] -async fn scope_of_equals_a_rank_sorted_forward_gather_for_every_node_mask() { - let (_generation, atlas) = publish("morton-proof-exhaustive-masks").await; - - let row_ids = atlas.rows.view(); - let ranks = atlas.ranks.view(); - let count = u32::try_from(atlas.morton.count()).expect("fixture counts fit u32"); - - // Structured masks first, then a deterministic random sweep. - let mut masks: Vec> = vec![ - Vec::new(), // saturated - (0..count).collect(), // empty view - (0..count).step_by(2).collect(), // every other row - (0..count.saturating_sub(1)).collect(), // one visible row - (count >> 1..count).collect(), // a position suffix - (0..count >> 1).collect(), // a position prefix - ]; - masks.extend((0..count).map(|row| vec![row])); // each single hidden row - - let mut state = 0x5EED_0F0F_5EED_0F0F_u64; - for _ in 0..256 { - let bits = replay_rng(&mut state); - masks.push( - (0..count) - .filter(|&row| bits & (1 << (row & 63)) != 0) - .collect(), - ); - } - - for hidden in &masks { - let proof = mask_hiding(&atlas, hidden); - - let built = ScopeSchedule::of(&atlas, &proof, PlacementCohort::EMPTY); - - // Gather forward in position order and sort by the corpus rank column, then re-index the - // ranks dense by enumeration. The result is the input `of` derives from the reverse - // column alone. - let mut naive_rows: Vec = row_ids - .iter_enumerated() - .filter(|&(_, &row)| proof.contains(row)) - .map(|(position, _)| ScopeRow { - vessel: ViewRow::Base(position), - key: atlas.morton.code(position), - rank: ranks[position], - }) - .collect(); - naive_rows.sort_unstable_by_key(|row| row.rank); - for (dense, row) in naive_rows.iter_mut().enumerate() { - row.rank = ImportanceRank::from_usize(dense); - } - let naive = - ScopeSchedule::over(naive_rows, Box::new_in([], MemoryUsageAllocator::global())); - - assert_eq!( - format!("{built:?}"), - format!("{naive:?}"), - "mask {hidden:?} builds a different schedule through the reverse column" - ); - } -} - -/// The generation's saturated memo is byte-identical to a schedule built directly under the -/// saturated mask proof - equality of content, past the sharing the memo's own test pins. -#[tokio::test] -async fn saturated_memo_equals_a_directly_built_scope_schedule() { - let (_generation, atlas) = publish("morton-proof-saturated-content").await; - - let shared = ViewSchedule::of(&atlas, &mask_hiding(&atlas, &[]), PlacementCohort::EMPTY); - let ViewSchedule::Scope(shared, _) = &shared else { - panic!("a saturated mask is a declared scope"); - }; - - let direct = ScopeSchedule::of(&atlas, &mask_hiding(&atlas, &[]), PlacementCohort::EMPTY); - assert_eq!( - format!("{shared:?}"), - format!("{direct:?}"), - "the memo serves different bytes than a direct build" - ); - - let full = ScopeSchedule::of( - &atlas, - &VisibilityProof::full_visibility(), - PlacementCohort::EMPTY, - ); - assert_eq!( - format!("{direct:?}"), - format!("{full:?}"), - "a saturated mask and the full proof gather different views" - ); -} - -/// A deterministic xorshift64* stream for adversarial cases without new dependencies. -fn replay_rng(state: &mut u64) -> u64 { - *state ^= *state << 13; - *state ^= *state >> 7; - *state ^= *state << 17; - state.wrapping_mul(0x2545_F491_4F6C_DD1D) -} - -/// Draws a value below `bound` from the stream. -#[expect( - clippy::integer_division_remainder_used, - clippy::cast_possible_truncation, - reason = "a bounded draw from the deterministic stream folds the word into the bound" -)] -fn replay_draw(state: &mut u64, bound: usize) -> usize { - (replay_rng(state) as usize) % bound -} - -/// The shared depth of two keys through `prefix` comparison alone, for the replay oracle. -fn replay_shared_depth(left: MortonKey, right: MortonKey) -> u8 { - (0..=32_u8) - .rev() - .find(|&at| { - let at = Depth::new(at).expect("the oracle sweeps the documented domain"); - left.prefix(at) == right.prefix(at) - }) - .expect("depth zero prefixes are always equal") -} - -/// Each row's natural bucket by the quadratic law, written without the production cascade. -fn replay_natural_buckets(rows: &[ScopeRow]) -> Vec { - rows.iter() - .map(|row| { - let mut best: Option = None; - for other in rows { - if other.rank < row.rank { - let shared = replay_shared_depth(row.key, other.key); - best = Some(best.map_or(shared, |held| held.max(shared))); - } - } - - best.map_or(0, |shared| (shared + 1).min(Depth::MAX.get())) - }) - .collect() -} - -/// Hand-built adversarial row sets, each with pairwise-distinct ranks and positions. -fn replay_row_sets() -> Vec> { - let row = |position: u32, key: u64, rank: u32| ScopeRow { - vessel: ViewRow::Base(BasePosition::from_u32(position)), - key: MortonKey::from_bits(key), - rank: ImportanceRank::from_u32(rank), - }; - - let mut state = 0x00DD_B01D_FACE_D00D_u64; - - let mut sets = vec![ - // Nothing, then one row. - Vec::new(), - vec![row(7, 0xDEAD_BEEF_0000_0000, 0)], - // Every row of this set shares one key, filling the catch-all. - (0..6) - .map(|at| row(at, 0xAAAA_0000_0000_1111, at)) - .collect(), - // Two co-resident clusters parting at the first subdivision. - (0..6) - .map(|at| row(at, u64::from(at & 1) << 63, at)) - .collect(), - // A nested-prefix chain, key `i` sharing exactly depth `i` with key zero. - (0..16) - .map(|at| row(at, 1_u64 << (63 - 2 * at), at)) - .collect(), - ]; - - // The chain again under monotone-descending and alternating ranks over key order. - let chain = |at: u32| 1_u64 << (63 - 2 * at); - sets.push((0..16).map(|at| row(at, chain(at), 15 - at)).collect()); - sets.push( - (0..16) - .map(|at| { - let rank = if at & 1 == 0 { at >> 1 } else { 15 - (at >> 1) }; - row(at, chain(at), rank) - }) - .collect(), - ); - - // Random draws from a four-key pool, and full-width randoms. - let pool = [ - 0x1234_5678_9ABC_DEF0_u64, - 0x1234_5678_9ABC_DEF1, - 0x1234_5678_0000_0000, - 0x9234_5678_9ABC_DEF0, - ]; - let mut ranks: Vec = (0..24).collect(); - for at in (1..ranks.len()).rev() { - let swap = replay_draw(&mut state, at + 1); - ranks.swap(at, swap); - } - sets.push( - (0..24) - .map(|at| { - let key = pool[replay_draw(&mut state, pool.len())]; - row(at, key, ranks[at as usize]) - }) - .collect(), - ); - sets.push( - (0..24) - .map(|at| row(at, replay_rng(&mut state), ranks[at as usize])) - .collect(), - ); - - sets -} - -/// The occupied-children bitmask by the quadratic law: bit `i` is one exactly when child `i` -/// holds a row whose clamped bucket exceeds the cut. -fn replay_children_mask(rows: &[ScopeRow], clamped: &[u8], cell: MortonCell, cut_depth: u8) -> u8 { - cell.children().map_or(0, |children| { - children - .iter() - .enumerate() - .fold(0_u8, |bits, (index, child)| { - let occupied = rows - .iter() - .zip(clamped) - .any(|(row, &bucket)| child.contains(row.key) && bucket > cut_depth); - bits | (u8::from(occupied) << index) - }) - }) -} - -/// One bound cut with the replay context its oracle assertions read. -/// -/// The driver walks every adversarial row set and every admissible offset, so a facet test -/// holds one query's law and the sweep stays shared. -struct ReplayCase<'cut> { - /// The row set's index in [`replay_row_sets`], for assertion messages. - case: usize, - /// The delivery-cut offset. - k: u8, - /// The grid's span, `span_log2`. - span: u8, - /// The grid's deepest served zoom. - max_tile: u8, - /// The cut's deepest bucket, `max_tile + span + k`. - deepest: u8, - /// The case's rows. - rows: &'cut [ScopeRow], - /// Each row's natural bucket by the quadratic law, row-parallel. - natural: &'cut [u8], - /// The natural buckets clamped to the cut's deepest, row-parallel. - clamped: &'cut [u8], - /// The bound cut under test. - cut: &'cut super::cut::ScheduleCut<'cut>, -} - -/// Hands every adversarial row set's every admissible bound cut to `check`. -fn replay_cuts(check: impl Fn(ReplayCase<'_>)) { - let grid = Grid::new(FIXTURE_LOD).expect("the fixture lod lies on the key width"); - let span = grid.span_log2(); - let max_tile = grid.max_tile_depth(); - - for (case, rows) in replay_row_sets().into_iter().enumerate() { - let schedule = ScopeSchedule::over( - rows.clone(), - Box::new_in([], MemoryUsageAllocator::global()), - ); - let natural = replay_natural_buckets(&rows); - let overlay = ArrivalOverlay::empty(); - - for k in 0..=33_u8 { - let Ok(cut) = schedule.cut(&overlay, grid, CutOffset::new(k)) else { - continue; - }; - let deepest = max_tile + span + k; - let clamped: Vec = natural.iter().map(|&at| at.min(deepest)).collect(); - check(ReplayCase { - case, - k, - span, - max_tile, - deepest, - rows: &rows, - natural: &natural, - clamped: &clamped, - cut: &cut, - }); - } - } -} - -/// Every occupied cell of the zoom, plus one cell nothing occupies. -fn replay_cells(rows: &[ScopeRow], zoom_grid: Depth) -> Vec { - let mut cells: Vec = rows.iter().map(|row| row.key.cell(zoom_grid)).collect(); - cells.sort_unstable_by_key(|cell| cell.min_key()); - cells.dedup(); - cells.push(MortonKey::from_bits(0x5555_5555_5555_5555).cell(zoom_grid)); - cells -} - -/// The expected delivery of `cell` over buckets `first..=last`: bucket-major, (key, -/// rank)-ascending per run, the deepest bucket absorbing the catch-all tail. -fn replay_delivery( - rows: &[ScopeRow], - natural: &[u8], - deepest: u8, - cell: MortonCell, - first: u8, - last: u8, -) -> (Vec, Vec) { - let mut positions = Vec::new(); - let mut runs = Vec::new(); - for bucket in first..=last { - let mut members: Vec<&ScopeRow> = rows - .iter() - .zip(natural) - .filter(|&(row, &at)| { - cell.contains(row.key) - && if bucket == deepest { - at >= deepest - } else { - at == bucket - } - }) - .map(|(row, _)| row) - .collect(); - members.sort_unstable_by_key(|row| (row.key, row.rank)); - runs.push(u32::try_from(members.len()).expect("fixture rows fit u32")); - positions.extend(members.iter().map(|row| row.vessel)); - } - (positions, runs) -} - -/// Binding admits an offset exactly when `max_tile + span + k` lies on the key width, and the -/// admitted cut's deepest bucket is that sum. -#[test] -fn cut_offset_admissibility() { - let grid = Grid::new(FIXTURE_LOD).expect("the fixture lod lies on the key width"); - let span = grid.span_log2(); - let max_tile = grid.max_tile_depth(); - - for (case, rows) in replay_row_sets().into_iter().enumerate() { - let schedule = ScopeSchedule::over(rows, Box::new_in([], MemoryUsageAllocator::global())); - let overlay = ArrivalOverlay::empty(); - - for k in 0..=33_u8 { - let bound = schedule.cut(&overlay, grid, CutOffset::new(k)); - let admissible = - u16::from(max_tile) + u16::from(span) + u16::from(k) <= u16::from(Depth::MAX.get()); - match bound { - Ok(cut) => { - assert!( - admissible, - "case {case}: binding accepted an offset past the width" - ); - assert_eq!( - cut.deepest().get(), - max_tile + span + k, - "case {case} k {k}" - ); - } - Err(_) => assert!( - !admissible, - "case {case}: binding refused an admissible offset" - ), - } - } - } -} - -/// Root delivery counts the rows at the root's cut depth and the resolution is the deepest -/// clamped bucket. -#[test] -fn root_aggregates_vs_replay() { - replay_cuts( - |ReplayCase { - case, - k, - span, - rows, - clamped, - cut, - .. - }| { - let cut_zero = span + k; - let delivered = rows - .iter() - .zip(clamped) - .filter(|&(_, &bucket)| bucket <= cut_zero) - .count() as u64; - assert_eq!(cut.root_delivered(), delivered, "case {case} k {k}"); - let resolution = clamped.iter().copied().max().map_or(0, u64::from); - assert_eq!(cut.min_resolution(), resolution, "case {case} k {k}"); - }, - ); -} - -/// Position lookups answer the clamped natural bucket and its first zoom, and a position the -/// view never held answers [`None`]. -#[test] -fn position_lookups_vs_natural_buckets() { - replay_cuts( - |ReplayCase { - case, - k, - span, - rows, - clamped, - cut, - .. - }| { - for (row, &bucket) in rows.iter().zip(clamped) { - let ViewRow::Base(position) = row.vessel else { - unreachable!("the replay sets hold base rows alone") - }; - let held = cut.bucket_of(position).map(Depth::get); - assert_eq!(held, Some(bucket), "case {case} k {k}"); - let zoom = cut.first_zoom(position); - assert_eq!( - zoom, - Some(bucket.saturating_sub(span + k)), - "case {case} k {k}" - ); - } - assert_eq!(cut.bucket_of(BasePosition::from_u32(9_999)), None); - assert_eq!(cut.first_zoom(BasePosition::from_u32(9_999)), None); - }, - ); -} - -/// A total delivery carries every bucket from the root to the cell's cut depth. -#[test] -fn total_delivery_bucket_major() { - replay_cuts( - |ReplayCase { - case, - k, - span, - max_tile, - deepest, - rows, - natural, - cut, - .. - }| { - for zoom in 0..=max_tile { - let cut_depth = zoom + span + k; - let zoom_grid = Depth::new(zoom).expect("served zooms lie within the key width"); - for cell in replay_cells(rows, zoom_grid) { - let total = cut.total(zoom, cell); - let (positions, runs) = - replay_delivery(rows, natural, deepest, cell, 0, cut_depth); - assert_eq!(total.rows, positions, "case {case} k {k} z {zoom}"); - assert_eq!(total.runs, runs, "case {case} k {k} z {zoom}"); - assert_eq!(total.first_bucket, 0); - } - } - }, - ); -} - -/// A delta delivery starts at the cell's cut depth, except the root's, which delivers whole. -#[test] -fn delta_delivery_first_bucket() { - replay_cuts( - |ReplayCase { - case, - k, - span, - max_tile, - deepest, - rows, - natural, - cut, - .. - }| { - for zoom in 0..=max_tile { - let cut_depth = zoom + span + k; - let zoom_grid = Depth::new(zoom).expect("served zooms lie within the key width"); - for cell in replay_cells(rows, zoom_grid) { - let delta = cut.delta(zoom, cell); - let first = if zoom == 0 { 0 } else { cut_depth }; - let (positions, runs) = - replay_delivery(rows, natural, deepest, cell, first, cut_depth); - assert_eq!(delta.rows, positions, "case {case} k {k} z {zoom}"); - assert_eq!(delta.runs, runs, "case {case} k {k} z {zoom}"); - assert_eq!(delta.first_bucket, first); - } - } - }, - ); -} - -/// The occupied-children bitmask matches the quadratic law, and a cut at or past the deepest -/// bucket has no occupied children. -#[test] -fn children_mask_vs_quadratic_law() { - replay_cuts( - |ReplayCase { - case, - k, - span, - max_tile, - deepest, - rows, - clamped, - cut, - .. - }| { - for zoom in 0..=max_tile { - let cut_depth = zoom + span + k; - let zoom_grid = Depth::new(zoom).expect("served zooms lie within the key width"); - for cell in replay_cells(rows, zoom_grid) { - let mask = cut.children(zoom, cell); - let expected = if cut_depth >= deepest { - 0_u8 - } else { - replay_children_mask(rows, clamped, cell, cut_depth) - }; - assert_eq!(mask, expected, "case {case} k {k} z {zoom}"); - } - } - }, - ); -} diff --git a/libs/@local/graph/atlas/src/serve/secret.rs b/libs/@local/graph/atlas/src/serve/secret.rs deleted file mode 100644 index c8361641e65..00000000000 --- a/libs/@local/graph/atlas/src/serve/secret.rs +++ /dev/null @@ -1,63 +0,0 @@ -//! The wire secret. -//! -//! Every wire-facing derivation keys from this server-held key material. The row-id codec's -//! per-generation permutation draws its round keys from the value, and [`WireSecret`] is the -//! configuration boundary. A value of the type is always a full-width key, so configuration parsing -//! rejects a weak or malformed secret before any key derivation reaches it. -//! -//! The format is exact. A secret is 32 bytes, configured as 64 lowercase hexadecimal characters in -//! the crate's canonical hexadecimal form. Rejecting every other shape keeps the key space honest, -//! because a memorable passphrase fails to parse rather than passing as low-entropy key material. -//! Generate one with `openssl rand -hex 32`. -//! -//! The module is crate-internal. Its examples carry `ignore` and spell each call as an in-crate -//! caller writes it. - -use core::fmt; - -use crate::integrity::SecretHexBytes; - -/// The server secret behind the wire row-id codec. -/// -/// A 256-bit key, held for the lifetime of an opened generation. The value zeroes its bytes on -/// drop, and formatting one for diagnostics is safe by construction: the [`fmt::Debug`] form is -/// fully redacted. -/// -/// The configured form, exactly 64 lowercase hexadecimal characters, decodes as a -/// [`SecretHexBytes`] and converts through [`From`]. -#[derive(Clone)] -pub(crate) struct WireSecret(SecretHexBytes<{ Self::BYTES }>); - -impl WireSecret { - /// The key width, bytes. - pub(crate) const BYTES: usize = 32; - - /// Wraps raw key bytes. - #[must_use] - #[cfg(test)] // The serve tests pin fixture secrets. - pub(crate) const fn new(bytes: [u8; Self::BYTES]) -> Self { - Self(SecretHexBytes::new(bytes)) - } - - /// Views the key bytes. - pub(crate) const fn as_bytes(&self) -> &[u8] { - self.0.as_bytes() - } - - /// Views the key as the typed value key derivations take. - pub(crate) const fn hex_bytes(&self) -> &SecretHexBytes<{ Self::BYTES }> { - &self.0 - } -} - -impl fmt::Debug for WireSecret { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - fmt.debug_tuple("WireSecret").finish_non_exhaustive() - } -} - -impl From> for WireSecret { - fn from(bytes: SecretHexBytes<{ Self::BYTES }>) -> Self { - Self(bytes) - } -} diff --git a/libs/@local/graph/atlas/src/serve/tests/arrival.rs b/libs/@local/graph/atlas/src/serve/tests/arrival.rs deleted file mode 100644 index 0e31dac3582..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/arrival.rs +++ /dev/null @@ -1,1183 +0,0 @@ -//! Cohort-serving witnesses: placed arrivals answer through the entry's retained cohort. -//! -//! Every case runs the served translate path with a real published snapshot, folded, classified, -//! and placed exactly as the consumer records them, so the witnesses cover the served path rather -//! than the map lookups alone. Translate is the first arrival-sensitive read. It resolves -//! identities against the cohort and encodes slots under the cohort's universe, and the ingress -//! capture's withdrawn identity set filters what the cohort retains. Each case carries a same-path -//! control whose delta touches nothing the request names. - -#![expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] - -use alloc::sync::Arc; - -use hash_graph_postgres_store::store::{EntityEnd, EntityEvent, EntityUpdate}; -use hash_graph_temporal_versioning::Timestamp; -use hashql_core::{ - collections::fast_hash_set, - id::{Id as _, IdSlice, IdVec, bit_vec::DenseBitSet}, -}; -use type_system::knowledge::entity::{ - EntityId, - id::{EntityEditionId, EntityUuid}, -}; -use uuid::Uuid; - -use super::{ - Atlas, Bound, CutOffset, FIXTURE_LOD, FULL, HEAD, ServeLimits, TileLimits, UntouchedStore, - coordinate_of, edges_request, entity_string_of, full_grid, head_global, locate_request, - mask_hiding, publish, request, section, test_codec, -}; -use crate::{ - bitset::{CompressedBitSet, DenseBitSlice}, - dataset::auxiliary::{Icon, Label, OwnedIcon, OwnedLabel, OwnedLegend}, - identity::{BasePosition, EdgeRowId, NodeRowId, OntologyRowId}, - math::Vec2, - morton::{Depth, MortonKey}, - postgres::{ - Classification, - id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedOntologyTypeUuid}, - }, - salt::{ - lod::stage::WIRE_FRAME, - wire::{ - Mode, - locate::{LocateResponse, LocateTrailer, PropertyMap}, - tile::{DeliveredSet, TileCoordinate, TileHead, TileResponse, TileTrailer}, - }, - }, - serve::{ - EdgesLimits, LocateRequest, ViewCensus, VisibilityProof, - delta::{ - DeltaEvent, DeltaRegister, DeltaRevision, DeltaSnapshot, PlacementCohort, - ProjectedArrival, - }, - hydrate::{ - DetailError, LocateHydration, LocateLinkHydration, LocateNodeHydration, LocateOrder, - LocateStore, - }, - locate::SourceSubject, - neighbourhood::EdgeColumns, - schedule::{ - ArrivalIndex, ArrivalOverlay, ArrivalRow, ScopeSchedule, Splice, ViewRow, ViewSchedule, - }, - tile::TileDetail, - translate::{TranslateLimits, TranslateRequest}, - }, -}; - -/// The arrival's seed, past every node and edge seed the fixture generation fits. -const ARRIVAL_SEED: u8 = 0xA0; - -/// The store-form entity id the seeding rule gives `seed`. -fn store_id(seed: u8) -> EntityId { - EntityId { - web_id: type_system::principal::actor_group::WebId::new(Uuid::from_bytes([seed; 16])), - entity_uuid: EntityUuid::new(Uuid::from_bytes([seed ^ 0xFF; 16])), - draft_id: None, - } -} - -/// The identity-table key the seeding rule gives `seed`. -fn archived_id(seed: u8) -> ArchivedEntityId { - ArchivedEntityId { - web_id: Uuid::from_bytes([seed; 16]).into(), - entity_uuid: ArchivedEntityUuid::from_bytes( - Uuid::from_bytes([seed ^ 0xFF; 16]).into_bytes(), - ), - } -} - -/// Folds, classifies, and places live arrivals, publishing the snapshot a resolution reads. -/// -/// The events travel the consumer's own conversion, and each placement takes the register's own -/// slot allocation with slots ascending in the given order, so the snapshot is the publication a -/// scope resolution would bind rather than a hand-assembled equivalent. -/// The fixture arrival's representative type, unknown to the generation, so the register's own -/// extension allocates its ontology row at the baked bound. -fn arrival_type() -> ArchivedOntologyTypeUuid { - ArchivedOntologyTypeUuid::from(Uuid::from_u128(0xA771)) -} - -/// The legend the fixture arrival publishes: the extension's first ontology row. -fn arrival_legend(atlas: &Atlas) -> OwnedLegend { - OwnedLegend::new( - OntologyRowId::from_usize(atlas.ontology_universe().size()), - Label::new("arrival"), - ) -} - -fn arriving_all(atlas: &Atlas, arrivals: &[(u8, Vec2)]) -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - for &(seed, wire) in arrivals { - let event = EntityEvent::Updated(EntityUpdate { - entity: store_id(seed), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - register - .classify(archived_id(seed), Classification::Node) - .expect("the fixture stays inside the edge universe"); - register - .place( - archived_id(seed), - &ProjectedArrival { - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - position: wire, - label: OwnedLabel::from("arrival"), - icon: OwnedIcon::from("arrival-icon"), - representative: arrival_type(), - }, - atlas, - ) - .expect("the fixture universe is far from the wire's row domain"); - } - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(1), - ) -} - -/// Folds, classifies, and places one live arrival, publishing the snapshot a resolution reads. -fn arriving(atlas: &Atlas, wire: Vec2) -> DeltaSnapshot { - arriving_all(atlas, &[(ARRIVAL_SEED, wire)]) -} - -/// Folds one `Ended` event per seed into an ingress snapshot withdrawing those identities. -fn withdrawing(atlas: &Atlas, seeds: &[u8]) -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - for &seed in seeds { - let event = EntityEvent::Ended(EntityEnd { - entity: store_id(seed), - ended_at: Timestamp::from_unix_timestamp(2), - }); - register.apply(DeltaEvent::from(&event)); - } - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(2), - ) -} - -/// A scoped proof admitting every fitted row outside `hidden`, plus the cohort slots. -/// -/// The shape the mask builder produces for a scope whose store resolution admitted the -/// arrivals: the fitted mask with the slots widened in. Hiding a fitted row keeps the proof -/// off the saturated-memo arm, and an empty `hidden` is the saturated shape, which reads the -/// shared cascade with the arrivals as its overlay. -fn widened(atlas: &Atlas, hidden: &[u32], slots: &[NodeRowId]) -> VisibilityProof { - let rows = u32::try_from(atlas.row_ids().len()).expect("fixture domains fit u32"); - let edges = u32::try_from(atlas.endpoints.view().len()).expect("fixture domains fit u32"); - - VisibilityProof::from_masks( - CompressedBitSet::from_rows( - (0..rows) - .filter(|row| !hidden.contains(row)) - .map(NodeRowId::from_u32) - .chain(slots.iter().copied()), - ), - CompressedBitSet::from_rows((0..edges).map(EdgeRowId::from_u32)), - fast_hash_set(), - ) -} - -/// A wire coordinate whose deepest-zoom cell holds no fitted row, with that cell's tile address. -/// -/// The candidates sweep distinct quadrants of the wire square, so one of them lands apart from -/// the fixture's handful of points and the tile witnesses the arrival alone. -pub(super) fn vacant_cell(atlas: &Atlas) -> (Vec2, TileCoordinate) { - let depth = Depth::new(FIXTURE_LOD.max_tile_depth).expect("the fixture depth is a depth"); - - 'candidates: for candidate in [ - Vec2::new(0.25, -0.5), - Vec2::new(-0.75, 0.75), - Vec2::new(0.8, 0.8), - Vec2::new(-0.3, -0.9), - Vec2::new(0.05, 0.6), - Vec2::new(-0.9, 0.1), - ] { - let [x, y] = WIRE_FRAME.quantize(candidate); - let cell = MortonKey::new(x, y).cell(depth); - for position in 0..atlas.row_ids().len() { - let position = BasePosition::from_usize(position); - if cell.contains(atlas.morton.code(position)) { - continue 'candidates; - } - } - - return (candidate, coordinate_of(cell)); - } - - unreachable!("every candidate cell holds a fixture point") -} - -/// The arrival's natural bucket under `proof`, clamped into `deepest`, by the quadratic law. -fn expected_bucket(atlas: &Atlas, proof: &VisibilityProof, wire: Vec2, deepest: u8) -> u8 { - let [x, y] = WIRE_FRAME.quantize(wire); - let key = MortonKey::new(x, y); - - let shared_depth = |left: MortonKey, right: MortonKey| { - (0..=Depth::MAX.get()) - .rev() - .find(|&at| { - let at = Depth::new(at).expect("the sweep stays on the documented domain"); - left.prefix(at) == right.prefix(at) - }) - .expect("depth zero prefixes are always equal") - }; - - let mut deepest_shared: Option = None; - for (position, &row) in atlas.row_ids().iter_enumerated() { - if proof.contains(row) { - let shared = shared_depth(key, atlas.morton.code(position)); - deepest_shared = Some(deepest_shared.map_or(shared, |held| held.max(shared))); - } - } - - deepest_shared.map_or(0, |shared| (shared + 1).min(deepest)) -} - -/// A scoped tile serves a placed arrival byte-exact against the directly built wire document: -/// the wire id and projected coordinate in the columns, the captured display in the trailer. -/// -/// The arrival's cell holds no fitted row, so every column of the response is the arrival's -/// alone and the expected envelope derives whole from the placement's own values. A second -/// resolution must produce identical bytes, which pins delivery against the cohort map's -/// iteration order. -#[tokio::test] -async fn scoped_tile_serves_placed_arrival_with_captured_display() { - let (_generation, atlas) = publish("arrival-tile").await; - let (wire, coordinate) = vacant_cell(&atlas); - let snapshot = arriving(&atlas, wire); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let cohort = PlacementCohort::of(Some(&snapshot)); - let proof = widened(&atlas, &[0], &[slot]); - - let mut request = request(coordinate.z, coordinate.x, coordinate.y, Mode::Total); - request.query.detail = TileDetail::Auxiliary; - - let assemble = || { - let bound = Bound::resolved(&atlas, &proof, cohort, CutOffset::ZERO); - atlas - .tile(&request, TileLimits::default(), bound.view(&atlas)) - .expect("the tile request is on the served grid") - }; - let bytes = assemble(); - assert_eq!(bytes, assemble(), "two resolutions serve identical bytes"); - - // d(z) at the deepest zoom is the catch-all, so the cut delivers buckets 0..=deepest and - // the children mask reads zero. The arrival's run sits at its natural bucket. - let deepest = FIXTURE_LOD.max_tile_depth + FIXTURE_LOD.span.get(); - let bucket = expected_bucket(&atlas, &proof, wire, deepest); - let runs: Vec = (0..=deepest).map(|at| u32::from(at == bucket)).collect(); - - let arrival_row = ArrivalRow { - identity: archived_id(ARRIVAL_SEED), - position: wire, - wire: test_codec(&atlas).encode(slot, snapshot.universe()), - legend: arrival_legend(&atlas), - }; - let expected = TileResponse { - head: TileHead { - generation: atlas.generation().digest(), - variant: 0, - coordinate, - mode: Mode::Total, - first_bucket: 0, - runs: &runs, - global: None, - children: 0, - }, - delivered: DeliveredSet::Positions(&[ViewRow::Arrival(ArrivalIndex::from_u32(0))]), - positions: atlas.positions(), - rows: atlas.wire_rows(), - arrivals: IdSlice::from_raw(core::slice::from_ref(&arrival_row)), - masks: None, - trailer: Some(TileTrailer { - labels: &[Label::new("arrival")], - icons: &[Icon::new("arrival-icon")], - }), - } - .encode(); - assert_eq!(bytes, expected, "the arrival tile is byte-exact"); -} - -/// An ingress withdrawal subtracts a retained arrival from tiles, leaving exactly the bytes a -/// view that never held it serves. -/// -/// The control withdraws an identity the cell never delivers and must leave the baseline bytes -/// untouched. -#[tokio::test] -async fn ingress_withdrawal_subtracts_retained_arrival_from_tiles() { - let (_generation, atlas) = publish("arrival-tile-withdrawn").await; - let (wire, coordinate) = vacant_cell(&atlas); - let snapshot = arriving(&atlas, wire); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let cohort = PlacementCohort::of(Some(&snapshot)); - let proof = widened(&atlas, &[0], &[slot]); - let request = request(coordinate.z, coordinate.x, coordinate.y, Mode::Total); - - let assemble = |delta: Option<&DeltaSnapshot>| { - let mut bound = Bound::resolved(&atlas, &proof, cohort, CutOffset::ZERO); - if let Some(delta) = delta { - bound = bound.withdrawing(delta); - } - atlas - .tile(&request, TileLimits::default(), bound.view(&atlas)) - .expect("the tile request is on the served grid") - }; - - let baseline = assemble(None); - - // The withdrawal leaves the bytes of a view that never held the arrival: same fitted mask, - // no slot, empty cohort. - let withdrawing_arrival = withdrawing(&atlas, &[ARRIVAL_SEED]); - let subtracted = assemble(Some(&withdrawing_arrival)); - let slotless = mask_hiding(&atlas, &[0]); - let never = { - let bound = Bound::resolved(&atlas, &slotless, PlacementCohort::EMPTY, CutOffset::ZERO); - atlas - .tile(&request, TileLimits::default(), bound.view(&atlas)) - .expect("the tile request is on the served grid") - }; - assert_eq!( - subtracted, never, - "the subtracted tile equals the never-held tile" - ); - assert_ne!(baseline, subtracted, "the baseline delivered the arrival"); - - // Same-path control: a withdrawal the cell never delivers moves nothing. - let withdrawing_other = withdrawing(&atlas, &[ARRIVAL_SEED ^ 0x11]); - let control = assemble(Some(&withdrawing_other)); - assert_eq!(control, baseline, "an unrelated withdrawal moves nothing"); -} - -/// The delivery-cut inputs never read the cohort: occupancy and census answer identically -/// under a widened proof and its slot-free counterpart. -/// -/// Both are position-bounded walks over the generation's columns, so the cut offset `k` an -/// issuance resolves from them cannot move when arrivals join a scope. -#[tokio::test] -async fn occupancy_and_census_never_read_cohort() { - let (_generation, atlas) = publish("arrival-occupancy").await; - let snapshot = arriving(&atlas, Vec2::new(0.25, -0.5)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - let with_slot = widened(&atlas, &[0], &[slot]); - let without = mask_hiding(&atlas, &[0]); - - assert_eq!( - atlas.visible_occupancy(&with_slot), - atlas.visible_occupancy(&without), - "the occupancy aggregate is position-bounded" - ); - assert_eq!( - atlas.census(&with_slot), - atlas.census(&without), - "the census is position-bounded" - ); - drop(snapshot); -} - -/// The translate request naming exactly the arrival. -fn ask() -> TranslateRequest { - TranslateRequest { - entity_ids: vec![entity_string_of(ARRIVAL_SEED)], - } -} - -/// Translate answers a placed arrival from the cohort, on its slot, under the grown universe. -/// -/// The wire id must agree with an independent codec derivation at the snapshot's own universe, -/// so the case pins that arrival egress reads the cohort's bound rather than the generation's. -/// The empty-cohort control runs the same request and must answer an absent key, which is the -/// resolution that read no publication. -#[tokio::test] -#[expect( - clippy::float_cmp, - reason = "the projected coordinate is copied verbatim into the response, so bit equality is \ - the intended test" -)] -async fn translate_answers_placed_arrival_from_entry_cohort() { - let (_generation, atlas) = publish("arrival-translate").await; - let wire = Vec2::new(0.25, -0.5); - let snapshot = arriving(&atlas, wire); - let key = entity_string_of(ARRIVAL_SEED); - - let response = atlas - .translate( - ask(), - TranslateLimits::default(), - &FULL, - None, - PlacementCohort::of(Some(&snapshot)), - ) - .expect("the request is under the cap"); - - let node = response - .nodes - .get(&key) - .expect("the cohort resolves the arrival"); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - assert_eq!( - node.id, - test_codec(&atlas).encode(slot, snapshot.universe()), - "the arrival encodes its slot under the cohort's universe" - ); - assert_eq!(node.x, wire.x(), "the projected coordinate answers"); - assert_eq!(node.y, wire.y(), "the projected coordinate answers"); - assert!( - response.edges.is_empty(), - "a node-classified arrival answers in the nodes map alone" - ); - - // The empty-cohort control runs the same request with no publication and must answer an - // absent key. - let unresolved = atlas - .translate( - ask(), - TranslateLimits::default(), - &FULL, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - assert!( - unresolved.nodes.is_empty() && unresolved.edges.is_empty(), - "an empty cohort answers absent keys" - ); -} - -/// A withdrawn arrival answers an absent key while the entry still retains its cohort. -/// -/// The filter is literal membership in the ingress capture's withdrawn identity set, in both -/// directions: the withdrawal hides the retained arrival on this request, and an ingress set no -/// longer holding the identity serves it again from the same retained cohort. The control -/// withdraws an identity the request never names and must leave the response equal to the -/// baseline. -#[tokio::test] -async fn withdrawn_arrival_answers_absent_key_while_cohort_retains_it() { - let (_generation, atlas) = publish("arrival-withdrawn").await; - let snapshot = arriving(&atlas, Vec2::new(0.25, -0.5)); - let cohort = PlacementCohort::of(Some(&snapshot)); - let key = entity_string_of(ARRIVAL_SEED); - - let translate = |delta: Option<&DeltaSnapshot>| { - atlas - .translate(ask(), TranslateLimits::default(), &FULL, delta, cohort) - .expect("the request is under the cap") - }; - - let baseline = translate(None); - assert!(baseline.nodes.contains_key(&key), "the arrival resolves"); - - let hidden = translate(Some(&withdrawing(&atlas, &[ARRIVAL_SEED]))); - assert!( - !hidden.nodes.contains_key(&key), - "the ingress withdrawal hides the retained arrival" - ); - - // Same-path control: a withdrawal the request never names moves nothing, which is the - // literal-membership reading in the serving direction. - let control = translate(Some(&withdrawing(&atlas, &[ARRIVAL_SEED ^ 0x11]))); - assert_eq!(control, baseline, "an unrelated withdrawal moves nothing"); -} - -/// A scoped proof answers an arrival exactly when its widened mask admits the slot. -/// -/// The mask builder admits a placed arrival on its cohort slot, and this case pins the serving -/// end of that law. A node mask holding the slot serves the arrival, and the control proof, -/// which admits every fitted row and no slot, answers an absent key for the same cohort. -#[tokio::test] -async fn scoped_proof_admits_arrival_only_through_widened_mask() { - let (_generation, atlas) = publish("arrival-scoped").await; - let snapshot = arriving(&atlas, Vec2::new(0.25, -0.5)); - let cohort = PlacementCohort::of(Some(&snapshot)); - let key = entity_string_of(ARRIVAL_SEED); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - let translate = |proof: &VisibilityProof| { - atlas - .translate(ask(), TranslateLimits::default(), proof, None, cohort) - .expect("the request is under the cap") - }; - - // The widened mask holds the slot alone, the shape the mask builder produces for a scope - // whose store resolution admitted exactly this arrival. - let widened = VisibilityProof::from_masks( - CompressedBitSet::from_rows([slot]), - CompressedBitSet::::from_rows([]), - fast_hash_set(), - ); - let admitted = translate(&widened); - assert!( - admitted.nodes.contains_key(&key), - "the widened mask admits the arrival on its slot" - ); - - // The control admits every fitted row and no slot, so the same cohort answers nothing. - let fitted_only = translate(&mask_hiding(&atlas, &[])); - assert!( - !fitted_only.nodes.contains_key(&key), - "a mask without the slot hides the arrival" - ); -} - -/// A corpus tile serves a placed arrival byte-exact against the directly built wire document: -/// the wire id and projected coordinate in the columns, the captured display in the trailer. -/// -/// The operator proof admits every slot, so the corpus delivery splices the whole cohort. The -/// arrival's cell holds no fitted row, so every column of the response is the arrival's alone -/// and the expected envelope derives whole from the placement's own values: the delivered set -/// is one splice at index zero, with one delivered run at the arrival's natural bucket clamped -/// into the corpus catch-all. A second resolution must produce identical bytes, which pins delivery -/// against the cohort map's iteration order. -#[tokio::test] -async fn corpus_tile_serves_placed_arrival_with_captured_display() { - let (_generation, atlas) = publish("arrival-corpus-tile").await; - let (wire, coordinate) = vacant_cell(&atlas); - let snapshot = arriving(&atlas, wire); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let mut request = request(coordinate.z, coordinate.x, coordinate.y, Mode::Total); - request.query.detail = TileDetail::Auxiliary; - - let assemble = || { - let bound = Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO); - atlas - .tile(&request, TileLimits::default(), bound.view(&atlas)) - .expect("the tile request is on the served grid") - }; - let bytes = assemble(); - assert_eq!(bytes, assemble(), "two resolutions serve identical bytes"); - - // The corpus catch-all is the deepest zoom's cut, so the total delivery covers buckets - // 0..=deepest with the vacant cell contributing the arrival alone, and the deepest zoom's - // child bitmask reads zero. - let deepest = FIXTURE_LOD.max_tile_depth + FIXTURE_LOD.span.get(); - let bucket = expected_bucket(&atlas, &FULL, wire, deepest); - let runs: Vec = (0..=deepest).map(|at| u32::from(at == bucket)).collect(); - - let arrival_row = ArrivalRow { - identity: archived_id(ARRIVAL_SEED), - position: wire, - wire: test_codec(&atlas).encode(slot, snapshot.universe()), - legend: arrival_legend(&atlas), - }; - let expected = TileResponse { - head: TileHead { - generation: atlas.generation().digest(), - variant: 0, - coordinate, - mode: Mode::Total, - first_bucket: 0, - runs: &runs, - global: None, - children: 0, - }, - delivered: DeliveredSet::Spliced { - ranges: &[], - splices: &[Splice { - at: 0, - arrival: ArrivalIndex::from_u32(0), - }], - }, - positions: atlas.positions(), - rows: atlas.wire_rows(), - arrivals: IdSlice::from_raw(core::slice::from_ref(&arrival_row)), - masks: None, - trailer: Some(TileTrailer { - labels: &[Label::new("arrival")], - icons: &[Icon::new("arrival-icon")], - }), - } - .encode(); - assert_eq!(bytes, expected, "the corpus arrival tile is byte-exact"); -} - -/// The corpus schedule's root count and depth from the operator proof's census. -fn corpus_census(atlas: &Atlas) -> (u64, u64) { - match atlas.census(&FULL) { - ViewCensus::Corpus { - visible, - min_resolution, - .. - } => (visible, min_resolution), - ViewCensus::Scope { .. } => panic!("the operator proof censuses the corpus"), - } -} - -/// The corpus splice, the saturated overlay, and the folded cascade agree byte for byte on -/// every tile. -/// -/// A saturated scope reads the shared memo with the cohort as its overlay, and a directly built -/// cascade folds the same arrivals into its own slots: the overlay must reproduce the fold at -/// every zoom, cell, mode, and admissible offset. At the zero offset the corpus contract serves -/// the same rows through the range-splice path, so all three must coincide there, which pins the -/// splice arithmetic against the two schedule mechanisms. -/// -/// One arrival sits in a cell of its own and the other co-locates with fitted row zero at the -/// full key width, so the sweep crosses vacant cells, mid-run splices, the catch-all, and the -/// equal-key ordering law in one pass. The root coordinates then check the merged global -/// aggregates against the census and the arrivals' own buckets. -#[tokio::test] -async fn corpus_saturated_and_folded_arrival_deliveries_agree() { - let (_generation, atlas) = publish("arrival-parity").await; - let (vacant, _) = vacant_cell(&atlas); - let co_located = atlas.positions()[BasePosition::MIN]; - let snapshot = arriving_all( - &atlas, - &[(ARRIVAL_SEED, vacant), (ARRIVAL_SEED + 1, co_located)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let base = atlas.node_universe().size(); - let slots = [base, base + 1].map(NodeRowId::from_usize); - let saturated = widened(&atlas, &[], &slots); - - for k in 0..=1_u8 { - let offset = CutOffset::new(k); - let fold = ViewSchedule::Scope( - Arc::new(ScopeSchedule::of(&atlas, &saturated, cohort)), - ArrivalOverlay::empty(), - ); - let folded_bound = Bound { - proof: &saturated, - census: atlas.census(&saturated), - schedule: fold, - k: offset, - cohort, - delta: None, - }; - - for z in 0..=FIXTURE_LOD.max_tile_depth { - let cells = 1_u32 << z; - for (x, y) in (0..cells).flat_map(|x| (0..cells).map(move |y| (x, y))) { - for mode in [Mode::Delta, Mode::Total] { - let at = format!("k={k} {mode:?} {z}/{x}/{y}"); - let tile = request(z, x, y, mode); - - let memo = { - let bound = Bound::resolved(&atlas, &saturated, cohort, offset); - atlas - .tile(&tile, TileLimits::default(), bound.view(&atlas)) - .expect("the saturated scope serves") - }; - let folded = atlas - .tile(&tile, TileLimits::default(), folded_bound.view(&atlas)) - .expect("the folded cascade serves"); - assert_eq!( - memo, folded, - "{at}: the overlay reproduces the folded cascade" - ); - - if k == 0 { - let corpus = { - let bound = Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO); - atlas - .tile(&tile, TileLimits::default(), bound.view(&atlas)) - .expect("the operator contract serves") - }; - assert_eq!( - corpus, memo, - "{at}: the corpus splice parts from the cascade" - ); - } - } - } - } - } - - // The corpus root publishes the census merged with the overlay. The vacant arrival counts - // toward the visible aggregate exactly when its bucket lies on the root's cumulative - // schedule, while the co-located arrival takes the catch-all and never does. The resolution - // deepens to the deepest clamped arrival bucket. - let deepest = FIXTURE_LOD.max_tile_depth + FIXTURE_LOD.span.get(); - let vacant_bucket = expected_bucket(&atlas, &FULL, vacant, deepest); - assert_eq!( - expected_bucket(&atlas, &FULL, co_located, deepest), - deepest, - "a full-key co-location takes the catch-all", - ); - - let root = { - let bound = Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO); - atlas - .tile( - &request(0, 0, 0, Mode::Delta), - TileLimits::default(), - bound.view(&atlas), - ) - .expect("the root serves") - }; - let (visible, _, min_resolution) = head_global(section(&root, HEAD).expect("HEAD is present")) - .expect("the root carries the global aggregates"); - - let (census_visible, census_resolution) = corpus_census(&atlas); - let cut = FIXTURE_LOD.span.get(); - assert_eq!( - visible, - census_visible + u64::from(vacant_bucket <= cut), - "the root's visible count folds the delivered arrivals in", - ); - assert_eq!( - min_resolution, - census_resolution.max(u64::from(deepest)), - "the root's resolution reaches the deepest clamped arrival bucket", - ); - assert_eq!( - Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO) - .view(&atlas) - .min_resolution(), - min_resolution, - "the manifest's maxZoom reads the root's resolution", - ); -} - -/// An ingress withdrawal subtracts spliced arrivals from corpus tiles, leaving exactly the -/// bytes a view that never held them serves, and a fitted withdrawal beside a retained arrival -/// subtracts identically through the spliced and the gathered shapes. -/// -/// The control withdraws an identity the cohort never held and must leave the baseline bytes -/// untouched. -#[tokio::test] -async fn ingress_withdrawal_subtracts_spliced_arrival_from_corpus_tiles() { - let (_generation, atlas) = publish("arrival-corpus-withdrawn").await; - let (vacant, coordinate) = vacant_cell(&atlas); - let co_located = atlas.positions()[BasePosition::MIN]; - let snapshot = arriving_all( - &atlas, - &[(ARRIVAL_SEED, vacant), (ARRIVAL_SEED + 1, co_located)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let tile = request(coordinate.z, coordinate.x, coordinate.y, Mode::Total); - - let corpus = |cohort: PlacementCohort<'_>, - delta: Option<&DeltaSnapshot>, - tile: &crate::serve::TileRequest| { - let mut bound = Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO); - if let Some(delta) = delta { - bound = bound.withdrawing(delta); - } - atlas - .tile(tile, TileLimits::default(), bound.view(&atlas)) - .expect("the tile request is on the served grid") - }; - - let baseline = corpus(cohort, None, &tile); - - // Withdrawing both arrivals leaves the bytes of a view that never read a publication. - let withdrawing_both = withdrawing(&atlas, &[ARRIVAL_SEED, ARRIVAL_SEED + 1]); - let subtracted = corpus(cohort, Some(&withdrawing_both), &tile); - let never = corpus(PlacementCohort::EMPTY, None, &tile); - assert_eq!( - subtracted, never, - "the subtracted tile equals the never-held tile" - ); - assert_ne!(baseline, subtracted, "the baseline delivered the arrival"); - - // Same-path control: a withdrawal the cohort never held moves nothing. - let withdrawing_other = withdrawing(&atlas, &[ARRIVAL_SEED ^ 0x11]); - let control = corpus(cohort, Some(&withdrawing_other), &tile); - assert_eq!(control, baseline, "an unrelated withdrawal moves nothing"); - - // A fitted withdrawal in the co-located cell shifts the splice sitting behind it. The - // saturated scope subtracts the same rows through the gathered shape, so byte parity pins - // the spliced subtraction against it. - let fitted_seed = - u8::try_from(atlas.row_ids()[BasePosition::MIN].as_u32()).expect("fixture rows fit u8"); - let withdrawing_fitted = withdrawing(&atlas, &[fitted_seed]); - let depth = Depth::new(FIXTURE_LOD.max_tile_depth).expect("the fixture depth is a depth"); - let shared_cell = coordinate_of(atlas.morton.code(BasePosition::MIN).cell(depth)); - let shared_tile = request(shared_cell.z, shared_cell.x, shared_cell.y, Mode::Total); - - let base = atlas.node_universe().size(); - let saturated = widened(&atlas, &[], &[base, base + 1].map(NodeRowId::from_usize)); - let gathered = { - let bound = Bound::resolved(&atlas, &saturated, cohort, CutOffset::ZERO) - .withdrawing(&withdrawing_fitted); - atlas - .tile(&shared_tile, TileLimits::default(), bound.view(&atlas)) - .expect("the saturated scope serves") - }; - let spliced = corpus(cohort, Some(&withdrawing_fitted), &shared_tile); - assert_eq!( - spliced, gathered, - "both shapes subtract the fitted row and keep the arrival" - ); -} - -/// A store answering that every delivered entity resolves with no recorded detail. -/// -/// The store resolution answers and every store-derived column stays empty, so an expectation built -/// over it pins the in-process columns - the captured display among them - without store-derived -/// content. -struct ResolvingEmptyStore; - -impl LocateStore for ResolvingEmptyStore { - fn hydrate(self, order: LocateOrder<'_>) -> Result { - Ok(LocateHydration { - nodes: LocateNodeHydration { - resolved: DenseBitSet::new_filled(order.nodes.count()), - type_urls: IdVec::from_elem(Vec::new(), order.nodes.count()), - source_properties: Some(Vec::new()), - source_properties_complete: true, - }, - links: LocateLinkHydration::empty(order.links.len()), - }) - } -} - -/// The locate request naming the arrival by its wire row id. -fn locate_by_row(wire: crate::serve::WireRow) -> LocateRequest { - LocateRequest { - entity_id: None, - row: Some(wire), - colored_type_ids: Vec::new(), - } -} - -/// Inverts the scope's cut rule over the arrival's bucket at `k = 0`. -/// -/// The scope folds its arrivals into its own cascade, so the bucket is the separation law over -/// the visible fitted keys, clamped into the catch-all, the zoom inverts it, and the fly-to -/// cell is the projected coordinate's tile at that zoom. -fn arrival_zoom_and_cell( - atlas: &Atlas, - proof: &VisibilityProof, - wire: Vec2, -) -> (u8, TileCoordinate) { - let deepest = FIXTURE_LOD.max_tile_depth + FIXTURE_LOD.span.get(); - let bucket = expected_bucket(atlas, proof, wire, deepest); - let zoom = bucket.saturating_sub(FIXTURE_LOD.span.get()); - let [x, y] = WIRE_FRAME.quantize(wire); - let cell = coordinate_of( - MortonKey::new(x, y).cell(Depth::new(zoom).expect("a served zoom is a depth")), - ); - (zoom, cell) -} - -/// Both ingress domains resolve a placed arrival to one source point under the cut rule. -/// -/// The entity-keyed and the wire-keyed resolutions must agree on one `SourcePoint`, whose zoom -/// is the scope's own cut rule inverted over the arrival's bucket and whose cell is the -/// projected coordinate's tile there. The saturated shape must agree on the zoom through its -/// overlay, which pins both scoped bucket sources against one law. -#[tokio::test] -async fn arrival_source_point_cut_rule() { - let (_, atlas) = publish("arrival-source-point").await; - let (wire, _) = vacant_cell(&atlas); - let snapshot = arriving(&atlas, wire); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let cohort = PlacementCohort::of(Some(&snapshot)); - let proof = widened(&atlas, &[0], &[slot]); - - let bound = Bound::resolved(&atlas, &proof, cohort, CutOffset::ZERO); - let view = bound.view(&atlas); - let (zoom, cell) = arrival_zoom_and_cell(&atlas, &proof, wire); - - let source = atlas - .resolve_source(&view, &entity_string_of(ARRIVAL_SEED)) - .expect("the cohort resolves the arrival"); - assert_eq!( - source.subject, - SourceSubject::Arrival(ArrivalIndex::from_u32(0)), - "the arrival addresses the view's table" - ); - assert_eq!(source.zoom, zoom, "the zoom inverts the cut rule"); - assert_eq!( - source.cell, cell, - "the fly-to cell is the projected coordinate's tile" - ); - - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()); - assert_eq!( - atlas.resolve_wire_source(&view, slot_wire), - Some(source), - "both ingress domains land on one source point" - ); - - // The saturated shape reads the shared memo with the cohort as its overlay, and must invert - // to the same zoom through the overlay's bucket, because hiding no row leaves the - // separation inputs identical. - let saturated = widened(&atlas, &[], &[slot]); - let saturated_bound = Bound::resolved(&atlas, &saturated, cohort, CutOffset::ZERO); - let through_overlay = atlas - .resolve_source( - &saturated_bound.view(&atlas), - &entity_string_of(ARRIVAL_SEED), - ) - .expect("the saturated scope resolves the arrival"); - assert_eq!( - through_overlay.zoom, zoom, - "the overlay inverts identically" - ); -} - -/// Locate serves a placed arrival from both ingress domains, byte-exact against the directly -/// built wire document. -/// -/// The full envelope delivers the arrival alone and complete, because the generation's -/// adjacency never names a cohort slot and this cohort publishes no link at it. The columns -/// carry its projected coordinate and slot wire id. The trailer carries the captured display -/// once the store resolution answers, and the edge columns stay empty. A second assembly must -/// produce identical bytes. -#[tokio::test] -async fn locate_serves_placed_arrival_from_both_ingress_domains() { - let (generation, atlas) = publish("arrival-locate").await; - let (wire, _) = vacant_cell(&atlas); - let snapshot = arriving(&atlas, wire); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let cohort = PlacementCohort::of(Some(&snapshot)); - let proof = widened(&atlas, &[0], &[slot]); - let (_, cell) = arrival_zoom_and_cell(&atlas, &proof, wire); - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()); - - let assemble = |request: &LocateRequest| { - let bound = Bound::resolved(&atlas, &proof, cohort, CutOffset::ZERO); - atlas - .locate( - request, - ServeLimits::default(), - bound.view(&atlas), - ResolvingEmptyStore, - ) - .expect("the arrival locate serves") - }; - let by_id = assemble(&locate_request(entity_string_of(ARRIVAL_SEED))); - assert_eq!( - by_id, - assemble(&locate_request(entity_string_of(ARRIVAL_SEED))), - "two assemblies serve identical bytes" - ); - assert_eq!( - by_id, - assemble(&locate_by_row(slot_wire)), - "both ingress domains serve identical bytes" - ); - - let arrival_row = ArrivalRow { - identity: archived_id(ARRIVAL_SEED), - position: wire, - wire: slot_wire, - legend: arrival_legend(&atlas), - }; - let empty_map = PropertyMap::new_unchecked(Vec::new()); - let no_edges = EdgeColumns::pinned([]); - let no_flags: Box> = DenseBitSlice::new_empty(0); - let expected = LocateResponse { - generation: generation.id().digest(), - variant: 0, - cell, - complete: true, - entity_id: archived_id(ARRIVAL_SEED), - // The store records no types, and coverage of an empty set attests nothing. - type_ids_complete: false, - properties_complete: true, - delivered: IdSlice::from_raw(&[ViewRow::Arrival(ArrivalIndex::from_u32(0))]), - arrivals: IdSlice::from_raw(core::slice::from_ref(&arrival_row)), - positions: atlas.positions(), - rows: atlas.wire_rows(), - masks: None, - edges: &no_edges, - trailer: LocateTrailer { - type_table: IdSlice::from_raw(&[]), - property_table: IdSlice::from_raw(&[]), - labels: IdSlice::from_raw(&[Label::new("arrival")]), - type_ids: IdSlice::from_raw(&[None]), - properties: Some(&empty_map), - link_labels: IdSlice::from_raw(&[]), - link_type_ids: IdSlice::from_raw(&[]), - link_type_ids_complete: &no_flags, - link_properties: IdSlice::from_raw(&[]), - link_properties_complete: &no_flags, - }, - } - .encode(); - assert_eq!(by_id, expected, "the arrival locate is byte-exact"); -} - -/// A corpus locate clamps the arrival's natural bucket into the catch-all before inverting. -/// -/// The vacant arrival inverts its natural bucket, and a full-key co-location saturates at -/// [`Depth::MAX`], whose clamp into the corpus catch-all inverts to exactly the deepest served -/// zoom. A fitted source's response must not move when a cohort rides the view, which pins the -/// fitted locate path against arrival ingress. -#[tokio::test] -async fn corpus_locate_clamps_arrival_zoom_into_catch_all() { - let (_generation, atlas) = publish("arrival-locate-corpus").await; - let (vacant, _) = vacant_cell(&atlas); - let co_located = atlas.positions()[BasePosition::MIN]; - let snapshot = arriving_all( - &atlas, - &[(ARRIVAL_SEED, vacant), (ARRIVAL_SEED + 1, co_located)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let bound = Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO); - let view = bound.view(&atlas); - - // Grid::deepest() is the corpus catch-all, and first_zoom subtracts the span. - let deepest = FIXTURE_LOD.max_tile_depth + FIXTURE_LOD.span.get(); - let vacant_zoom = - expected_bucket(&atlas, &FULL, vacant, deepest).saturating_sub(FIXTURE_LOD.span.get()); - let source = atlas - .resolve_source(&view, &entity_string_of(ARRIVAL_SEED)) - .expect("the corpus view resolves the arrival"); - assert_eq!(source.zoom, vacant_zoom, "the natural bucket inverts"); - - // The co-location shares every key bit with fitted row zero, so its natural bucket - // saturates past the catch-all and the clamp answers the deepest served zoom. - let saturated = atlas - .resolve_source(&view, &entity_string_of(ARRIVAL_SEED + 1)) - .expect("the corpus view resolves the co-located arrival"); - assert_eq!( - saturated.zoom, FIXTURE_LOD.max_tile_depth, - "the catch-all clamp inverts to the deepest served zoom" - ); - - // A fitted source's bytes are cohort-independent. The same request and the same store - // answer run with and without the retained cohort. - let fitted = |cohort: PlacementCohort<'_>| { - let bound = Bound::resolved(&atlas, &FULL, cohort, CutOffset::ZERO); - atlas - .locate( - &locate_request(entity_string_of(0)), - ServeLimits::default(), - bound.view(&atlas), - ResolvingEmptyStore, - ) - .expect("the fitted locate serves") - }; - assert_eq!( - fitted(cohort), - fitted(PlacementCohort::EMPTY), - "a fitted source's bytes do not move under a cohort" - ); -} - -/// A withdrawn or unadmitted arrival locates nowhere, in both ingress domains. -/// -/// The ingress withdrawal filter hides a retained arrival on this request alone, a scoped mask -/// without the slot never admits it, and the empty cohort refuses its wire id at decode. The -/// control withdraws an identity the request never names and must leave both resolutions -/// standing. -#[tokio::test] -async fn withdrawn_or_unadmitted_arrival_locates_nowhere() { - let (_generation, atlas) = publish("arrival-locate-withdrawn").await; - let (wire, _) = vacant_cell(&atlas); - let snapshot = arriving(&atlas, wire); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let cohort = PlacementCohort::of(Some(&snapshot)); - let proof = widened(&atlas, &[0], &[slot]); - let id = entity_string_of(ARRIVAL_SEED); - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()); - - // The ingress withdrawal covers both domains through the one convergence point. - let hidden = withdrawing(&atlas, &[ARRIVAL_SEED]); - let bound = Bound::resolved(&atlas, &proof, cohort, CutOffset::ZERO).withdrawing(&hidden); - let view = bound.view(&atlas); - assert!( - atlas.resolve_source(&view, &id).is_none(), - "a withdrawn arrival is unknown by entity id" - ); - assert!( - atlas.resolve_wire_source(&view, slot_wire).is_none(), - "the wire-keyed path refuses the same way" - ); - - // Same-path control: an unrelated withdrawal leaves both resolutions standing. - let unrelated = withdrawing(&atlas, &[ARRIVAL_SEED ^ 0x11]); - let bound = Bound::resolved(&atlas, &proof, cohort, CutOffset::ZERO).withdrawing(&unrelated); - let view = bound.view(&atlas); - assert!( - atlas.resolve_source(&view, &id).is_some() - && atlas.resolve_wire_source(&view, slot_wire).is_some(), - "an unrelated withdrawal moves nothing" - ); - - // A mask without the slot never admitted the arrival: the view's table does not hold it, - // and both domains answer the same absence. - let slotless = mask_hiding(&atlas, &[0]); - let bound = Bound::resolved(&atlas, &slotless, cohort, CutOffset::ZERO); - let view = bound.view(&atlas); - assert!( - atlas.resolve_source(&view, &id).is_none(), - "an unadmitted arrival is unknown by entity id" - ); - assert!( - atlas.resolve_wire_source(&view, slot_wire).is_none(), - "an unadmitted slot refuses at resolution" - ); - - // The empty cohort is the resolution that read no publication: the slot wire id lies past - // the generation's universe and refuses at decode. - let bound = Bound::resolved(&atlas, &proof, PlacementCohort::EMPTY, CutOffset::ZERO); - let view = bound.view(&atlas); - assert!( - atlas.resolve_wire_source(&view, slot_wire).is_none(), - "an empty cohort refuses the slot at decode" - ); -} - -/// A cohort publishing no links moves no edge byte: arrivals alone contribute no edge. -/// -/// An arrival contributes no generation edge, and a delta edge exists only where the cohort -/// publishes a link. Corpus and saturated views holding a link-free cohort must answer the -/// exact bytes their arrival-free counterparts answer over the whole served grid. -#[tokio::test] -async fn arrival_bounds_no_edge_in_edges_response() { - let (_generation, atlas) = publish("arrival-edges").await; - let (vacant, _) = vacant_cell(&atlas); - let co_located = atlas.positions()[BasePosition::MIN]; - let snapshot = arriving_all( - &atlas, - &[(ARRIVAL_SEED, vacant), (ARRIVAL_SEED + 1, co_located)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let base = atlas.node_universe().size(); - let slots = [base, base + 1].map(NodeRowId::from_usize); - - let edges = |proof: &VisibilityProof, cohort: PlacementCohort<'_>| { - let bound = Bound::resolved(&atlas, proof, cohort, CutOffset::ZERO); - atlas - .edges( - &edges_request(full_grid()), - EdgesLimits::default(), - bound.view(&atlas), - UntouchedStore, - ) - .expect("the edges request is on the served grid") - }; - - assert_eq!( - edges(&FULL, cohort), - edges(&FULL, PlacementCohort::EMPTY), - "the corpus edge set ignores the cohort" - ); - - let saturated = widened(&atlas, &[], &slots); - let slotless = mask_hiding(&atlas, &[]); - assert_eq!( - edges(&saturated, cohort), - edges(&slotless, PlacementCohort::EMPTY), - "the scoped edge set ignores the cohort" - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/authorization.rs b/libs/@local/graph/atlas/src/serve/tests/authorization.rs deleted file mode 100644 index bf231722902..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/authorization.rs +++ /dev/null @@ -1,394 +0,0 @@ -//! The authority token's contract, driven through [`TokenAuthority`] alone. -#![expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] - -use core::time::Duration; -use std::time::SystemTime; - -use rand::{SeedableRng as _, rngs::ChaCha20Rng}; -use type_system::principal::actor::{ActorId, UserId}; -use uuid::Uuid; - -use crate::{ - integrity::SecretHexBytes, - serve::{ - DeltaEpoch, - authorization::{AuthorityError, Scope, TOKEN_BYTES, TokenAuthority}, - cache::scope::FilterDigest, - density::CutOffset, - }, -}; - -/// The acceptance window of every fixture authority. -const HARD: Duration = Duration::from_mins(10); - -/// The canonical filter document of the filtered fixtures. -const FILTER: &[u8] = b"{\"kind\":\"all\"}"; - -/// A deterministic CSPRNG for the fixtures. -/// -/// Seeded rather than drawn from the operating system: these cases never depend on the nonce's -/// value, and a fixed stream keeps a failure reproducible. -fn rng() -> ChaCha20Rng { - ChaCha20Rng::from_seed([7; 32]) -} - -/// The fixture secret: 32 bytes of key material, value arbitrary. -fn secret() -> SecretHexBytes<32> { - SecretHexBytes::new([0x5A; 32]) -} - -/// The fixture generation. -fn generation() -> crate::file::generation::GenerationId { - "07".repeat(32) - .parse() - .expect("64 hexadecimal digits name a generation") -} - -/// The fixture issue time, a round wall-clock second. -fn issued_at() -> SystemTime { - SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000) -} - -/// The actor identity `actor` names, as a token's presenter. -fn actor(actor: u128) -> ActorId { - ActorId::User(UserId::new(Uuid::from_u128(actor))) -} - -/// The view of one actor, resolved at offset `k`, over the filter digest of `filter` when present. -fn scope(actor_id: u128, k: u8, filter: Option<&[u8]>) -> Scope { - Scope::new( - actor(actor_id), - filter.map(FilterDigest::of), - CutOffset::new(k), - ) -} - -/// An authority over the fixture generation and secret, holding `epoch`. -fn authority(epoch: Option) -> TokenAuthority { - TokenAuthority::new(generation(), &secret(), HARD, epoch, rng()) -} - -/// An issued token opens to the scope it sealed, filtered or not, under either epoch form. -#[test] -fn open_roundtrip() { - let epoch = DeltaEpoch::fresh(&mut rng()).expect("the seeded generator is infallible"); - - for authority in [authority(None), authority(Some(epoch))] { - for (k, filter) in [(0, None), (5, Some(FILTER))] { - let minted = authority - .issue(scope(11, k, filter), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority - .open(&minted, actor(11), issued_at()) - .expect("an issued token opens"), - scope(11, k, filter), - "the opened scope differs from the sealed one" - ); - } - } -} - -/// Every byte of the envelope is under the tag, and the version byte leads it. -/// -/// The first byte fails the cast ahead of the tag. Every other byte, the issue time included, fails -/// the tag ahead of the window. -#[test] -fn open_tampered_byte() { - let authority = authority(None); - let minted = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - for index in 0..TOKEN_BYTES { - let mut tampered = minted; - tampered[index] ^= 1; - - let expected = if index == 0 { - AuthorityError::Envelope - } else { - AuthorityError::Authentication - }; - assert_eq!( - authority.open(&tampered, actor(11), issued_at()), - Err(expected), - "a token with a flipped bit at byte {index} opened, or refused for another cause" - ); - } -} - -/// A token at or past the hard window refuses, and so does a future-dated one. -#[test] -fn open_outside_window() { - let authority = authority(None); - let minted = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority.open(&minted, actor(11), issued_at() + HARD), - Err(AuthorityError::Stale), - "a token at the hard window opened" - ); - assert_eq!( - authority.open(&minted, actor(11), issued_at() - Duration::from_secs(1)), - Err(AuthorityError::Stale), - "a future-dated token opened" - ); - authority - .open( - &minted, - actor(11), - issued_at() + HARD - Duration::from_secs(1), - ) - .expect("a token one second inside the window opens"); -} - -/// A token issued under another generation refuses at the tag, because the generation salts the -/// key. -#[test] -fn open_foreign_generation() { - let minted = authority(None) - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - let foreign = TokenAuthority::new( - "1a".repeat(32) - .parse() - .expect("64 hexadecimal digits name a generation"), - &secret(), - HARD, - None, - rng(), - ); - - assert_eq!( - foreign.open(&minted, actor(11), issued_at()), - Err(AuthorityError::Authentication), - "a token from another generation opened" - ); -} - -/// A token opened under another secret refuses at the tag. -#[test] -fn open_foreign_secret() { - let minted = authority(None) - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - let foreign = TokenAuthority::new( - generation(), - &SecretHexBytes::new([0xA5; 32]), - HARD, - None, - rng(), - ); - - assert_eq!( - foreign.open(&minted, actor(11), issued_at()), - Err(AuthorityError::Authentication), - "a token opened under another secret" - ); -} - -/// A valid token presented by an actor it does not name refuses. -/// -/// The tag proves the server issued the token, not that the presenter is its subject: without this -/// refusal a leaked token would grant any authenticated actor the subject's scope. -#[test] -fn open_foreign_actor() { - let authority = authority(None); - let minted = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority.open(&minted, actor(12), issued_at()), - Err(AuthorityError::Actor), - "another actor's presentation opened" - ); -} - -/// A token sealed under a dead delta epoch refuses to open. -/// -/// Issuer and successor share one generation and secret, a serving process before and after a -/// restart. -#[test] -fn open_dead_epoch() { - let mut draws = rng(); - let first = DeltaEpoch::fresh(&mut draws).expect("the seeded generator is infallible"); - let second = DeltaEpoch::fresh(&mut draws).expect("the seeded generator is infallible"); - - let minted = authority(Some(first)) - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority(Some(second)).open(&minted, actor(11), issued_at()), - Err(AuthorityError::Epoch), - "a dead epoch's token opened" - ); -} - -/// The absent and present epoch forms refuse each other in both directions. -/// -/// A delta-serving process must not accept the tokens of a no-delta predecessor, whose sessions -/// never learned any slot. A no-delta process must not accept tokens whose sessions may hold slot -/// rows from a delta predecessor, and exact equality of the sealed form is the one rule covering -/// both. -#[test] -fn open_epoch_presence_mismatch() { - let epoch = DeltaEpoch::fresh(&mut rng()).expect("the seeded generator is infallible"); - let without = authority(None); - let with = authority(Some(epoch)); - - let absent_sealed = without - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - let present_sealed = with - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - with.open(&absent_sealed, actor(11), issued_at()), - Err(AuthorityError::Epoch), - "a no-delta token opened under a live epoch" - ); - assert_eq!( - without.open(&present_sealed, actor(11), issued_at()), - Err(AuthorityError::Epoch), - "a delta token opened under a no-delta authority" - ); -} - -/// Two no-delta authorities accept each other's tokens. -#[test] -fn open_absent_epoch_after_restart() { - let minted = authority(None) - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority(None) - .open(&minted, actor(11), issued_at()) - .expect("a no-delta token opens after a no-delta restart"), - scope(11, 5, None), - "the opened scope differs from the sealed one" - ); -} - -/// An expired token no longer opens, yet still yields its view state for a renewal. -/// -/// Hard invalidation forces a fresh token without perturbing the view. -#[test] -fn continuity_expired() { - let authority = authority(None); - let minted = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - let later = issued_at() + HARD + Duration::from_mins(1); - - assert_eq!( - authority.open(&minted, actor(11), later), - Err(AuthorityError::Stale), - "an expired token opened" - ); - - let carried = authority - .continuity(&minted, actor(11)) - .expect("an expired token still yields its view state"); - assert_eq!( - carried, - scope(11, 5, None), - "the continuity read differs from the sealed scope" - ); - - let renewed = authority - .issue(carried, later) - .expect("the seeded generator is infallible"); - assert_eq!( - authority - .open(&renewed, actor(11), later) - .expect("the renewed token opens"), - scope(11, 5, None), - "the renewal perturbed the view" - ); -} - -/// The continuity read skips the window and keeps the tag: a flipped tag byte refuses. -#[test] -fn continuity_tampered() { - let authority = authority(None); - let mut tampered = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - tampered[TOKEN_BYTES - 1] ^= 1; - - assert_eq!( - authority.continuity(&tampered, actor(11)), - Err(AuthorityError::Authentication), - "a tampered token carried" - ); -} - -/// The continuity read skips the window and keeps the actor comparison: a presenter the token does -/// not name refuses. -#[test] -fn continuity_foreign_actor() { - let authority = authority(None); - let minted = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority.continuity(&minted, actor(12)), - Err(AuthorityError::Actor), - "another actor's presentation carried" - ); -} - -/// A token sealed under a dead delta epoch refuses to carry into a renewal. -/// -/// The renewal read is the path the session-replacement contract names: view state accumulated -/// beside a dead register must not carry into a token issued under the live one. -#[test] -fn continuity_dead_epoch() { - let mut draws = rng(); - let first = DeltaEpoch::fresh(&mut draws).expect("the seeded generator is infallible"); - let second = DeltaEpoch::fresh(&mut draws).expect("the seeded generator is infallible"); - - let minted = authority(Some(first)) - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_eq!( - authority(Some(second)).continuity(&minted, actor(11)), - Err(AuthorityError::Epoch), - "a dead epoch's token carried into a renewal" - ); -} - -/// A second issuance draws a different nonce. -/// -/// Tokens of one scope at one instant differ through the nonce alone. -#[test] -fn issue_distinct_nonces() { - let authority = authority(None); - let first = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - let second = authority - .issue(scope(11, 5, None), issued_at()) - .expect("the seeded generator is infallible"); - - assert_ne!(first, second, "two issuances shared a nonce"); - for minted in [&first, &second] { - authority - .open(minted, actor(11), issued_at()) - .expect("an equal-scope token opens"); - } -} diff --git a/libs/@local/graph/atlas/src/serve/tests/auxiliary.rs b/libs/@local/graph/atlas/src/serve/tests/auxiliary.rs deleted file mode 100644 index f77527fd863..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/auxiliary.rs +++ /dev/null @@ -1,395 +0,0 @@ -//! The detail trailer's payload hydration over a displaying fixture. -//! -//! The all-empty instance in `tile.rs` proves the trailer's encode path with details injected at -//! the transport splice. These cases instead feed real payloads through the production pipeline, -//! dataset labels and icons entering the published identity artifacts and coming out through the -//! tile arm's own hydration. They can therefore fail on what an all-empty instance cannot: -//! per-position alignment to the delivered order, icon-precedence selection across a point's -//! direct types, and non-empty auxiliary column encoding. -//! -//! Expectations derive from the fixture's own assignment tables and independently opened -//! artifacts, never from the assembly under test. The grid sweep spells its trailer expectation -//! byte-by-byte from the wire contract (`docs/wire.md`), independently of the producer's -//! encoder. - -use std::collections::{HashMap, HashSet}; - -use hashql_core::id::{Id as _, IdSlice}; -use smallvec::smallvec; -use zerocopy::{LE, U64}; - -use super::{ - Artifacts, Bound, FIXTURE_EDGES, FIXTURE_LOD, FULL, NODES, ROW_IDS, decode_rows, fixture_nodes, - fixture_row_ids, full_grid, open_artifacts, publish_dataset, test_codec, -}; -use crate::{ - dataset::{ - Edge, Ontology, - auxiliary::{Icon, Label, OwnedIcon, OwnedLegend}, - card::Card, - memory::MemoryDataset, - }, - identity::{BasePosition, NodeRowId, OntologyRowId}, - math::{Bounds2, Vec2}, - salt::wire::{ - Mode, - cbor::CborWriter, - tests::section, - tile::{DeliveredSet, GlobalHead, TileCoordinate, TileHead, TileResponse, TileTrailer}, - }, - serve::{CutOffset, TileLimits, TileQuery, TileRequest, tile::TileDetail}, -}; - -/// Each ontology row's own icon, by row, empty for a row that displays none. -/// -/// The rows build every icon-resolution shape the trailer serves: -/// -/// | row | parents | icon | memo (source, depth) | -/// |-----|---------|----------------|----------------------| -/// | 0 | - | `icon-alpha` | (0, 0) | -/// | 1 | - | `icon-beta` | (1, 0) | -/// | 2 | [1] | - | (1, 1) | -/// | 3 | [2] | - | (1, 2) | -/// | 4 | - | `icon-gamma` | (4, 0) | -/// | 5 | - | `icon-delta` | (5, 0) | -/// | 6 | - | - | none | -/// | 7 | - | `icon-epsilon` | (7, 0) | -/// | 8 | - | - | none, the link type | -/// -/// Rows 1 through 3 are one inheritance chain under row 1's icon. Rows 1 and 2 carry no -/// instances and exist as its interior. -const ICONS: [&str; 9] = [ - "icon-alpha", - "icon-beta", - "", - "", - "icon-gamma", - "icon-delta", - "", - "icon-epsilon", - "", -]; - -/// Returns the direct types node row `row` carries: the type list of case `row % 6`. -/// -/// Each list is strictly ascending, as the dataset contract requires, and each case resolves a -/// different icon, so a resolution mixing up two cases changes bytes. Candidates read -/// `(index, icon, depth)` per direct type; the resolution law is the minimum over -/// `(depth, index)`. -/// -/// | case | types | candidates | resolved | falsifies | -/// |------|--------|---------------------------------|----------------|--------------------------| -/// | 0 | [0] | (0, alpha, 0) | `icon-alpha` | the depth-zero base case | -/// | 1 | [3] | (0, beta, 2) | `icon-beta` | inheritance at depth two | -/// | 2 | [3, 4] | (0, beta, 2), (1, gamma, 0) | `icon-gamma` | first-listed-type-wins | -/// | 3 | [5, 7] | (0, delta, 0), (1, epsilon, 0) | `icon-delta` | last-wins on a depth tie | -/// | 4 | [6] | - | empty | an icon-free cone serves | -/// | 5 | [6, 7] | (1, epsilon, 0) | `icon-epsilon` | a skipped icon-free type | -#[expect( - clippy::integer_division_remainder_used, - reason = "the modulus folds node rows onto the six-entry case table" -)] -fn case_types(row: usize) -> smallvec::SmallVec { - match row % 6 { - 0 => smallvec![OntologyRowId::new(0)], - 1 => smallvec![OntologyRowId::new(3)], - 2 => smallvec![OntologyRowId::new(3), OntologyRowId::new(4)], - 3 => smallvec![OntologyRowId::new(5), OntologyRowId::new(7)], - 4 => smallvec![OntologyRowId::new(6)], - _ => smallvec![OntologyRowId::new(6), OntologyRowId::new(7)], - } -} - -/// Returns the icon the trailer must deliver for node row `row`, hand-resolved from -/// [`case_types`]'s table, empty for the icon-free case. -#[expect( - clippy::integer_division_remainder_used, - reason = "the modulus folds node rows onto the six-entry case table" -)] -fn expected_icon(row: u32) -> &'static str { - match row % 6 { - 0 => "icon-alpha", - 1 => "icon-beta", - 2 => "icon-gamma", - 3 => "icon-delta", - 4 => "", - _ => "icon-epsilon", - } -} - -/// Returns the label node row `row` displays: distinct per row, empty at row 13. -/// -/// Distinctness is what makes the alignment claim falsifiable: permuting any two delivered -/// labels changes bytes. Row 13 displays nothing, so exactly one label crosses the wire as -/// `null` at a position whose icon stays non-empty - a response swapping the two columns -/// cannot encode it. -fn expected_label(row: u32) -> String { - if row == 13 { - return String::new(); - } - - format!("entity {row:02}") -} - -/// The displaying fixture: [`fixture_nodes`]'s geometry with display payloads assigned. -/// -/// [`case_types`] types each node row and [`expected_label`] labels it. [`ICONS`] gives each -/// ontology row its icon, and the fixture edges take the instance-free link type row 8. -fn displaying_dataset() -> MemoryDataset { - let (nodes, canonical) = fixture_nodes(case_types); - - let edges = FIXTURE_EDGES - .into_iter() - .map(|(id, source, target)| Edge { - id: U64::::new(id), - source: NodeRowId::new(source), - target: NodeRowId::new(target), - ontology: smallvec![OntologyRowId::new(8)], - embedding: None, - confidence: None, - source_confidence: None, - target_confidence: None, - }) - .collect(); - - let parents_of: [&[u64]; 9] = [&[], &[], &[1], &[2], &[], &[], &[], &[], &[]]; - let ontology = parents_of - .into_iter() - .enumerate() - .map(|(row, parents)| Ontology { - id: U64::::new(row as u64), - parents: parents - .iter() - .map(|&parent| OntologyRowId::new(parent)) - .collect(), - }) - .collect(); - - let cards = (0..parents_of.len() as u64) - .map(|row| (row, Card::verbatim(format!("Type {row} card")))) - .collect(); - - let mut dataset = MemoryDataset::new(nodes, edges, ontology, canonical, cards); - for (row, legend) in dataset.node_legends.iter_mut().enumerate() { - let row = u32::try_from(row).expect("fixture counts fit u32"); - *legend = OwnedLegend::new( - legend.representative_ontology(), - Label::new(&expected_label(row)), - ); - } - dataset.ontology_icons = ICONS.iter().map(|&icon| OwnedIcon::from(icon)).collect(); - - dataset -} - -/// Encodes the detail trailer the wire contract pins: one CBOR map of two arrays, text entries -/// with `null` for a row that displays nothing. -/// -/// Spelled from `docs/wire.md`, so the expectation stays independent of the assembly under -/// test. -fn expected_trailer(labels: &[String], icons: &[&str]) -> Vec { - fn details<'entry>( - cbor: &mut CborWriter<'_>, - entries: impl ExactSizeIterator, - ) { - cbor.array(entries.len() as u64); - for entry in entries { - if entry.is_empty() { - cbor.null(); - } else { - cbor.text(entry); - } - } - } - - let mut bytes = Vec::new(); - let mut cbor = CborWriter::over(&mut bytes); - cbor.map(2); - cbor.uint(0); - details(&mut cbor, labels.iter().map(String::as_str)); - cbor.uint(1); - details(&mut cbor, icons.iter().copied()); - - bytes -} - -/// The detailed-tile production path, hydrating real generation payloads. -/// -/// One `Atlas::tile` call with `detail: "auxiliary"` over the displaying fixture, byte-compared -/// against the wire document built directly over the fixture's own assignment tables: each -/// delivered position's label is its row's, each icon its row's case resolution. -#[tokio::test] -#[expect( - clippy::single_range_in_vec_init, - reason = "an array of one range is what a root delta delivery IS" -)] -async fn detailed_tiles_hydrate_labels_and_icons_from_the_generation() { - let (generation, atlas) = publish_dataset("auxiliary-trailer", &displaying_dataset()).await; - let Artifacts { - quad, - morton, - coordinates, - rows, - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - // The arm under test: assembly, in-process payload resolution, and encoding through - // `Atlas::tile` itself in one call. - let detailed = TileRequest { - coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, - query: TileQuery { - mode: Mode::Delta, - detail: TileDetail::Auxiliary, - ..TileQuery::default() - }, - }; - let bytes = atlas - .tile( - &detailed, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the detailed root tile serves"); - - let delivered: u64 = morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let delivered = usize::try_from(delivered).expect("fixture counts fit usize"); - - // The trailer columns the response must carry: the delivered prefix's rows, each mapped - // through the fixture's own assignment. - let labels_text: Vec = row_ids[..delivered] - .iter() - .map(|&row| expected_label(row)) - .collect(); - let labels: Vec<&Label> = labels_text.iter().map(|text| Label::new(text)).collect(); - let icons: Vec<&Icon> = row_ids[..delivered] - .iter() - .map(|&row| Icon::new(expected_icon(row))) - .collect(); - - let end = u32::try_from(delivered).expect("fixture counts fit u32"); - let expected = TileResponse { - head: TileHead { - generation: atlas.generation().digest(), - variant: 0, - coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, - mode: Mode::Delta, - first_bucket: 0, - runs: &morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .map(|&length| u32::try_from(length).expect("fixture counts fit u32")) - .collect::>(), - global: Some(GlobalHead { - visible: delivered as u64, - // Normalization maps each attained world axis onto the frame edges, so the - // extent anchors at the full wire square. - bounds: Some( - Bounds2::new(Vec2::new(-1.0, -1.0), Vec2::new(1.0, 1.0)) - .expect("the wire square is a valid extent"), - ), - min_resolution: morton - .fenceposts() - .lengths() - .iter() - .rposition(|&length| length > 0) - .map_or(0, |bucket| bucket as u64), - }), - children: (0..4).fold(0_u8, |bits, quadrant| { - bits | (u8::from(quad.nodes()[0].child(quadrant).is_some()) << quadrant) - }), - }, - delivered: DeliveredSet::Ranges(&[BasePosition::from_u32(0)..BasePosition::from_u32(end)]), - positions: IdSlice::from_raw(points), - rows: IdSlice::from_raw(&{ - let node_codec = test_codec(&atlas); - row_ids - .iter() - .map(|&row| node_codec.encode(NodeRowId::from_u32(row), atlas.node_universe())) - .collect::>() - }), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: Some(TileTrailer { - labels: &labels, - icons: &icons, - }), - } - .encode(); - assert_eq!(bytes, expected, "the hydrated trailer path is byte-exact"); -} - -/// Every icon-precedence case, witnessed over the whole corpus. -/// -/// The deepest grid's cut reaches the catch-all bucket, so total-mode tiles over it deliver -/// every node exactly once - all 48 rows, eight instances of each of [`case_types`]'s six -/// cases, whatever the fit placed where. Per tile, the envelope's tail must be exactly the -/// trailer the delivered rows' assignments encode. The envelope writes the trailer last, so the -/// tail comparison addresses it. -#[tokio::test] -async fn deepest_grid_tiles_align_every_icon_precedence_case() { - let (_generation, atlas) = publish_dataset("auxiliary-grid", &displaying_dataset()).await; - - // Wire row id to fixture row, through the independently derived codec. - let node_codec = test_codec(&atlas); - let rows = u32::try_from(NODES).expect("fixture counts fit u32"); - let decode: HashMap = (0..rows) - .map(|row| { - ( - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get(), - row, - ) - }) - .collect(); - - let mut seen: HashSet = HashSet::new(); - for coordinate in full_grid() { - let detailed = TileRequest { - coordinate, - query: TileQuery { - mode: Mode::Total, - detail: TileDetail::Auxiliary, - ..TileQuery::default() - }, - }; - let bytes = atlas - .tile( - &detailed, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("every deepest tile serves"); - - let delivered: Vec = decode_rows(section(&bytes, ROW_IDS).unwrap_or(&[])) - .into_iter() - .map(|wire| { - *decode - .get(&wire) - .expect("every delivered id decodes to a fixture row") - }) - .collect(); - for &row in &delivered { - assert!( - seen.insert(row), - "row {row} is delivered by exactly one deepest tile", - ); - } - - let labels: Vec = delivered.iter().map(|&row| expected_label(row)).collect(); - let icons: Vec<&str> = delivered.iter().map(|&row| expected_icon(row)).collect(); - let tail = expected_trailer(&labels, &icons); - assert!( - bytes.len() >= tail.len() && bytes[bytes.len() - tail.len()..] == tail[..], - "tile {coordinate:?} ends with the trailer its delivered rows encode", - ); - } - - assert_eq!( - seen.len(), - NODES, - "the deepest grid delivers the whole corpus", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/delta_edges.rs b/libs/@local/graph/atlas/src/serve/tests/delta_edges.rs deleted file mode 100644 index ba90793bac7..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/delta_edges.rs +++ /dev/null @@ -1,2475 +0,0 @@ -//! Delta-edge witnesses: the entry cohort's published links serve through the edges, -//! translate, and locate routes. -//! -//! Every case runs the route's own assembly with a real published snapshot, folded, classified, -//! and placed exactly as the consumer records them, so the witnesses cover the served path rather -//! than the map lookups alone. A delta link serves when the proof's identity set admits it, the -//! ingress capture does not withdraw it, and the response's delivered sets hold both of its -//! endpoints, and it merges into the same ascending identity order the fitted edges answer in. -//! Translate reads the same publication keyed by identity, its endpoints qualified through the -//! proof and the cohort's retention rather than a delivered bound, and the locate fold takes -//! the same endpoint rule around one source. Each refusal case runs beside a same-path control -//! whose delta touches nothing the request names. - -use core::num::NonZero; - -use hash_graph_postgres_store::store::{EntityEnd, EntityEvent, EntityUpdate}; -use hash_graph_temporal_versioning::Timestamp; -use hashql_core::{ - collections::FastHashMap, - id::{Id as _, IdSlice, IdVec, bit_vec::DenseBitSet}, -}; -use type_system::{ - knowledge::entity::{ - EntityId, - id::{EntityEditionId, EntityUuid}, - }, - ontology::id::VersionedUrl, -}; -use uuid::Uuid; - -use super::{ - Atlas, Bound, CutOffset, EDGE_IDS, FIXTURE_LOD, FULL, Generation, UntouchedStore, - arrival::vacant_cell, coordinate_of, edge_identity_of, edges_request, entity_string_of, - expected_edges_bytes, fixture_type_url, full_grid, locate_request, open_edge_artifacts, - publish, section, test_codec, type_expectations, -}; -use crate::{ - bitset::{CompressedBitSet, DenseBitSlice}, - dataset::auxiliary::{Icon, Label, OwnedIcon, OwnedLabel}, - identity::{BasePosition, EdgeRowId, ImportanceRank, NodeRowId}, - math::Vec2, - morton::Depth, - postgres::{ - Classification, - edition_display::DisplayParts, - id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedOntologyTypeUuid}, - }, - random::{keyed_rng, uniform_below}, - salt::wire::{ - edges::{EdgesResponse, EdgesTrailer}, - locate::{LocateResponse, LocateTrailer, PropertyMap}, - }, - serve::{ - EdgesLimits, ServeLimits, VisibilityProof, - delta::{ - DeltaEvent, DeltaRegister, DeltaRevision, DeltaSnapshot, PlacementCohort, - ProjectedArrival, - }, - edges::EdgesDetail, - hydrate::{ - DetailError, EdgesStore, LocateHydration, LocateLinkHydration, LocateNodeHydration, - LocateOrder, LocateStore, TypeSlot, - }, - locate::LocateLimits, - neighbourhood::{DeltaEdge, DeltaEndpoint, EdgeColumns, ServedEdge}, - schedule::{ArrivalIndex, ViewRow}, - translate::{TranslateLimits, TranslateRequest, TranslatedEdge}, - }, -}; - -/// A link identity sorting before every fitted edge identity, whose seeds start at 64. -const LOW_LINK: u8 = 50; - -/// A link identity sorting after the arrival band. -const HIGH_LINK: u8 = 0xB0; - -/// The arrival's seed, past every node and edge seed the fixture generation fits. -const ARRIVAL: u8 = 0xA0; - -/// Link seeds for the differential's random cohorts, disjoint from node seeds `0..48`, edge -/// seeds `64..70`, and `ARRIVAL`. -const LINK_SEEDS: [u8; 8] = [48, 52, 58, 63, 72, 90, 150, 200]; - -/// An endpoint identity the view cannot deliver. -const REFUSED: u8 = 0xD0; - -/// An oracle candidate pairs the full-sort selection key with the delivered wire triple. -type OracleCandidate = ((Rank, ArchivedEntityId), (u32, u32, ArchivedEntityId)); - -/// The store-form entity id the seeding rule gives `seed`. -fn store_id(seed: u8) -> EntityId { - EntityId { - web_id: type_system::principal::actor_group::WebId::new(Uuid::from_bytes([seed; 16])), - entity_uuid: EntityUuid::new(Uuid::from_bytes([seed ^ 0xFF; 16])), - draft_id: None, - } -} - -/// The identity-table key the seeding rule gives `seed`. -fn archived_id(seed: u8) -> ArchivedEntityId { - ArchivedEntityId { - web_id: Uuid::from_bytes([seed; 16]).into(), - entity_uuid: ArchivedEntityUuid::from_bytes( - Uuid::from_bytes([seed ^ 0xFF; 16]).into_bytes(), - ), - } -} - -/// Folds, classifies, and places live arrivals and links, publishing one snapshot. -/// -/// The events travel the consumer's own conversion. Each arrival classifies as a node and takes -/// the register's own slot allocation, and each link classifies with its endpoint pair, so the -/// snapshot is the publication a scope resolution would bind rather than a hand-assembled -/// equivalent. A link's endpoints are seeds under the same rule, so a link can attach fitted -/// rows, arrivals, and itself. -fn publishing(atlas: &Atlas, arrivals: &[(u8, Vec2)], links: &[(u8, u8, u8)]) -> DeltaSnapshot { - publishing_displayed( - atlas, - arrivals, - links, - &DisplayParts { - label: OwnedLabel::from("link"), - icon: OwnedIcon::from("link-icon"), - representative: fixture_type(), - }, - ) -} - -/// The fixture's shared link representative type, unknown to the generation, so the register's -/// own extension allocates its ontology row. -fn fixture_type() -> ArchivedOntologyTypeUuid { - ArchivedOntologyTypeUuid::from(Uuid::from_u128(0x117C)) -} - -/// [`publishing`], with every link capturing `link_display` instead of the default. -fn publishing_displayed( - atlas: &Atlas, - arrivals: &[(u8, Vec2)], - links: &[(u8, u8, u8)], - link_display: &DisplayParts, -) -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - for &(seed, wire) in arrivals { - let event = EntityEvent::Updated(EntityUpdate { - entity: store_id(seed), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - register - .classify(archived_id(seed), Classification::Node) - .expect("the fixture stays inside the edge universe"); - register - .place( - archived_id(seed), - &ProjectedArrival { - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - position: wire, - label: OwnedLabel::from("arrival"), - icon: OwnedIcon::from("arrival-icon"), - representative: fixture_type(), - }, - atlas, - ) - .expect("the fixture universe is far from the wire's row domain"); - } - for &(seed, source, target) in links { - let event = EntityEvent::Updated(EntityUpdate { - entity: store_id(seed), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - register - .classify( - archived_id(seed), - Classification::Edge { - source: Some(archived_id(source)), - target: Some(archived_id(target)), - }, - ) - .expect("the fixture stays inside the edge universe"); - // Publication withholds an uncaptured link, so the fixture captures exactly as the - // poll's own display read does. - let DisplayParts { - label, - icon, - representative, - } = link_display.clone(); - register - .capture_display( - archived_id(seed), - EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - &label, - &icon, - representative, - atlas, - ) - .expect("the fixture ontology domain has room"); - } - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(1), - ) -} - -/// Folds one `Ended` event per seed into an ingress snapshot withdrawing those identities. -fn withdrawing(atlas: &Atlas, seeds: &[u8]) -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - for &seed in seeds { - let event = EntityEvent::Ended(EntityEnd { - entity: store_id(seed), - ended_at: Timestamp::from_unix_timestamp(2), - }); - register.apply(DeltaEvent::from(&event)); - } - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(2), - ) -} - -/// The full-coverage fitted wire triples, ascending identity bytes. -/// -/// Every fixture edge qualifies under a full grid with everything visible, and edge identities -/// ascend with the edge row by the seeding rule, so the artifact order is the delivery order. -fn fitted_triples(atlas: &Atlas, generation: &Generation) -> Vec<(u32, u32, ArchivedEntityId)> { - let artifacts = open_edge_artifacts(generation); - let endpoints = artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let codec = test_codec(atlas); - let wire = |row: u64| { - codec - .encode( - NodeRowId::from_u32(u32::try_from(row).expect("fixture rows fit u32")), - atlas.node_universe(), - ) - .get() - }; - - endpoints - .iter() - .enumerate() - .map(|(row, pair)| { - let [source, target] = pair.map(zerocopy::U64::get); - ( - wire(source), - wire(target), - edge_identity_of(u32::try_from(row).expect("fixture edge rows fit u32")), - ) - }) - .collect() -} - -/// The wire id of the fitted node owning `seed`, whose row is the seed by the seeding rule. -fn node_wire(atlas: &Atlas, seed: u8) -> u32 { - test_codec(atlas) - .encode(NodeRowId::from_u32(u32::from(seed)), atlas.node_universe()) - .get() -} - -/// The delivered `EDGE_IDS` records of one response. -fn edge_ids_of(bytes: &[u8]) -> Vec { - section(bytes, EDGE_IDS) - .expect("EDGE_IDS is present") - .as_chunks::<32>() - .0 - .iter() - .map(|record| { - *zerocopy::FromBytes::ref_from_bytes(record.as_slice()) - .expect("EDGE_IDS records are identity-sized") - }) - .collect() -} - -/// A scoped proof admitting every fitted row, `slots`, and exactly `links`. -fn admitting(atlas: &Atlas, slots: &[NodeRowId], links: &[u8]) -> VisibilityProof { - let rows = u32::try_from(atlas.row_ids().len()).expect("fixture domains fit u32"); - let edges = u32::try_from(atlas.endpoints.view().len()).expect("fixture domains fit u32"); - - VisibilityProof::from_masks( - CompressedBitSet::from_rows( - (0..rows) - .map(NodeRowId::from_u32) - .chain(slots.iter().copied()), - ), - CompressedBitSet::from_rows((0..edges).map(EdgeRowId::from_u32)), - links.iter().map(|&seed| archived_id(seed)).collect(), - ) -} - -/// Serves one edges request over `proof` and `cohort`, with `ingress` as the request's capture. -fn edges_with( - atlas: &Atlas, - proof: &VisibilityProof, - cohort: PlacementCohort<'_>, - ingress: Option<&DeltaSnapshot>, - tiles: Vec, - limits: EdgesLimits, -) -> Vec { - let mut bound = Bound::resolved(atlas, proof, cohort, CutOffset::ZERO); - if let Some(ingress) = ingress { - bound = bound.withdrawing(ingress); - } - - atlas - .edges( - &edges_request(tiles), - limits, - bound.view(atlas), - UntouchedStore, - ) - .expect("the edges request is on the served grid") -} - -/// A corpus view serves every published link, merged into the identity order, byte-exactly. -/// -/// The link identities straddle the fitted edge identities, so the expectation pins the merge -/// rather than an appended tail: one delta edge leads the column and one closes it. The repeat -/// call pins determinism across the cohort map's iteration order, and the empty-cohort control -/// answers the fitted set alone on the same path. -#[tokio::test] -async fn corpus_link_identity_order() { - let (generation, atlas) = publish("delta-edges-corpus").await; - let snapshot = publishing(&atlas, &[], &[(LOW_LINK, 0, 1), (HIGH_LINK, 2, 3)]); - let cohort = PlacementCohort::of(Some(&snapshot)); - let fitted = fitted_triples(&atlas, &generation); - - let mut merged = vec![( - node_wire(&atlas, 0), - node_wire(&atlas, 1), - archived_id(LOW_LINK), - )]; - merged.extend(fitted.iter().copied()); - merged.push(( - node_wire(&atlas, 2), - node_wire(&atlas, 3), - archived_id(HIGH_LINK), - )); - let expected = expected_edges_bytes(&generation, true, &EdgeColumns::pinned(merged)); - - let serve = |cohort| { - edges_with( - &atlas, - &FULL, - cohort, - None, - full_grid(), - EdgesLimits::default(), - ) - }; - - let first = serve(cohort); - assert_eq!(first, expected, "delta links merge into the identity order"); - assert_eq!( - serve(cohort), - first, - "the delta-bearing response is deterministic" - ); - - let control = expected_edges_bytes(&generation, true, &EdgeColumns::pinned(fitted)); - assert_eq!( - serve(PlacementCohort::EMPTY), - control, - "an empty cohort serves the fitted set alone" - ); -} - -/// A scoped proof serves exactly the links its own resolution admitted. -/// -/// Both directions on one path: widening the admitted set from one link to both adds exactly -/// the second link's row, and the empty set serves the fitted edges alone even while the cohort -/// publishes both links. -#[tokio::test] -async fn scoped_admitted_links() { - let (generation, atlas) = publish("delta-edges-admission").await; - let snapshot = publishing(&atlas, &[], &[(LOW_LINK, 0, 1), (HIGH_LINK, 2, 3)]); - let cohort = PlacementCohort::of(Some(&snapshot)); - let fitted = fitted_triples(&atlas, &generation); - - let serve = |links: &[u8]| { - edges_with( - &atlas, - &admitting(&atlas, &[], links), - cohort, - None, - full_grid(), - EdgesLimits::default(), - ) - }; - - let low = ( - node_wire(&atlas, 0), - node_wire(&atlas, 1), - archived_id(LOW_LINK), - ); - let high = ( - node_wire(&atlas, 2), - node_wire(&atlas, 3), - archived_id(HIGH_LINK), - ); - - let mut one = vec![low]; - one.extend(fitted.iter().copied()); - assert_eq!( - serve(&[LOW_LINK]), - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(one.clone())), - "the admitted link serves" - ); - - let mut both = one; - both.push(high); - assert_eq!( - serve(&[LOW_LINK, HIGH_LINK]), - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(both)), - "widening the admitted set adds exactly the second link" - ); - - assert_eq!( - serve(&[]), - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(fitted)), - "an unadmitted link never serves, whatever the cohort publishes" - ); -} - -/// An arrival endpoint qualifies exactly when the arrival serves. -/// -/// One link attaches the anchor node to a placed arrival and a control link attaches the anchor -/// to itself. Listing both tiles serves both links. Dropping the arrival's tile from the list -/// drops the arrival-endpoint link alone, and hiding the slot from the proof drops it again -/// with both tiles listed, while the self-loop control survives every case. -#[tokio::test] -async fn arrival_endpoint_qualification() { - let (_generation, atlas) = publish("delta-edges-arrival").await; - let (vacant, arrival_tile) = vacant_cell(&atlas); - - let anchor_position = BasePosition::MIN; - let anchor_row = atlas.row_ids()[anchor_position]; - let anchor_seed = u8::try_from(anchor_row.as_u32()).expect("fixture rows fit u8"); - let depth = Depth::new(FIXTURE_LOD.max_tile_depth).expect("the fixture depth is a depth"); - let anchor_tile = coordinate_of(atlas.morton.code(anchor_position).cell(depth)); - - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[ - (HIGH_LINK, anchor_seed, ARRIVAL), - (LOW_LINK, anchor_seed, anchor_seed), - ], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - let ids = |proof: &VisibilityProof, tiles: Vec<_>| { - edge_ids_of(&edges_with( - &atlas, - proof, - cohort, - None, - tiles, - EdgesLimits::default(), - )) - }; - - let widened = admitting(&atlas, &[slot], &[LOW_LINK, HIGH_LINK]); - let served = ids(&widened, vec![anchor_tile, arrival_tile]); - assert!( - served.contains(&archived_id(HIGH_LINK)) && served.contains(&archived_id(LOW_LINK)), - "both links serve while the arrival delivers: {served:?}" - ); - - let undelivered = ids(&widened, vec![anchor_tile]); - assert!( - !undelivered.contains(&archived_id(HIGH_LINK)), - "the arrival-endpoint link drops when its tile is unlisted" - ); - assert!( - undelivered.contains(&archived_id(LOW_LINK)), - "the self-loop control survives the narrowed tile list" - ); - - let slotless = admitting(&atlas, &[], &[LOW_LINK, HIGH_LINK]); - let hidden = ids(&slotless, vec![anchor_tile, arrival_tile]); - assert!( - !hidden.contains(&archived_id(HIGH_LINK)), - "the arrival-endpoint link drops when the proof hides the slot" - ); - assert!( - hidden.contains(&archived_id(LOW_LINK)), - "the self-loop control survives the hidden slot" - ); -} - -/// The ingress capture's withdrawn identity set filters what the retained cohort serves. -/// -/// The entry keeps its cohort while three later captures withdraw the link itself, a fitted -/// endpoint, and an arrival endpoint, each killing exactly its own edge at the next request. -/// The control capture withdraws an identity the response never names and answers byte-exactly -/// the capture-free response. -#[tokio::test] -async fn withdrawal_kills_retained_link() { - let (generation, atlas) = publish("delta-edges-ingress").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()); - let fitted = fitted_triples(&atlas, &generation); - - let serve = |ingress: Option<&DeltaSnapshot>| { - edges_with( - &atlas, - &FULL, - cohort, - ingress, - full_grid(), - EdgesLimits::default(), - ) - }; - - let mut merged = vec![( - node_wire(&atlas, 0), - node_wire(&atlas, 1), - archived_id(LOW_LINK), - )]; - merged.extend(fitted.iter().copied()); - merged.push(( - node_wire(&atlas, 0), - slot_wire.get(), - archived_id(HIGH_LINK), - )); - - let held = serve(None); - assert_eq!( - held, - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(merged)), - "the retained cohort serves both links before any withdrawal" - ); - - let link_withdrawn = edge_ids_of(&serve(Some(&withdrawing(&atlas, &[LOW_LINK])))); - assert!( - !link_withdrawn.contains(&archived_id(LOW_LINK)), - "withdrawing the link itself kills its edge" - ); - assert!( - link_withdrawn.contains(&archived_id(HIGH_LINK)), - "the unrelated link survives the link withdrawal" - ); - - let endpoint_withdrawn = edge_ids_of(&serve(Some(&withdrawing(&atlas, &[1])))); - assert!( - !endpoint_withdrawn.contains(&archived_id(LOW_LINK)), - "withdrawing a fitted endpoint kills the edge through the delivered set" - ); - assert!( - endpoint_withdrawn.contains(&archived_id(HIGH_LINK)), - "the unrelated link survives the endpoint withdrawal" - ); - - let arrival_withdrawn = edge_ids_of(&serve(Some(&withdrawing(&atlas, &[ARRIVAL])))); - assert!( - !arrival_withdrawn.contains(&archived_id(HIGH_LINK)), - "withdrawing the arrival endpoint kills its incident delta edge" - ); - assert!( - arrival_withdrawn.contains(&archived_id(LOW_LINK)), - "the fitted-endpoint link survives the arrival withdrawal" - ); - - assert_eq!( - serve(Some(&withdrawing(&atlas, &[60]))), - held, - "a capture withdrawing nothing the response names moves no byte" - ); -} - -/// The rank-ordered cap selects over the fitted-plus-delta union, arrivals ranking last. -/// -/// A cap of the fitted count drops exactly the arrival-endpoint link, because an arrival -/// endpoint ranks past every generation row, and the head reports the truncation. One more -/// slot serves the whole union complete. -#[tokio::test] -async fn cap_ranks_arrivals_last() { - let (generation, atlas) = publish("delta-edges-cap").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing(&atlas, &[(ARRIVAL, vacant)], &[(HIGH_LINK, 0, ARRIVAL)]); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()); - let fitted = fitted_triples(&atlas, &generation); - let fitted_count = u32::try_from(fitted.len()).expect("fixture edge counts fit u32"); - - let serve = |edges: u32| { - edges_with( - &atlas, - &FULL, - cohort, - None, - full_grid(), - EdgesLimits { - edges, - ..EdgesLimits::default() - }, - ) - }; - - assert_eq!( - serve(fitted_count), - expected_edges_bytes(&generation, false, &EdgeColumns::pinned(fitted.clone())), - "the cap drops the arrival-endpoint link first and reports the truncation" - ); - - let mut whole = fitted; - whole.push(( - node_wire(&atlas, 0), - slot_wire.get(), - archived_id(HIGH_LINK), - )); - assert_eq!( - serve(fitted_count + 1), - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(whole)), - "one more slot serves the whole union complete" - ); -} - -/// A union candidate's selection key beside its expected wire triple. -type KeyedTriple = ((u32, ArchivedEntityId), (u32, u32, ArchivedEntityId)); - -/// The fixture's endpoint row pairs and each node row's importance rank, from the edge artifacts. -fn endpoint_ranks(generation: &Generation) -> (Vec<[u64; 2]>, Vec) { - let artifacts = open_edge_artifacts(generation); - let endpoints = artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs") - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let ranks: Vec = artifacts - .ranks - .column::() - .expect("the rank column holds importance ranks") - .as_raw() - .iter() - .map(|rank| rank.as_u32()) - .collect(); - let row_ranks = artifacts - .positions - .column::() - .expect("the position column holds base positions") - .as_raw() - .iter() - .map(|position| ranks[position.as_usize()]) - .collect(); - - (endpoints, row_ranks) -} - -/// A truncating cap admits a winning delta link and drops arrival-endpoint links unpriced. -/// -/// The cap sits strictly below the fitted count, so the fitted walk itself truncates. The -/// cohort publishes one link between the two best-ranked rows, whose identity sorts before -/// every fitted edge, so its key wins a seat in a selection that is already dropping fitted -/// edges - the fold must keep pricing delta links after truncation begins. The arrival-endpoint -/// link keys past every fitted candidate and stays out. The expectation is the full-sort law -/// over the union, the same selection the bounded fold must reproduce. -#[tokio::test] -async fn truncating_cap_admits_winner() { - let (generation, atlas) = publish("delta-edges-rank-cap").await; - let (endpoints, row_ranks) = endpoint_ranks(&generation); - let rank_of_row = |row: usize| row_ranks[row]; - - // The published link attaches the best-ranked row pair, ties on the row. Its worse-endpoint - // rank therefore ties or beats every fitted edge's, and any tie falls to its lower identity. - let mut by_rank: Vec<(u32, usize)> = (0..row_ranks.len()) - .map(|row| (rank_of_row(row), row)) - .collect(); - by_rank.sort_unstable(); - let (best, second) = (by_rank[0].1, by_rank[1].1); - let best_seed = u8::try_from(best).expect("fixture rows fit u8"); - let second_seed = u8::try_from(second).expect("fixture rows fit u8"); - - let fitted = fitted_triples(&atlas, &generation); - // The union under the selection key: every fitted edge and the published link, keyed by - // worse-endpoint rank with ties on identity bytes. The arrival-endpoint link keys past - // every fitted candidate, so the fitted-domain key covers the whole competition and the - // link's absence is asserted on the delivered ids instead. - let mut union: Vec = endpoints - .iter() - .zip(&fitted) - .map(|(&[source, target], &triple)| { - let worse = rank_of_row(usize::try_from(source).expect("fixture rows fit usize")).max( - rank_of_row(usize::try_from(target).expect("fixture rows fit usize")), - ); - ((worse, triple.2), triple) - }) - .collect(); - union.push(( - ( - rank_of_row(best).max(rank_of_row(second)), - archived_id(LOW_LINK), - ), - ( - node_wire(&atlas, best_seed), - node_wire(&atlas, second_seed), - archived_id(LOW_LINK), - ), - )); - union.sort_unstable_by_key(|&(key, _)| key); - union.truncate(2); - assert!( - union - .iter() - .any(|&(_, (.., id))| id == archived_id(LOW_LINK)), - "the charter needs the published link winning a seat" - ); - let mut kept: Vec<(u32, u32, ArchivedEntityId)> = - union.into_iter().map(|(_, triple)| triple).collect(); - kept.sort_unstable_by_key(|&(.., id)| id); - - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, best_seed, second_seed), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let capped = edges_with( - &atlas, - &FULL, - cohort, - None, - full_grid(), - EdgesLimits { - edges: 2, - ..EdgesLimits::default() - }, - ); - assert_eq!( - capped, - expected_edges_bytes(&generation, false, &EdgeColumns::pinned(kept)), - "the truncating cap keeps exactly the full-sort head of the union" - ); - assert!( - !edge_ids_of(&capped).contains(&archived_id(HIGH_LINK)), - "the arrival-endpoint link stays out of a selection full of fitted keys" - ); - - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()); - let mut whole = vec![( - node_wire(&atlas, best_seed), - node_wire(&atlas, second_seed), - archived_id(LOW_LINK), - )]; - whole.extend(fitted.iter().copied()); - whole.push(( - node_wire(&atlas, 0), - slot_wire.get(), - archived_id(HIGH_LINK), - )); - assert_eq!( - edges_with( - &atlas, - &FULL, - cohort, - None, - full_grid(), - EdgesLimits::default() - ), - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(whole)), - "the uncapped serve delivers the whole union complete" - ); -} - -/// A full cap stays complete when the delta link refuses, and truncates when it qualifies. -/// -/// Both serves cap at exactly the fitted count, and the cohort publishes one arrival and one -/// link onto it. Withdrawing the arrival removes the link's endpoint from the delivered sets, -/// so the link refuses and the full selection stays complete - a full cap alone is not a -/// truncation. Retaining the arrival makes the link qualify, so it falls to the same cap and -/// the head reports the truncation. The delivered columns are the fitted set either way, and -/// only the completeness bit separates the two. -#[tokio::test] -async fn full_cap_refusal_vs_truncation() { - let (generation, atlas) = publish("delta-edges-full-cap").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing(&atlas, &[(ARRIVAL, vacant)], &[(HIGH_LINK, 0, ARRIVAL)]); - let cohort = PlacementCohort::of(Some(&snapshot)); - let fitted = fitted_triples(&atlas, &generation); - let limits = EdgesLimits { - edges: u32::try_from(fitted.len()).expect("fixture edge counts fit u32"), - ..EdgesLimits::default() - }; - - let withdrawn = withdrawing(&atlas, &[ARRIVAL]); - assert_eq!( - edges_with(&atlas, &FULL, cohort, Some(&withdrawn), full_grid(), limits), - expected_edges_bytes(&generation, true, &EdgeColumns::pinned(fitted.clone())), - "a link refused at the endpoint rule leaves the full cap complete" - ); - - assert_eq!( - edges_with(&atlas, &FULL, cohort, None, full_grid(), limits), - expected_edges_bytes(&generation, false, &EdgeColumns::pinned(fitted)), - "the qualifying link falls to the same cap and the head reports it" - ); -} - -/// A store answering per uuid map, asserting the one order the trailer places. -/// -/// The expected order is the distinct representative uuids in first-occurrence order over the -/// delivered slots, delta and fitted alike, so the assertion pins that hydration stopped -/// splitting the arms. The order is an internal convention rather than a wire property - -/// `Table::new` bytewise-sorts every trailer table, so the wire is order-invariant - and the -/// assertion pins the convention the store observes. -struct ExpectedTypesStore { - /// The distinct representative uuids the order must name, in first-occurrence order. - expected: Vec, - /// The resolved URL per uuid. An unlisted uuid answers `None`. - urls: FastHashMap, -} - -impl EdgesStore for ExpectedTypesStore { - #[expect( - clippy::panic_in_result_fn, - reason = "the merged order is the contract under test, and the assertion is its witness" - )] - fn hydrate( - self, - types: &IdSlice, - ) -> Result>, DetailError> { - assert_eq!( - types.iter().copied().collect::>(), - self.expected, - "the one order names the distinct representative uuids in first-occurrence order" - ); - - Ok(types - .iter() - .map(|uuid| self.urls.get(uuid).cloned()) - .collect()) - } -} - -/// The detail trailer carries each delta link's captured display at its own merged slot. -/// -/// The delta link's identity sorts before every fitted edge, so its captured label and interned -/// type occupy the column head rather than a tail, and its representative uuid takes the merged -/// order's first slot. The fitted labels stay the generation's payloads, empty under the -/// fixture's identity rewrite. -#[tokio::test] -async fn delta_display_head_slot() { - let (generation, atlas) = publish("delta-edges-trailer").await; - let url: VersionedUrl = "https://example.com/wired/v/1" - .parse() - .expect("the fixture URL parses"); - let delta_uuid = ArchivedOntologyTypeUuid::from_url(&url); - let wired = OwnedLabel::from("wired"); - let snapshot = publishing_displayed( - &atlas, - &[], - &[(LOW_LINK, 0, 1)], - &DisplayParts { - label: wired.clone(), - icon: OwnedIcon::from("wired-icon"), - representative: delta_uuid, - }, - ); - let fitted = fitted_triples(&atlas, &generation); - - let mut request = edges_request(full_grid()); - request.detail = EdgesDetail::Auxiliary; - let bound = Bound::resolved( - &atlas, - &FULL, - PlacementCohort::of(Some(&snapshot)), - CutOffset::ZERO, - ); - let bytes = atlas - .edges( - &request, - EdgesLimits::default(), - bound.view(&atlas), - ExpectedTypesStore { - expected: { - let rows: Vec = (0..u32::try_from(fitted.len()) - .expect("fixture edge rows fit u32")) - .collect(); - let mut expected = vec![delta_uuid]; - expected.extend(type_expectations(&generation, &rows).2); - expected - }, - urls: FastHashMap::from_iter([(delta_uuid, url.clone())]), - }, - ) - .expect("the detail request serves"); - - let mut merged = vec![( - node_wire(&atlas, 0), - node_wire(&atlas, 1), - archived_id(LOW_LINK), - )]; - merged.extend(fitted.iter().copied()); - let columns = EdgeColumns::pinned(merged); - - let type_table = [alloc::borrow::Cow::Owned(url.to_string())]; - let mut labels: Vec<&Label> = vec![&wired]; - labels.extend(core::iter::repeat_n(Label::EMPTY, fitted.len())); - let mut type_ids = vec![Some(crate::serve::intern::TableIndex::new(0))]; - type_ids.extend(core::iter::repeat_n(None, fitted.len())); - - let expected = EdgesResponse { - generation: generation.id().digest(), - variant: 0, - complete: true, - edges: &columns, - trailer: Some(EdgesTrailer { - type_table: IdSlice::from_raw(&type_table), - link_labels: IdSlice::from_raw(&labels), - link_type_ids: IdSlice::from_raw(&type_ids), - }), - } - .encode(); - assert_eq!( - bytes, expected, - "the delta display rides the head slot of the merged trailer" - ); -} - -/// The fixture geometry with each edge's type cycling over the three ontology rows. -/// -/// The shared fixture gives every edge one ontology row, so a single uuid satisfies any -/// scatter of the fitted order. Cycling the rows delivers repeated representatives in an -/// interleaved order, which makes the trailer's dedup and its first-occurrence order both -/// observable. -fn cycling_types_dataset() -> crate::dataset::memory::MemoryDataset { - use smallvec::smallvec; - use zerocopy::{LE, U64}; - - use crate::{ - dataset::{Edge, Ontology, card::Card, memory::MemoryDataset}, - identity::OntologyRowId, - }; - - let (nodes, canonical) = - super::fixture_nodes(|row| smallvec![OntologyRowId::from_usize(row & 1)]); - - let edges = super::FIXTURE_EDGES - .into_iter() - .zip([0_usize, 1, 2].into_iter().cycle()) - .map(|((id, source, target), ontology_row)| Edge { - id: U64::::new(id), - source: NodeRowId::new(source), - target: NodeRowId::new(target), - ontology: smallvec![OntologyRowId::from_usize(ontology_row)], - embedding: None, - confidence: None, - source_confidence: None, - target_confidence: None, - }) - .collect(); - - let ontology = vec![ - Ontology { - id: U64::::new(0), - parents: smallvec![], - }, - Ontology { - id: U64::::new(1), - parents: smallvec![], - }, - Ontology { - id: U64::::new(2), - parents: smallvec![], - }, - ]; - let cards = std::collections::HashMap::from([ - (0, Card::verbatim("Person entity card".to_owned())), - (1, Card::verbatim("Company entity card".to_owned())), - (2, Card::verbatim("Employment link card".to_owned())), - ]); - - MemoryDataset::new(nodes, edges, ontology, canonical, cards) -} - -/// The trailer resolves several fitted representatives beside a delta display in one response. -/// -/// The cycling dataset delivers three distinct representative uuids, so the merged order's -/// dedup and first-occurrence contract are exercised beyond one uuid while the delta link's -/// captured display occupies the head slot of the same trailer. Expected values derive from the -/// published artifacts alone, through [`type_expectations`]. -#[tokio::test] -async fn compose_trailer_types() { - let (generation, atlas) = - super::publish_dataset("delta-edges-composed-trailer", &cycling_types_dataset()).await; - let delta_url: VersionedUrl = "https://example.com/wired/v/1" - .parse() - .expect("the fixture URL parses"); - let delta_uuid = ArchivedOntologyTypeUuid::from_url(&delta_url); - let wired = OwnedLabel::from("wired"); - let snapshot = publishing_displayed( - &atlas, - &[], - &[(LOW_LINK, 0, 1)], - &DisplayParts { - label: wired.clone(), - icon: OwnedIcon::from("wired-icon"), - representative: delta_uuid, - }, - ); - let fitted = fitted_triples(&atlas, &generation); - - let rows: Vec = - (0..u32::try_from(fitted.len()).expect("fixture edge rows fit u32")).collect(); - let (mut urls, fitted_urls, expected_asked) = type_expectations(&generation, &rows); - assert!( - expected_asked.len() > 1, - "the cycling dataset delivers more than one distinct representative" - ); - urls.insert(delta_uuid, delta_url.clone()); - - let mut request = edges_request(full_grid()); - request.detail = EdgesDetail::Auxiliary; - let bound = Bound::resolved( - &atlas, - &FULL, - PlacementCohort::of(Some(&snapshot)), - CutOffset::ZERO, - ); - let bytes = atlas - .edges( - &request, - EdgesLimits::default(), - bound.view(&atlas), - ExpectedTypesStore { - expected: { - let mut expected = vec![delta_uuid]; - expected.extend(expected_asked); - expected - }, - urls, - }, - ) - .expect("the detail request serves"); - - let mut merged = vec![( - node_wire(&atlas, 0), - node_wire(&atlas, 1), - archived_id(LOW_LINK), - )]; - merged.extend(fitted.iter().copied()); - let columns = EdgeColumns::pinned(merged); - - let merged_urls: Vec> = core::iter::once(Some(delta_url)) - .chain(fitted_urls) - .collect(); - let table = crate::serve::intern::Table::new(merged_urls.iter().flatten()); - let type_ids: Vec>> = merged_urls - .iter() - .map(|url| url.as_ref().map(|url| table.index_of(url))) - .collect(); - let mut labels: Vec<&Label> = vec![&wired]; - labels.extend(core::iter::repeat_n(Label::EMPTY, fitted.len())); - - let expected = EdgesResponse { - generation: generation.id().digest(), - variant: 0, - complete: true, - edges: &columns, - trailer: Some(EdgesTrailer { - type_table: table.entries(), - link_labels: IdSlice::from_raw(&labels), - link_type_ids: IdSlice::from_raw(&type_ids), - }), - } - .encode(); - assert_eq!( - bytes, expected, - "the merged trailer carries each fitted representative's URL beside the delta display" - ); -} - -/// A revised fitted link's captured display reaches the edges trailer, overriding the -/// generation legend at that edge's own slot alone. -#[tokio::test] -async fn revised_fitted_trailer_overlay() { - let (generation, atlas) = publish("revised-fitted-trailer").await; - let fitted = fitted_triples(&atlas, &generation); - - // Fitted edge row 0's identity under the rewrite rule. - let revised_seed = 64_u8; - assert_eq!( - archived_id(revised_seed), - edge_identity_of(0), - "the witness revises fitted edge row 0" - ); - - let captured: VersionedUrl = "https://example.com/wired/v/1" - .parse() - .expect("the fixture URL parses"); - let captured_uuid = ArchivedOntologyTypeUuid::from_url(&captured); - let revised_label = OwnedLabel::from("revised"); - - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - let event = EntityEvent::Updated(EntityUpdate { - entity: store_id(revised_seed), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(revised_seed))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - register - .capture_display( - archived_id(revised_seed), - EntityEditionId::new(Uuid::from_u128(u128::from(revised_seed))), - &revised_label, - Icon::new("revised-icon"), - captured_uuid, - &atlas, - ) - .expect("the fixture ontology domain has room"); - let snapshot = register.snapshot( - &atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(2), - ); - assert!( - snapshot.legend_of(archived_id(revised_seed)).is_some(), - "publication carries the revised fitted identity's capture" - ); - - let fitted_url: VersionedUrl = fixture_type_url(2).parse().expect("the fixture URL parses"); - let fitted_uuid = ArchivedOntologyTypeUuid::from_url(&fitted_url); - - let mut request = edges_request(full_grid()); - request.detail = EdgesDetail::Auxiliary; - let bound = Bound::resolved( - &atlas, - &FULL, - PlacementCohort::of(Some(&snapshot)), - CutOffset::ZERO, - ); - let bytes = atlas - .edges( - &request, - EdgesLimits::default(), - bound.view(&atlas), - ExpectedTypesStore { - // The revised slot delivers first, so its captured uuid heads the merged order. - expected: vec![captured_uuid, fitted_uuid], - urls: FastHashMap::from_iter([ - (captured_uuid, captured.clone()), - (fitted_uuid, fitted_url.clone()), - ]), - }, - ) - .expect("the detail request serves"); - - let columns = EdgeColumns::pinned(fitted.iter().copied()); - let type_table = [ - alloc::borrow::Cow::Owned(fitted_url.to_string()), - alloc::borrow::Cow::Owned(captured.to_string()), - ]; - let mut labels: Vec<&Label> = vec![&revised_label]; - labels.extend(core::iter::repeat_n(Label::EMPTY, fitted.len() - 1)); - let mut type_ids = vec![Some(crate::serve::intern::TableIndex::new(1))]; - type_ids.extend(core::iter::repeat_n( - Some(crate::serve::intern::TableIndex::new(0)), - fitted.len() - 1, - )); - - let expected = EdgesResponse { - generation: generation.id().digest(), - variant: 0, - complete: true, - edges: &columns, - trailer: Some(EdgesTrailer { - type_table: IdSlice::from_raw(&type_table), - link_labels: IdSlice::from_raw(&labels), - link_type_ids: IdSlice::from_raw(&type_ids), - }), - } - .encode(); - assert_eq!( - bytes, expected, - "the revised fitted link's capture overrides its legend at its own slot alone" - ); -} - -/// A candidate's rank in the union order, derived independently of `EndpointRank`. -#[derive(Debug, Copy, Clone, PartialEq, Eq, PartialOrd, Ord)] -enum OracleRank { - /// A generation row's importance rank. - Fitted(u32), - /// A placed arrival's identity. - Arrival(ArchivedEntityId), -} - -/// The oracle union over one trial's survivors, keyed by (worse endpoint rank, identity). -/// -/// Fitted rows enter minus the withdrawn identities, and every drawn link whose endpoints both -/// resolve enters with the worse endpoint's rank, so the union is exactly what the fold may -/// select from. -fn fold_oracle_union( - atlas: &Atlas, - endpoints: &[[u64; 2]], - row_ranks: &[u32], - fitted: &[(u32, u32, ArchivedEntityId)], - withdrawn_edges: &[u8], - links: &[(u8, u8, u8)], - slot_wire: u32, -) -> Vec> { - let mut union: Vec> = Vec::new(); - for (row, (&[source, target], &triple)) in endpoints.iter().zip(fitted).enumerate() { - if withdrawn_edges.contains(&(super::EDGE_SEED + u8::try_from(row).expect("small"))) { - continue; - } - let worse = row_ranks[usize::try_from(source).expect("small")] - .max(row_ranks[usize::try_from(target).expect("small")]); - union.push(((OracleRank::Fitted(worse), triple.2), triple)); - } - for &(seed, source, target) in links { - let resolve = |endpoint: u8| -> Option<(OracleRank, u32)> { - if endpoint == ARRIVAL { - Some((OracleRank::Arrival(archived_id(ARRIVAL)), slot_wire)) - } else if usize::from(endpoint) < row_ranks.len() { - Some(( - OracleRank::Fitted(row_ranks[usize::from(endpoint)]), - node_wire(atlas, endpoint), - )) - } else { - None - } - }; - let (Some((source_rank, source_wire)), Some((target_rank, target_wire))) = - (resolve(source), resolve(target)) - else { - continue; - }; - - union.push(( - (source_rank.max(target_rank), archived_id(seed)), - (source_wire, target_wire, archived_id(seed)), - )); - } - union.sort_unstable_by_key(|&(key, _)| key); - union -} - -/// Differential: the fold's selection equals full-sort-then-truncate, at every cap. -/// -/// Randomised cohorts over the published fixture, each swept across every cap from zero to one -/// past the union size, against an oracle built from the artifacts alone: the union sorted by -/// (worse endpoint rank, identity), truncated, re-sorted by identity, with `complete` read as -/// `union.len() <= cap`. Withdrawn fitted link identities put non-qualifying candidates in the -/// walk, so the exactly-cap completeness law is under test at every trial. -#[tokio::test] -async fn fold_matches_full_sort() { - let (generation, atlas) = publish("review-fold-differential").await; - let (endpoints, row_ranks) = endpoint_ranks(&generation); - let fitted = fitted_triples(&atlas, &generation); - let (vacant, _) = vacant_cell(&atlas); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - let mut rng = keyed_rng(0x243F_6A88_85A3_08D3, 0, 0); - let mut next = - move |bound: u64| uniform_below(&mut rng, NonZero::new(bound).expect("a positive bound")); - - let mut mismatches: Vec = Vec::new(); - for trial in 0..16_u32 { - let count = if trial < 3 { - 0 - } else { - 1 + usize::try_from(next(LINK_SEEDS.len() as u64)).expect("small") - }; - let mut links: Vec<(u8, u8, u8)> = Vec::new(); - for &seed in &LINK_SEEDS[..count] { - let endpoint = |kind: u64, row: u64| -> u8 { - match kind { - 0 | 1 => ARRIVAL, - 2 => REFUSED, - _ => u8::try_from(row).expect("fixture rows fit u8"), - } - }; - let source = endpoint(next(16), next(48)); - let target = endpoint(next(16), next(48)); - links.push((seed, source, target)); - } - - // Withdrawn fitted link identities: candidates the walk reaches and the rule refuses. - let withdrawn_edges: Vec = (0..u8::try_from(fitted.len()).expect("small")) - .filter(|_| next(4) == 0) - .map(|row| super::EDGE_SEED + row) - .collect(); - let capture = withdrawing(&atlas, &withdrawn_edges); - let ingress = (!withdrawn_edges.is_empty()).then_some(&capture); - - let snapshot = publishing(&atlas, &[(ARRIVAL, vacant)], &links); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot_wire = test_codec(&atlas).encode(slot, snapshot.universe()).get(); - - let union = fold_oracle_union( - &atlas, - &endpoints, - &row_ranks, - &fitted, - &withdrawn_edges, - &links, - slot_wire, - ); - - for cap in 0..=(union.len() + 1) { - let mut kept: Vec<(u32, u32, ArchivedEntityId)> = - union.iter().take(cap).map(|&(_, triple)| triple).collect(); - kept.sort_unstable_by_key(|&(.., id)| id); - - let served = edges_with( - &atlas, - &FULL, - cohort, - ingress, - full_grid(), - EdgesLimits { - edges: u32::try_from(cap).expect("small"), - ..EdgesLimits::default() - }, - ); - let expected = expected_edges_bytes( - &generation, - union.len() <= cap, - &EdgeColumns::pinned(kept.clone()), - ); - if served != expected { - mismatches.push(format!( - "trial {trial} cap {cap} union {} links {links:?} withdrawn \ - {withdrawn_edges:?}", - union.len() - )); - } - } - } - - assert!( - mismatches.is_empty(), - "{} mismatches:\n{}", - mismatches.len(), - mismatches.join("\n") - ); -} - -/// The fixture geometry densified with reciprocal edge pairs. -/// -/// The first half of the stream carries each pair's high-row-first direction and the second half -/// its low-row-first direction, so the walk (rows ascending) offers the high-identity twin first -/// and the equal-ranked low-identity twin later - the arrival order the cap's tie-break rule has -/// to answer. -fn reciprocal_pairs_dataset() -> crate::dataset::memory::MemoryDataset { - use smallvec::smallvec; - use zerocopy::{LE, U64}; - - use crate::{ - dataset::{Edge, Ontology, card::Card, memory::MemoryDataset}, - identity::OntologyRowId, - }; - - const PAIRS: [(u64, u64); 12] = [ - (0, 1), - (2, 3), - (4, 5), - (6, 7), - (8, 9), - (10, 11), - (12, 13), - (14, 15), - (16, 17), - (18, 19), - (20, 21), - (22, 23), - ]; - - let (nodes, canonical) = - super::fixture_nodes(|row| smallvec![OntologyRowId::from_usize(row & 1)]); - - let directed: Vec<(u64, u64)> = PAIRS - .iter() - .map(|&(low, high)| (high, low)) - .chain(PAIRS.iter().copied()) - .collect(); - - let edges = directed - .iter() - .enumerate() - .map(|(row, &(source, target))| Edge { - id: U64::::new(100 + row as u64), - source: NodeRowId::new(source), - target: NodeRowId::new(target), - ontology: smallvec![OntologyRowId::new(2)], - embedding: None, - confidence: None, - source_confidence: None, - target_confidence: None, - }) - .collect(); - - let ontology = vec![ - Ontology { - id: U64::::new(0), - parents: smallvec![], - }, - Ontology { - id: U64::::new(1), - parents: smallvec![], - }, - Ontology { - id: U64::::new(2), - parents: smallvec![], - }, - ]; - let cards = std::collections::HashMap::from([ - (0, Card::verbatim("Person entity card".to_owned())), - (1, Card::verbatim("Company entity card".to_owned())), - (2, Card::verbatim("Employment link card".to_owned())), - ]); - - MemoryDataset::new(nodes, edges, ontology, canonical, cards) -} - -/// Differential over the reciprocal-pair fixture: rank ties at the cap boundary, every cap. -/// -/// Edges sharing their less-prominent endpoint share a key rank, so the tie-break law is under -/// test at every cap that cuts at a tie, where a rank equal to the kept worst's must still price -/// the identity, and the strict form of the rank exclusion is what a `>=` mutation breaks here -/// while every targeted witness stays green. -#[tokio::test] -async fn rank_tie_admits_better_identity() { - let (generation, atlas) = - super::publish_dataset("review-dense", &reciprocal_pairs_dataset()).await; - let (endpoints, row_ranks) = endpoint_ranks(&generation); - let fitted = fitted_triples(&atlas, &generation); - - let ties = { - let mut keys: Vec = endpoints - .iter() - .map(|&[source, target]| { - row_ranks[usize::try_from(source).expect("small")] - .max(row_ranks[usize::try_from(target).expect("small")]) - }) - .collect(); - keys.sort_unstable(); - keys.len() - { - keys.dedup(); - keys.len() - } - }; - assert!(ties > 0, "the dense fixture needs worse-endpoint rank ties"); - - let mut mismatches: Vec = Vec::new(); - let mut rng = keyed_rng(0x0BAD_C0FF_EE0D_DF00, 0, 0); - let mut next = - move |bound: u64| uniform_below(&mut rng, NonZero::new(bound).expect("a positive bound")); - - for trial in 0..4_u32 { - let withdrawn_edges: Vec = (0..u8::try_from(fitted.len()).expect("small")) - .filter(|_| trial > 0 && next(5) == 0) - .map(|row| super::EDGE_SEED + row) - .collect(); - let capture = withdrawing(&atlas, &withdrawn_edges); - let ingress = (!withdrawn_edges.is_empty()).then_some(&capture); - - let mut union: Vec> = Vec::new(); - for (row, (&[source, target], &triple)) in endpoints.iter().zip(&fitted).enumerate() { - if withdrawn_edges.contains(&(super::EDGE_SEED + u8::try_from(row).expect("small"))) { - continue; - } - let worse = row_ranks[usize::try_from(source).expect("small")] - .max(row_ranks[usize::try_from(target).expect("small")]); - union.push(((worse, triple.2), triple)); - } - union.sort_unstable_by_key(|&(key, _)| key); - - for cap in 0..=(union.len() + 1) { - let mut kept: Vec<(u32, u32, ArchivedEntityId)> = - union.iter().take(cap).map(|&(_, triple)| triple).collect(); - kept.sort_unstable_by_key(|&(.., id)| id); - - let served = edges_with( - &atlas, - &FULL, - PlacementCohort::EMPTY, - ingress, - full_grid(), - EdgesLimits { - edges: u32::try_from(cap).expect("small"), - ..EdgesLimits::default() - }, - ); - let expected = expected_edges_bytes( - &generation, - union.len() <= cap, - &EdgeColumns::pinned(kept.clone()), - ); - if served != expected { - mismatches.push(format!("trial {trial} cap {cap} union {}", union.len())); - } - } - } - - assert!( - mismatches.is_empty(), - "{} mismatches ({ties} rank ties):\n{}", - mismatches.len(), - mismatches.join("\n") - ); -} - -/// The translate request naming both published links. -fn link_ask() -> TranslateRequest { - TranslateRequest { - entity_ids: vec![entity_string_of(LOW_LINK), entity_string_of(HIGH_LINK)], - } -} - -/// Translate answers a published link's endpoint rows from the cohort, in both endpoint domains. -/// -/// One link joins two fitted rows and one joins a fitted row to a placed arrival, so the case -/// pins the domain split: fitted endpoints encode their generation rows and the arrival endpoint -/// its cohort slot, all under the snapshot's own universe. The empty-cohort control runs the -/// same request and must answer absent keys, which is the resolution that read no publication. -#[tokio::test] -async fn translate_answers_published_link_endpoints() { - let (_generation, atlas) = publish("delta-translate").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let codec = test_codec(&atlas); - let wire = |row: NodeRowId| codec.encode(row, snapshot.universe()); - - let response = atlas - .translate(link_ask(), TranslateLimits::default(), &FULL, None, cohort) - .expect("the request is under the cap"); - - assert_eq!( - response.edges.get(&entity_string_of(LOW_LINK)), - Some(&TranslatedEdge { - source: wire(NodeRowId::from_u32(0)), - target: wire(NodeRowId::from_u32(1)), - }), - "the fitted-endpoint link answers its generation rows" - ); - assert_eq!( - response.edges.get(&entity_string_of(HIGH_LINK)), - Some(&TranslatedEdge { - source: wire(NodeRowId::from_u32(0)), - target: wire(slot), - }), - "the arrival-endpoint link answers the cohort slot" - ); - assert!( - response.nodes.is_empty(), - "link-classified identities answer in the edges map alone" - ); - - let unresolved = atlas - .translate( - link_ask(), - TranslateLimits::default(), - &FULL, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - assert!( - unresolved.nodes.is_empty() && unresolved.edges.is_empty(), - "an empty cohort answers absent keys" - ); -} - -/// The ingress capture's withdrawn identity set filters translated links, per direction. -/// -/// The entry keeps its cohort while three later captures withdraw the link itself, a fitted -/// endpoint, and the arrival endpoint, each answering an absent key for exactly its own link at -/// the next request. The control capture withdraws an identity the request never names and must -/// leave the response equal to the baseline. -#[tokio::test] -async fn withdrawal_answers_absent_key_for_translated_link() { - let (_generation, atlas) = publish("delta-translate-ingress").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let translate = |ingress: Option<&DeltaSnapshot>| { - atlas - .translate( - link_ask(), - TranslateLimits::default(), - &FULL, - ingress, - cohort, - ) - .expect("the request is under the cap") - }; - - let baseline = translate(None); - assert!( - baseline.edges.contains_key(&entity_string_of(LOW_LINK)) - && baseline.edges.contains_key(&entity_string_of(HIGH_LINK)), - "the retained cohort answers both links before any withdrawal" - ); - - let link_withdrawn = translate(Some(&withdrawing(&atlas, &[LOW_LINK]))); - assert!( - !link_withdrawn - .edges - .contains_key(&entity_string_of(LOW_LINK)), - "withdrawing the link itself answers its absent key" - ); - assert!( - link_withdrawn - .edges - .contains_key(&entity_string_of(HIGH_LINK)), - "the unrelated link survives the link withdrawal" - ); - - let endpoint_withdrawn = translate(Some(&withdrawing(&atlas, &[1]))); - assert!( - !endpoint_withdrawn - .edges - .contains_key(&entity_string_of(LOW_LINK)), - "withdrawing a fitted endpoint kills the translated link" - ); - assert!( - endpoint_withdrawn - .edges - .contains_key(&entity_string_of(HIGH_LINK)), - "the unrelated link survives the endpoint withdrawal" - ); - - let arrival_withdrawn = translate(Some(&withdrawing(&atlas, &[ARRIVAL]))); - assert!( - !arrival_withdrawn - .edges - .contains_key(&entity_string_of(HIGH_LINK)), - "withdrawing the arrival endpoint kills its incident link" - ); - assert!( - arrival_withdrawn - .edges - .contains_key(&entity_string_of(LOW_LINK)), - "the fitted-endpoint link survives the arrival withdrawal" - ); - - assert_eq!( - translate(Some(&withdrawing(&atlas, &[60]))), - baseline, - "a capture withdrawing nothing the request names moves no key" - ); -} - -/// A scoped proof answers exactly the links its own resolution admitted, endpoint mask included. -/// -/// The identity set decides the link itself. Admitting one link answers it alone, and widening -/// the set adds exactly the second. The node mask decides the arrival endpoint. A proof -/// admitting both links while hiding the slot refuses the arrival-endpoint link whole, and the -/// fitted-endpoint link survives as the same-path control. -#[tokio::test] -async fn scoped_translate_admits_links_through_identity_set_and_mask() { - let (_generation, atlas) = publish("delta-translate-scope").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - let translate = |proof: &VisibilityProof| { - atlas - .translate(link_ask(), TranslateLimits::default(), proof, None, cohort) - .expect("the request is under the cap") - }; - - let one = translate(&admitting(&atlas, &[slot], &[LOW_LINK])); - assert!( - one.edges.contains_key(&entity_string_of(LOW_LINK)), - "the admitted link answers" - ); - assert!( - !one.edges.contains_key(&entity_string_of(HIGH_LINK)), - "an unadmitted link answers an absent key, whatever the cohort publishes" - ); - - let both = translate(&admitting(&atlas, &[slot], &[LOW_LINK, HIGH_LINK])); - assert!( - both.edges.contains_key(&entity_string_of(LOW_LINK)) - && both.edges.contains_key(&entity_string_of(HIGH_LINK)), - "widening the identity set adds exactly the second link" - ); - - let slotless = translate(&admitting(&atlas, &[], &[LOW_LINK, HIGH_LINK])); - assert!( - !slotless.edges.contains_key(&entity_string_of(HIGH_LINK)), - "a hidden slot refuses the arrival-endpoint link whole" - ); - assert!( - slotless.edges.contains_key(&entity_string_of(LOW_LINK)), - "the fitted-endpoint link survives the hidden slot" - ); -} - -/// The bound view over `proof` and `cohort`, with `ingress` as the request's capture. -fn viewing_delta<'scope>( - atlas: &'scope Atlas, - proof: &'scope VisibilityProof, - cohort: PlacementCohort<'scope>, - ingress: Option<&'scope DeltaSnapshot>, -) -> Bound<'scope> { - let mut bound = Bound::resolved(atlas, proof, cohort, CutOffset::ZERO); - if let Some(ingress) = ingress { - bound = bound.withdrawing(ingress); - } - - bound -} - -/// The locate ego-graph folds the cohort's incident links into both endpoint domains. -/// -/// Fitted row 0 carries one fitted edge, one published link to fitted row 1, and one published -/// link to a placed arrival. The subgraph merges all three ascending by identity, delivers the -/// arrival partner as its table vessel, and the empty-cohort control answers the fitted edge -/// alone on the same path. -#[tokio::test] -async fn locate_ego_graph_folds_cohort_links() { - let (_generation, atlas) = publish("delta-locate-fold").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let bound = viewing_delta(&atlas, &FULL, cohort, None); - let view = bound.view(&atlas); - - let source = atlas - .resolve_source(&view, &entity_string_of(0)) - .expect("fixture node ids resolve"); - let subgraph = atlas.locate_subgraph(source, LocateLimits::default(), &view); - - assert!(subgraph.complete, "three edges sit under the default cap"); - let ids: Vec = subgraph.edges.iter().map(|&(_, id)| id).collect(); - assert_eq!( - ids, - [ - archived_id(LOW_LINK), - edge_identity_of(0), - archived_id(HIGH_LINK) - ], - "both links merge into the identity order around the fitted edge" - ); - assert_eq!( - subgraph.edges[0].0, - ServedEdge::Delta(DeltaEdge { - source: DeltaEndpoint::Fitted(NodeRowId::from_u32(0)), - target: DeltaEndpoint::Fitted(NodeRowId::from_u32(1)), - }), - "the fitted-endpoint link resolves both rows" - ); - assert_eq!( - subgraph.edges[2].0, - ServedEdge::Delta(DeltaEdge { - source: DeltaEndpoint::Fitted(NodeRowId::from_u32(0)), - target: DeltaEndpoint::Arrival { - slot, - identity: archived_id(ARRIVAL), - }, - }), - "the arrival-endpoint link resolves the cohort slot" - ); - - // The delivered partners follow ascending wire id, whichever domain each encodes from. - let positions_of_row = atlas.positions_of_row(); - let codec = test_codec(&atlas); - let mut expected_partners = [ - ( - codec.encode(NodeRowId::from_u32(1), snapshot.universe()), - ViewRow::Base(positions_of_row[NodeRowId::from_u32(1)]), - ), - ( - view.arrivals()[ArrivalIndex::from_u32(0)].wire, - ViewRow::Arrival(ArrivalIndex::from_u32(0)), - ), - ]; - expected_partners.sort_unstable_by_key(|&(wire, _)| wire); - let mut expected = vec![ViewRow::Base(positions_of_row[NodeRowId::from_u32(0)])]; - expected.extend(expected_partners.iter().map(|&(_, vessel)| vessel)); - assert_eq!( - subgraph.delivered.as_raw(), - expected, - "partners deliver in their own vessels, ascending wire id" - ); - - // Same-path control: the empty cohort serves the fitted ego-graph alone. - let bare = viewing_delta(&atlas, &FULL, PlacementCohort::EMPTY, None); - let control = atlas.locate_subgraph(source, LocateLimits::default(), &bare.view(&atlas)); - assert_eq!( - control.edges.iter().map(|&(_, id)| id).collect::>(), - [edge_identity_of(0)], - "an empty cohort serves the fitted edge alone" - ); -} - -/// An arrival source's ego-graph serves its cohort links instead of a lone node. -/// -/// The link's fitted partner delivers beside the arrival source, and the no-links control keeps -/// the lone-node answer on the same path, so the fold widens the arrival source without moving -/// the linkless case. -#[tokio::test] -async fn arrival_source_ego_graph_serves_cohort_links() { - let (_generation, atlas) = publish("delta-locate-arrival-source").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing(&atlas, &[(ARRIVAL, vacant)], &[(HIGH_LINK, 0, ARRIVAL)]); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - let bound = viewing_delta(&atlas, &FULL, cohort, None); - let view = bound.view(&atlas); - - let source = atlas - .resolve_source(&view, &entity_string_of(ARRIVAL)) - .expect("the cohort resolves the arrival"); - let subgraph = atlas.locate_subgraph(source, LocateLimits::default(), &view); - - assert!(subgraph.complete, "one edge sits under the default cap"); - assert_eq!( - subgraph.edges, - vec![( - ServedEdge::Delta(DeltaEdge { - source: DeltaEndpoint::Fitted(NodeRowId::from_u32(0)), - target: DeltaEndpoint::Arrival { - slot, - identity: archived_id(ARRIVAL), - }, - }), - archived_id(HIGH_LINK), - )], - "the arrival source serves its incident link" - ); - assert_eq!( - subgraph.delivered.as_raw(), - [ - ViewRow::Arrival(ArrivalIndex::from_u32(0)), - ViewRow::Base(atlas.positions_of_row()[NodeRowId::from_u32(0)]), - ], - "the fitted partner delivers beside the arrival source" - ); - - // Same-path control: a cohort publishing no link keeps the lone-node answer. - let bare_snapshot = publishing(&atlas, &[(ARRIVAL, vacant)], &[]); - let bare_cohort = PlacementCohort::of(Some(&bare_snapshot)); - let bare = viewing_delta(&atlas, &FULL, bare_cohort, None); - let bare_view = bare.view(&atlas); - let bare_source = atlas - .resolve_source(&bare_view, &entity_string_of(ARRIVAL)) - .expect("the cohort resolves the arrival"); - let control = atlas.locate_subgraph(bare_source, LocateLimits::default(), &bare_view); - assert!(control.complete, "no edge qualifies"); - assert!(control.edges.is_empty(), "no link publishes at the source"); - assert_eq!( - control.delivered.as_raw(), - [ViewRow::Arrival(ArrivalIndex::from_u32(0))], - "the linkless arrival delivers alone" - ); -} - -/// The ingress capture's withdrawn identity set filters the locate fold, per direction. -/// -/// Withdrawing the link kills its edge alone. Withdrawing fitted row 1 kills the published link -/// and the fitted edge through one rule. Withdrawing the arrival kills its incident link alone, -/// and the unrelated control moves nothing. -#[tokio::test] -async fn withdrawal_subtracts_from_locate_fold() { - let (_generation, atlas) = publish("delta-locate-ingress").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let ids = |ingress: Option<&DeltaSnapshot>| -> Vec { - let bound = viewing_delta(&atlas, &FULL, cohort, ingress); - let view = bound.view(&atlas); - let source = atlas - .resolve_source(&view, &entity_string_of(0)) - .expect("fixture node ids resolve"); - - atlas - .locate_subgraph(source, LocateLimits::default(), &view) - .edges - .iter() - .map(|&(_, id)| id) - .collect() - }; - - let baseline = ids(None); - assert_eq!( - baseline, - [ - archived_id(LOW_LINK), - edge_identity_of(0), - archived_id(HIGH_LINK) - ], - "the retained cohort serves the whole fold before any withdrawal" - ); - - assert_eq!( - ids(Some(&withdrawing(&atlas, &[LOW_LINK]))), - [edge_identity_of(0), archived_id(HIGH_LINK)], - "withdrawing the link kills its edge alone" - ); - assert_eq!( - ids(Some(&withdrawing(&atlas, &[1]))), - [archived_id(HIGH_LINK)], - "withdrawing the shared partner kills the published link and the fitted edge" - ); - assert_eq!( - ids(Some(&withdrawing(&atlas, &[ARRIVAL]))), - [archived_id(LOW_LINK), edge_identity_of(0)], - "withdrawing the arrival kills its incident link alone" - ); - assert_eq!( - ids(Some(&withdrawing(&atlas, &[60]))), - baseline, - "a withdrawal the fold never names moves nothing" - ); -} - -/// A scoped locate serves exactly the links its own resolution admitted, slot mask included. -/// -/// The identity set decides each link, and hiding the slot empties the view's arrival table, so -/// the arrival-endpoint link refuses whole while the fitted-endpoint link and the fitted edge -/// survive as the same-path controls. -#[tokio::test] -async fn scoped_locate_admits_links_through_identity_set_and_mask() { - let (_generation, atlas) = publish("delta-locate-scope").await; - let (vacant, _) = vacant_cell(&atlas); - let snapshot = publishing( - &atlas, - &[(ARRIVAL, vacant)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - let ids = |proof: &VisibilityProof| -> Vec { - let bound = viewing_delta(&atlas, proof, cohort, None); - let view = bound.view(&atlas); - let source = atlas - .resolve_source(&view, &entity_string_of(0)) - .expect("fixture node ids resolve"); - - atlas - .locate_subgraph(source, LocateLimits::default(), &view) - .edges - .iter() - .map(|&(_, id)| id) - .collect() - }; - - assert_eq!( - ids(&admitting(&atlas, &[slot], &[LOW_LINK])), - [archived_id(LOW_LINK), edge_identity_of(0)], - "the admitted link serves and the unadmitted one refuses" - ); - assert_eq!( - ids(&admitting(&atlas, &[slot], &[LOW_LINK, HIGH_LINK])), - [ - archived_id(LOW_LINK), - edge_identity_of(0), - archived_id(HIGH_LINK) - ], - "widening the identity set adds exactly the second link" - ); - assert_eq!( - ids(&admitting(&atlas, &[], &[LOW_LINK, HIGH_LINK])), - [archived_id(LOW_LINK), edge_identity_of(0)], - "a hidden slot refuses the arrival-endpoint link whole" - ); -} - -/// The nearest-partner truncation prices arrival partners on their recorded coordinates. -/// -/// The arrival places at the wire corner opposite the source, strictly farther than fitted row -/// 1, so a cap of two keeps both row-1 edges and drops the arrival-endpoint link with its -/// partner. One more slot serves the whole fold complete. -#[tokio::test] -async fn truncation_prices_arrival_partners() { - let (_generation, atlas) = publish("delta-locate-truncation").await; - let positions = atlas.positions(); - let positions_of_row = atlas.positions_of_row(); - let origin = positions[positions_of_row[NodeRowId::from_u32(0)]]; - let near = positions[positions_of_row[NodeRowId::from_u32(1)]]; - let far = Vec2::new( - 0.99_f32.copysign(-origin.x()), - 0.99_f32.copysign(-origin.y()), - ); - let distance = |point: Vec2| { - let (dx, dy) = (point.x() - origin.x(), point.y() - origin.y()); - #[expect( - clippy::suboptimal_flops, - reason = "unfused arithmetic mirrors the selection key exactly" - )] - (dx * dx + dy * dy).to_bits() - }; - assert!( - distance(near) < distance(far), - "the charter needs the arrival strictly farther than fitted row 1" - ); - - let snapshot = publishing( - &atlas, - &[(ARRIVAL, far)], - &[(LOW_LINK, 0, 1), (HIGH_LINK, 0, ARRIVAL)], - ); - let cohort = PlacementCohort::of(Some(&snapshot)); - let bound = viewing_delta(&atlas, &FULL, cohort, None); - let view = bound.view(&atlas); - let source = atlas - .resolve_source(&view, &entity_string_of(0)) - .expect("fixture node ids resolve"); - - let capped = |edges: u32| { - atlas.locate_subgraph( - source, - LocateLimits { - edges, - ..LocateLimits::default() - }, - &view, - ) - }; - - let two = capped(2); - assert!(!two.complete, "the cap truncated the farthest partner"); - assert_eq!( - two.edges.iter().map(|&(_, id)| id).collect::>(), - [archived_id(LOW_LINK), edge_identity_of(0)], - "both row-1 edges outrank the arrival-endpoint link" - ); - assert_eq!( - two.delivered.as_raw(), - [ - ViewRow::Base(positions_of_row[NodeRowId::from_u32(0)]), - ViewRow::Base(positions_of_row[NodeRowId::from_u32(1)]), - ], - "the truncated arrival partner leaves with its edge" - ); - - let whole = capped(3); - assert!(whole.complete, "one more slot serves the whole fold"); - assert_eq!( - whole.edges.iter().map(|&(_, id)| id).collect::>(), - [ - archived_id(LOW_LINK), - edge_identity_of(0), - archived_id(HIGH_LINK) - ], - ); -} - -/// A store answering every delivered node and link as resolved with no recorded detail. -/// -/// The resolution flags open and every store-derived column stays empty, so an expectation -/// built over it pins the in-process label columns - the captured displays among them - -/// without store-derived content. -struct ResolvedEmptyDetails; - -impl LocateStore for ResolvedEmptyDetails { - fn hydrate(self, order: LocateOrder<'_>) -> Result { - Ok(LocateHydration { - nodes: LocateNodeHydration { - resolved: DenseBitSet::new_filled(order.nodes.count()), - type_urls: IdVec::from_elem(Vec::new(), order.nodes.count()), - source_properties: Some(Vec::new()), - source_properties_complete: true, - }, - links: LocateLinkHydration { - type_urls: IdVec::from_elem(Vec::new(), order.links.len()), - type_urls_complete: DenseBitSlice::new_empty(order.links.len()), - properties: IdVec::from_elem(Some(Vec::new()), order.links.len()), - properties_complete: DenseBitSlice::new_empty(order.links.len()), - }, - }) - } -} - -/// Captures each `(seed, label)`'s display with a revised icon over one register and publishes. -fn capturing_displays(atlas: &Atlas, captures: &[(u8, &OwnedLabel)]) -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - for &(seed, label) in captures { - let event = EntityEvent::Updated(EntityUpdate { - entity: store_id(seed), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - register - .capture_display( - archived_id(seed), - EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - label, - Icon::new("revised-icon"), - fixture_type(), - atlas, - ) - .expect("the fixture ontology domain has room"); - } - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(2), - ) -} - -/// Serves fitted node 3's locate under `snapshot`. -fn locate_node_three(atlas: &Atlas, snapshot: Option<&DeltaSnapshot>) -> Vec { - let bound = viewing_delta(atlas, &FULL, PlacementCohort::of(snapshot), None); - atlas - .locate( - &locate_request(entity_string_of(3)), - ServeLimits::default(), - bound.view(atlas), - ResolvedEmptyDetails, - ) - .expect("the locate request serves") -} - -/// The expected node-3 locate envelope, built directly. -/// -/// Fixture edge row 5 joins rows 3 and 7, the only edge at either row, so only the two label -/// arguments separate a capture's envelope from the baseline. -fn expected_node_three_envelope( - atlas: &Atlas, - generation: &Generation, - source_label: &Label, - link_label: &Label, -) -> Vec { - let source_row = NodeRowId::from_u32(3); - let partner_row = NodeRowId::from_u32(7); - let link_seed = 64 + 5; - assert_eq!( - archived_id(link_seed), - edge_identity_of(5), - "the witness revises fitted edge row 5" - ); - - let bound = viewing_delta(atlas, &FULL, PlacementCohort::EMPTY, None); - let view = bound.view(atlas); - let cell = atlas - .resolve_source(&view, &entity_string_of(3)) - .expect("fixture node ids resolve") - .cell; - let positions_of_row = atlas.positions_of_row(); - let codec = test_codec(atlas); - let wire = |row: NodeRowId| codec.encode(row, atlas.node_universe()); - let columns = EdgeColumns::pinned([( - wire(source_row).get(), - wire(partner_row).get(), - archived_id(link_seed), - )]); - let empty_map = PropertyMap::new_unchecked(Vec::new()); - let link_flags: Box> = - DenseBitSlice::new_empty(1); - - LocateResponse { - generation: generation.id().digest(), - variant: 0, - cell, - complete: true, - entity_id: archived_id(3), - type_ids_complete: false, - properties_complete: true, - delivered: IdSlice::from_raw(&[ - ViewRow::Base(positions_of_row[source_row]), - ViewRow::Base(positions_of_row[partner_row]), - ]), - arrivals: IdSlice::from_raw(&[]), - positions: atlas.positions(), - rows: atlas.wire_rows(), - masks: None, - edges: &columns, - trailer: LocateTrailer { - type_table: IdSlice::from_raw(&[]), - property_table: IdSlice::from_raw(&[]), - labels: IdSlice::from_raw(&[source_label, Label::EMPTY]), - type_ids: IdSlice::from_raw(&[None, None]), - properties: Some(&empty_map), - link_labels: IdSlice::from_raw(&[link_label]), - link_type_ids: IdSlice::from_raw(&[Vec::new()]), - link_type_ids_complete: &link_flags, - link_properties: IdSlice::from_raw(&[Some(&empty_map)]), - link_properties_complete: &link_flags, - }, - } - .encode() -} - -/// Captured displays reach the locate trailer's node and link labels, byte-exact. -/// -/// The register captures a revised display for fitted node 3 and for its one incident fitted -/// link, and the locate response serves both captured labels at their own slots. -#[tokio::test] -async fn captured_displays_reach_locate_labels() { - let (generation, atlas) = publish("delta-locate-labels").await; - - let renamed = OwnedLabel::from("renamed"); - let rewired = OwnedLabel::from("rewired"); - let captured = capturing_displays(&atlas, &[(3, &renamed), (64 + 5, &rewired)]); - - assert_eq!( - locate_node_three(&atlas, Some(&captured)), - expected_node_three_envelope(&atlas, &generation, &renamed, &rewired), - "both captured labels serve at their own slots" - ); -} - -/// An empty cohort serves the payload labels. -#[tokio::test] -async fn locate_baseline_payload_labels() { - let (generation, atlas) = publish("delta-locate-baseline").await; - - assert_eq!( - locate_node_three(&atlas, None), - expected_node_three_envelope(&atlas, &generation, Label::EMPTY, Label::EMPTY), - "the baseline serves the payload labels" - ); -} - -/// Same-path control: a capture the response never delivers moves no byte. -#[tokio::test] -async fn unrelated_capture_baseline_bytes() { - let (_generation, atlas) = publish("delta-locate-unrelated").await; - - let renamed = OwnedLabel::from("renamed"); - let unrelated = capturing_displays(&atlas, &[(60, &renamed)]); - - assert_eq!( - locate_node_three(&atlas, Some(&unrelated)), - locate_node_three(&atlas, None), - "a capture the response never names moves nothing" - ); -} - -/// Publishes the dormant-arrival shape into one register. -/// -/// The arrival arrives live, classifies, and places, so its slot allocates and stands for the -/// register's life. The link then attaches fitted row 0 to that arrival, live and captured. The -/// arrival's own end lands last, the feed order the register cannot forbid. The link therefore -/// keeps standing live while the `nodes` map drops the dormant holder. Every read must refuse the -/// link through `node_at`'s absent answer instead of resolving a slot no arrival table holds. -fn dormant_arrival_snapshot(atlas: &Atlas) -> DeltaSnapshot { - let (vacant, _) = vacant_cell(atlas); - - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - - let arrival_live = EntityEvent::Updated(EntityUpdate { - entity: store_id(ARRIVAL), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(ARRIVAL))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&arrival_live)); - register - .classify(archived_id(ARRIVAL), Classification::Node) - .expect("the fixture stays inside the edge universe"); - register - .place( - archived_id(ARRIVAL), - &ProjectedArrival { - edition: EntityEditionId::new(Uuid::from_u128(u128::from(ARRIVAL))), - position: vacant, - label: OwnedLabel::from("arrival"), - icon: OwnedIcon::from("arrival-icon"), - representative: fixture_type(), - }, - atlas, - ) - .expect("the fixture universe is far from the wire's row domain"); - - let link_live = EntityEvent::Updated(EntityUpdate { - entity: store_id(HIGH_LINK), - edition: EntityEditionId::new(Uuid::from_u128(u128::from(HIGH_LINK))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&link_live)); - register - .classify( - archived_id(HIGH_LINK), - Classification::Edge { - source: Some(archived_id(0)), - target: Some(archived_id(ARRIVAL)), - }, - ) - .expect("the fixture stays inside the edge universe"); - register - .capture_display( - archived_id(HIGH_LINK), - EntityEditionId::new(Uuid::from_u128(u128::from(HIGH_LINK))), - &OwnedLabel::from("link"), - &OwnedIcon::from("link-icon"), - fixture_type(), - atlas, - ) - .expect("the fixture ontology domain has room"); - - let arrival_end = EntityEvent::Ended(EntityEnd { - entity: store_id(ARRIVAL), - ended_at: Timestamp::from_unix_timestamp(3), - }); - register.apply(DeltaEvent::from(&arrival_end)); - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(3), - ) -} - -/// The dormant-arrival publication keeps the link and withdraws the identity, while the holder -/// leaves the nodes map. -#[tokio::test] -async fn dormant_arrival_publication_shape() { - let (_generation, atlas) = publish("probe-dormant-shape").await; - let snapshot = dormant_arrival_snapshot(&atlas); - let slot = NodeRowId::from_usize(atlas.node_universe().size()); - - assert!( - snapshot.edge(archived_id(HIGH_LINK)).is_some(), - "the link publishes even though its arrival endpoint stands withdrawn", - ); - assert!( - snapshot.node_at(slot).is_none(), - "the dormant holder leaves the nodes map, so the slot resolves to no arrival", - ); - assert!( - snapshot.withdraws(archived_id(ARRIVAL)), - "the same publication withdraws the arrival identity", - ); - - let cohort = PlacementCohort::of(Some(&snapshot)); - let bound = viewing_delta(&atlas, &FULL, cohort, None); - assert!( - bound.view(&atlas).arrivals().is_empty(), - "the view's arrival table holds no dormant holder", - ); -} - -/// Locate refuses the dormant-endpoint link rather than fabricating an unindexable endpoint. -#[tokio::test] -async fn locate_refuses_dormant_link() { - let (_generation, atlas) = publish("probe-dormant-locate").await; - let snapshot = dormant_arrival_snapshot(&atlas); - let bound = viewing_delta(&atlas, &FULL, PlacementCohort::of(Some(&snapshot)), None); - let view = bound.view(&atlas); - - let source = atlas - .resolve_source(&view, &entity_string_of(0)) - .expect("fixture node ids resolve"); - let subgraph = atlas.locate_subgraph(source, LocateLimits::default(), &view); - assert!( - !subgraph - .edges - .iter() - .any(|&(_, id)| id == archived_id(HIGH_LINK)), - "the dormant-endpoint link never reaches the locate ego graph", - ); - assert_eq!( - subgraph.delivered.len(), - 2, - "the fitted ego graph delivers the source and its one fitted partner alone", - ); -} - -/// Translate answers an absent key for the dormant-endpoint link. -#[tokio::test] -async fn translate_dormant_link_absent() { - let (_generation, atlas) = publish("probe-dormant-translate").await; - let snapshot = dormant_arrival_snapshot(&atlas); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let response = atlas - .translate( - TranslateRequest { - entity_ids: vec![entity_string_of(HIGH_LINK)], - }, - TranslateLimits::default(), - &FULL, - None, - cohort, - ) - .expect("the request is under the cap"); - assert!( - response.edges.is_empty() && response.nodes.is_empty(), - "translate answers an absent key for the dormant-endpoint link", - ); -} - -/// The edges route's delivered set refuses the dormant-endpoint link. -#[tokio::test] -async fn edges_refuse_dormant_link() { - let (_generation, atlas) = publish("probe-dormant-edges").await; - let snapshot = dormant_arrival_snapshot(&atlas); - let cohort = PlacementCohort::of(Some(&snapshot)); - - let served = edges_with( - &atlas, - &FULL, - cohort, - None, - full_grid(), - EdgesLimits::default(), - ); - assert!( - !edge_ids_of(&served).contains(&archived_id(HIGH_LINK)), - "the edges route refuses the dormant-endpoint link", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/density.rs b/libs/@local/graph/atlas/src/serve/tests/density.rs deleted file mode 100644 index 9293143ff26..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/density.rs +++ /dev/null @@ -1,226 +0,0 @@ -//! The visible-view occupancy census over a real published generation. -//! -//! The aggregate a delivery-cut policy reads is a statement about the authorized view. These cases -//! prove it over the fixture's own artifacts: expectations count distinct key prefixes with a set, -//! independently of the histogram derivation the census uses. -//! -//! The fixture's 48 rows share 8 distinct keys, so which rows a mask hides decides whether the -//! aggregate can move at all - the two cases below are exactly those two regimes. - -use core::num::NonZero; -use std::collections::{HashMap, HashSet}; - -use hashql_core::id::Id as _; - -use super::{FIXTURE_LOD, FULL, mask_hiding, publish}; -use crate::{ - identity::{BasePosition, NodeRowId}, - morton::{Depth, MortonKey}, - serve::{ - Atlas, CutOffset, DensityBand, DensityPolicy, ViewOccupancy, visibility::ProofKind, - walk::Walk, - }, -}; - -/// The keys of the rows a mask leaves visible, read straight from the code column. -fn visible_keys(atlas: &Atlas, hidden: &HashSet) -> Vec { - let rows = atlas.row_ids(); - - rows.ids() - .filter(|&position| !hidden.contains(&rows[position])) - .map(|position| atlas.morton.code(position)) - .collect() -} - -/// Groups the generation's rows by the complete key they occupy. -fn clusters(atlas: &Atlas) -> HashMap> { - let rows = atlas.row_ids(); - let mut clusters: HashMap> = HashMap::new(); - for (position, &row) in rows.iter_enumerated() { - let key = atlas.morton.code(position); - clusters.entry(key.to_bits()).or_default().push(row); - } - - clusters -} - -/// Counts the distinct depth-`depth` cells of `keys` with a set. -fn prefix_census(keys: &[MortonKey], depth: Depth) -> u64 { - let cells: HashSet = keys.iter().map(|key| key.prefix(depth)).collect(); - - u64::try_from(cells.len()).expect("a fixture cell count fits u64") -} - -/// Asserts a census matches the independent prefix count at every depth. -#[track_caller] -fn assert_census(occupancy: &ViewOccupancy, keys: &[MortonKey]) { - let distinct: HashSet = keys.iter().map(|key| key.to_bits()).collect(); - assert_eq!( - occupancy.distinct_keys(), - u64::try_from(distinct.len()).expect("a fixture key count fits u64") - ); - assert_eq!(occupancy.is_empty(), keys.is_empty()); - - for depth in Depth::all() { - assert_eq!( - occupancy.occupied_cells(depth), - prefix_census(keys, depth), - "depth {}", - depth.get() - ); - } -} - -/// Hiding a whole cluster removes its cell from every depth's count. -/// -/// The delivery-cut policy rests on this bug class. A census reaching past the mask would let a -/// hidden row set how deep an authorized view goes. The mask must therefore move the aggregate by -/// exactly the cell the hidden cluster occupied, which is why the case hides every row sharing one -/// key rather than an arbitrary subset. -#[tokio::test] -async fn hiding_a_cluster_removes_its_cell() { - let (_generation, atlas) = publish("cluster-occupancy").await; - - let full = Walk::of(&atlas, &FULL).visible_occupancy(); - assert_census(&full, &visible_keys(&atlas, &HashSet::new())); - - let clusters = clusters(&atlas); - assert_eq!( - full.distinct_keys(), - u64::try_from(clusters.len()).expect("a fixture cluster count fits u64"), - "one occupied cell per distinct key at the deepest depth" - ); - - // The cluster of the first position's key: hiding a whole cluster is what vacates a cell, - // since the rows sharing a key occupy one cell between them. - let key = atlas.morton.code(BasePosition::MIN); - let mut hidden_rows = clusters - .get(&key.to_bits()) - .expect("the first position's key holds its own rows") - .clone(); - hidden_rows.sort_unstable(); - let hidden: HashSet = hidden_rows.iter().copied().collect(); - - // The proof builder speaks the fixture's raw-u32 row vocabulary; narrow checked, once. - let hidden_mask: Vec = hidden_rows - .iter() - .map(|&row| u32::try_from(row).expect("fixture rows fit u32")) - .collect(); - let proof = mask_hiding(&atlas, &hidden_mask); - let masked = Walk::of(&atlas, &proof).visible_occupancy(); - assert_census(&masked, &visible_keys(&atlas, &hidden)); - - // The mask left exactly one cell fewer at the depths that separated the vacated cell, and - // never one more at any depth. - assert_eq!(masked.distinct_keys() + 1, full.distinct_keys()); - for depth in Depth::all() { - assert!( - masked.occupied_cells(depth) <= full.occupied_cells(depth), - "the mask grew occupancy at depth {}", - depth.get() - ); - } - assert!(masked.saturation_depth() <= full.saturation_depth()); -} - -/// Hiding rows that share their cells with visible rows moves no count. -/// -/// Occupancy counts cells, not rows: a row-counting aggregate would resolve a coarser cut for a -/// scope than for the operator over identical geometry, and every count the policy reads would -/// drift with permissions that changed nothing about the view's shape. The fixture's clusters make -/// this observable - every third row leaves each cell still occupied. -#[tokio::test] -async fn hiding_co_located_rows_moves_no_count() { - let (_generation, atlas) = publish("co-located-occupancy").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - - let hidden_rows: Vec = (0..universe).filter(|row| row.is_multiple_of(3)).collect(); - let hidden: HashSet = hidden_rows - .iter() - .copied() - .map(NodeRowId::from_u32) - .collect(); - let survivors = visible_keys(&atlas, &hidden); - let all = visible_keys(&atlas, &HashSet::new()); - assert!( - survivors.len() < all.len(), - "the mask hides rows even though it vacates no cell" - ); - - let proof = mask_hiding(&atlas, &hidden_rows); - let masked = Walk::of(&atlas, &proof).visible_occupancy(); - let full = Walk::of(&atlas, &FULL).visible_occupancy(); - - assert_census(&masked, &survivors); - assert_eq!(masked, full); -} - -/// A proof admitting nothing yields an empty aggregate. -/// -/// The degenerate view a scope with no permissions presents. An aggregate carrying the domain's -/// floor of one occupied cell would report geometry to a caller who may see none. -#[tokio::test] -async fn proof_admitting_nothing_occupies_nothing() { - let (_generation, atlas) = publish("empty-occupancy").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - - let hidden: Vec = (0..universe).collect(); - let proof = mask_hiding(&atlas, &hidden); - let occupancy = Walk::of(&atlas, &proof).visible_occupancy(); - - assert!(occupancy.is_empty()); - assert_eq!(occupancy.distinct_keys(), 0); - assert_eq!(occupancy.saturation_depth(), Depth::MIN); -} - -/// The delivery-cut policy is offered no operator view to resolve. -/// -/// An operator proof serves the corpus schedule, whose one cut per zoom takes no offset, so an -/// issuance over one seals zero. The case pins that at the proof-kind check rather than at the -/// issuance. The -/// fixture's own corpus occupancy resolves nonzero under the band below, which the first assertion -/// states, and the absent aggregate is therefore about the proof's kind. A view that happened to -/// resolve zero would pass the same check for the wrong reason. -/// -/// A mask admitting every row of the generation is the other half. It sees exactly what the -/// operator proof sees and still answers its own aggregate, because it serves its own cascade: a -/// declared scope stays a scope whatever its masks admit. -#[tokio::test] -async fn operator_proof_offers_the_policy_no_view() { - let (_generation, atlas) = publish("operator-offset").await; - let policy = DensityPolicy::new( - DensityBand::new( - NonZero::new(8).expect("the band's lower bound is positive"), - NonZero::new(8).expect("the band's upper bound is positive"), - ) - .expect("the band is ordered"), - FIXTURE_LOD.span, - FIXTURE_LOD.max_tile_depth, - ) - .expect("the fixture schedule admits an offset"); - - let corpus = Walk::of(&atlas, &FULL).visible_occupancy(); - assert_ne!( - policy.resolve(&corpus), - CutOffset::ZERO, - "the fixture must resolve a deeper cut for the corpus view, or this case cannot fail", - ); - - assert_eq!( - FULL.kind(), - ProofKind::Corpus, - "an operator proof has no view for the policy to resolve", - ); - - let all_rows = mask_hiding(&atlas, &[]); - assert_eq!( - all_rows.kind(), - ProofKind::Scope, - "a scope admitting every row still serves its own cascade", - ); - assert_eq!( - atlas.visible_occupancy(&all_rows), - corpus, - "that scope's aggregate is the corpus one: the two differ by declared contract alone", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/frame_channel.rs b/libs/@local/graph/atlas/src/serve/tests/frame_channel.rs deleted file mode 100644 index 9845d422fe1..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/frame_channel.rs +++ /dev/null @@ -1,220 +0,0 @@ -//! Controlled proof that the visible set's key assignment is corpus-dependent. -//! -//! The delivery contract's fixed-view comparison keeps visible row identity, visible Morton keys, -//! and relative importance order among visible rows identical across two worlds. The scope cascade -//! closes the channel that runs through the *bucket assignment*; this module is about the channel -//! one layer further up, in the keys the cascade reads. -//! -//! Keys are not row-local. `salt/lod/stage.rs` fits the world frame from every corpus coordinate -//! (`Bounds2::from_slice_par`), maps that frame onto the fixed `[-1, 1]` wire frame -//! (`Bounds2::normalize_into`), and quantizes the mapped column (`key::keys`). Each step is a -//! function of the frame, so a corpus row *outside* the visible rows' bounding box widens the frame -//! and rescales the visible rows' wire coordinates. An unchanged frame is sufficient for unchanged -//! keys; a widened frame is a demonstrated cause of movement rather than a guarantee of it - the -//! rescaling fixes whichever row sits at the frame's own lower corner, and finite quantization can -//! absorb a small widening for the rest. Morton cell membership reads absolute bit prefixes rather -//! than order, so where coordinates move, cells the visible rows occupy merge and split: the -//! occupied-cell curve `C(d, V)` moves with them. -//! -//! That curve is the delivery policy's own input. A view-wide cut chosen from `C(m + k, V)` is -//! therefore a function of the corpus bounding box, not of the visible set alone. -//! -//! These witnesses need two corpora, for the same reason the metadata channel does. One corpus -//! under two proofs shares one frame, so no comparison that varies the mask alone can observe this. -//! -//! Scope of these witnesses: they establish what the *key assignment* does. They assert nothing -//! about bucket assignment, which is [`super::metadata_channel`]'s subject and which an interior -//! hidden row also moves when it outranks a visible row for a contested cell. - -use hashql_core::id::IdSlice; - -use crate::{ - identity::NodeRowId, - math::{Bounds2, Vec2}, - morton::{Depth, MortonKey}, - salt::lod::{ - cascade, key, - rank::{RankInputs, Ranking}, - }, -}; - -/// The deepest cascade grid of these fixtures. -const DEEPEST: u8 = 6; - -/// The fixed wire frame every generation normalizes onto, mirroring `stage::WIRE_FRAME`. -fn wire_frame() -> Bounds2 { - Bounds2::new(Vec2::new(-1.0, -1.0), Vec2::new(1.0, 1.0)).expect("the wire frame is ordered") -} - -/// One world's reading of the first `visible` rows. -#[derive(Debug, PartialEq, Eq)] -struct Reading { - /// The visible rows' quantized keys, in row order. - keys: Vec<[u32; 2]>, - /// Occupied cells of the visible set at depths `0..=4`: `C(d, V)`. - occupied: Vec, -} - -/// Reproduces the production key assignment over `points`, then reads the first `visible` rows. -/// -/// The staged steps are `stage.rs`'s own: fit the frame over every coordinate, normalize onto the -/// wire frame, quantize. The ranking ranks row `r` at position `r`, and the cascade runs only to -/// confirm the columns agree - no expectation here reads a bucket. -fn read(points: &[Vec2], visible: usize) -> (Bounds2, Reading) { - let world = Bounds2::from_slice_par(points).expect("the fixtures are finite and non-empty"); - let wire = world.normalize_into(wire_frame(), points); - let keys = key::keys(&wire, wire_frame()); - - let importance: Vec = (0..points.len()) - .map(|row| -f32::from(u16::try_from(row).expect("fixture rows fit u16"))) - .collect(); - let priority = vec![0.0_f32; points.len()]; - let identities: Vec = - (0..u64::try_from(points.len()).expect("fixture rows fit u64")).collect(); - let inputs = RankInputs::new( - IdSlice::from_raw(&importance), - IdSlice::from_raw(&priority), - IdSlice::from_raw(&identities), - ) - .expect("the columns agree in length"); - let ranking = Ranking::new(inputs, 0); - let _buckets = cascade::buckets( - IdSlice::::from_raw(&keys), - &ranking, - depth(DEEPEST), - ); - - let occupied = (0..=4) - .map(|level| { - let mut cells: Vec = keys[..visible] - .iter() - .map(|key: &MortonKey| key.prefix(depth(level))) - .collect(); - cells.sort_unstable(); - cells.dedup(); - cells.len() - }) - .collect(); - - ( - world, - Reading { - keys: keys[..visible] - .iter() - .map(|key| key.coordinates()) - .collect(), - occupied, - }, - ) -} - -fn depth(value: u8) -> Depth { - Depth::new(value).expect("fixture depths lie within the key width") -} - -/// The visible rows every world here shares, in world coordinates. -/// -/// Their bounding box is `[0, 3]^2`, and the array lists them in rank order. -const VISIBLE: [Vec2; 3] = [ - Vec2::new(0.0, 0.0), - Vec2::new(1.0, 1.0), - Vec2::new(3.0, 3.0), -]; - -/// A hidden row outside the visible bounding box can move the visible rows' keys and `C(d, V)`. -/// -/// One witness of the channel, exact in its own numbers rather than a law over every outside row. -/// `V0` is the counterexample to the stronger reading: it sits at the fitted frame's lower corner, -/// so it keys to zero in both worlds while its neighbours move. -/// -/// Both worlds contain the same three visible rows with the same identities and the same relative -/// importance order. `read` ranks row `r` at position `r` in either world, so the comparison fixes -/// the rank inputs alongside the identities. The blocked world adds one hidden row at `(7, 7)`, -/// widening the fitted frame from `[0, 3]^2` to `[0, 7]^2`. Each axis then maps `world -> [-1, 1]` -/// with a different scale, so the visible unit coordinates change from `{0, 1/3, 1}` to `{0, 1/7, -/// 3/7}`: -/// -/// | Row | World | Unit, sparse | Key `x`, sparse | Unit, blocked | Key `x`, blocked | -/// | ---- | ------ | ------------ | --------------- | ------------- | ---------------- | -/// | `V0` | `0, 0` | `0` | `0x0000_0000` | `0` | `0x0000_0000` | -/// | `V1` | `1, 1` | `1/3` | `0x5555_5540` | `1/7` | `0x2492_4900` | -/// | `V2` | `3, 3` | `1` | `0xffff_ffff` | `3/7` | `0x6db6_db60` | -/// -/// The low bits carry the one f32 rounding the normalization admits. The cell prefixes are exact. -/// Because prefixes move, occupancy moves with them: at depth 1 the sparse world holds `V0`, `V1` -/// in one cell and `V2` in another, while the blocked world holds all three in one, so `C(1, V)` -/// reads 2 and then 1. `C(d, V)` is the cut policy's input, so the corpus bounding box is an input -/// to a scope-local delivery decision. -#[test] -fn hidden_row_outside_the_visible_frame_changes_the_visible_key_assignment() { - let (sparse_frame, sparse) = read(&VISIBLE, VISIBLE.len()); - assert_eq!( - sparse_frame, - Bounds2::new(Vec2::new(0.0, 0.0), Vec2::new(3.0, 3.0)).expect("ordered"), - "the sparse world's frame is the visible rows' own bounding box", - ); - assert_eq!( - sparse, - Reading { - keys: vec![ - [0, 0], - [0x5555_5540, 0x5555_5540], - [0xFFFF_FFFF, 0xFFFF_FFFF] - ], - occupied: vec![1, 2, 3, 3, 3], - }, - ); - - let mut blocked_points = VISIBLE.to_vec(); - blocked_points.push(Vec2::new(7.0, 7.0)); - let (blocked_frame, blocked) = read(&blocked_points, VISIBLE.len()); - assert_eq!( - blocked_frame, - Bounds2::new(Vec2::new(0.0, 0.0), Vec2::new(7.0, 7.0)).expect("ordered"), - "the hidden row widened the fitted frame", - ); - assert_eq!( - blocked, - Reading { - keys: vec![ - [0, 0], - [0x2492_4900, 0x2492_4900], - [0x6DB6_DB60, 0x6DB6_DB60] - ], - occupied: vec![1, 1, 2, 3, 3], - }, - ); - - assert_ne!( - sparse, blocked, - "the hidden row moved the visible key assignment" - ); -} - -/// A hidden row inside the visible bounding box leaves the visible key assignment fixed. -/// -/// This is the other half of the condition, and it is what makes the fixed-view premise -/// constructible: the frame is the tight box over every corpus coordinate, so a hidden row at -/// `(1.5, 1.5)` - interior to `[0, 3]^2` - leaves the frame, the visible keys, and `C(d, V)` -/// exactly as the sparse world reads them. -/// -/// It asserts nothing about buckets. An interior hidden row still moves the bucket assignment when -/// it outranks a visible row for a contested cell, which is [`super::metadata_channel`]'s witness. -/// Asserting bucket equality here would generalize one lucky instance into a false rule. -#[test] -fn interior_hidden_row_leaves_the_visible_key_assignment_fixed() { - let (sparse_frame, sparse) = read(&VISIBLE, VISIBLE.len()); - - let mut interior_points = VISIBLE.to_vec(); - interior_points.push(Vec2::new(1.5, 1.5)); - let (interior_frame, interior) = read(&interior_points, VISIBLE.len()); - - assert_eq!( - interior_frame, sparse_frame, - "an interior row cannot widen the tight bounding box", - ); - assert_eq!( - interior, sparse, - "the visible keys and C(d, V) are invariant to interior hidden rows", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/masking.rs b/libs/@local/graph/atlas/src/serve/tests/masking.rs deleted file mode 100644 index 9fc603ea878..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/masking.rs +++ /dev/null @@ -1,1208 +0,0 @@ -//! The visibility masking battery, covering every read path computed over the masked view. - -use core::assert_matches; - -use rand::{RngExt as _, SeedableRng as _}; - -use super::{ - Artifacts, Atlas, BasePosition, Bound, CutOffset, Depth, EDGE_SEED, EdgesLimits, FIXTURE_EDGES, - FIXTURE_LOD, FULL, Generation, HEAD, HashMap, HashSet, Id as _, IdSlice, Mode, MortonCell, - NodeRowId, POSITIONS, ROW_IDS, ServeLimits, TileHead, TileLimits, TileQuery, TileRequest, - TileResponse, UntouchedStore, VisibilityProof, Xoshiro256PlusPlus, codec, coordinate_of, - decode_rows, domain_mask, edges_request, entity_string_of, expected_edges_bytes, - extremes_vacating_a_root_cell, fixture_row_ids, full_grid, head_global, locate_request, - mask_hiding, mask_hiding_rows, narrow_usize, open_artifacts, publish, qualifying_columns, - request, section, test_codec, viewing, walk, wire_columns, -}; -use crate::{ - math::Bounds2, - serve::{delta::PlacementCohort, visibility::ResolvedRow}, -}; - -/// Resolution collapses every failure to one [`None`]. -/// -/// Under the full proof every in-universe wire id resolves to its row, and under a mask the hidden -/// row's wire id answers exactly the [`None`] an out-of-universe value answers, so forbidden and -/// nonexistent are indistinguishable downstream of the resolution. -#[tokio::test] -async fn resolve_collapses_every_failure_to_one_none() { - let (_generation, atlas) = publish("resolve-seam").await; - let node_codec = test_codec(&atlas); - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - - // The empty cohort holds no slot, so every resolution here answers in the fitted domain. - let fitted = - |proof: &VisibilityProof, wire| match atlas.resolve(proof, PlacementCohort::EMPTY, wire) { - Some(ResolvedRow::Fitted(row)) => Some(row.get()), - Some(ResolvedRow::Arrival(identity)) => { - panic!("an empty cohort resolved the arrival {identity:?}") - } - None => None, - }; - - let masked = mask_hiding(&atlas, &[7]); - for row in 0..universe { - let wire = node_codec.encode(NodeRowId::from_u32(row), atlas.node_universe()); - assert_eq!(fitted(&FULL, wire), Some(NodeRowId::from_u32(row))); - assert_eq!( - fitted(&masked, wire), - (row != 7).then(|| NodeRowId::from_u32(row)), - ); - } - assert!(fitted(&FULL, codec::WireRow::pinned(universe)).is_none()); - assert!(fitted(&masked, codec::WireRow::pinned(universe)).is_none()); -} - -/// The masked root serves the scope cascade's rows, in both modes. -/// -/// The delivered rows and their positions column equal the reference over exactly the visible -/// rows - a visible row may sit shallower than the corpus schedule placed it, because its -/// hidden competitor is out of its view. -#[tokio::test] -async fn masked_root_serves_the_scope_cascades_rows_in_both_modes() { - let (_, atlas) = publish("masked-tile").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let node_codec = test_codec(&atlas); - - // Hide every third row; the hidden set crosses every bucket of - // the 48-point fixture. - let hidden: Vec = (0..universe).filter(|row| row.is_multiple_of(3)).collect(); - let proof = mask_hiding(&atlas, &hidden); - - let schedule = super::schedule::reference::Schedule::new( - super::schedule::reference::rows(&atlas, &proof), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - 0, - ); - let root_cell = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - let position_of: HashMap = atlas - .row_ids() - .iter() - .enumerate() - .map(|(position, &row)| (row, u32::try_from(position).expect("positions fit u32"))) - .collect(); - let wire_points = atlas.positions(); - - for mode in [Mode::Delta, Mode::Total] { - let masked_bytes = atlas - .tile( - &request(0, 0, 0, mode), - TileLimits::default(), - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - ) - .expect("the masked root serves"); - - let masked_rows = decode_rows(section(&masked_bytes, ROW_IDS).expect("ROW_IDS is present")); - let delivered: Vec = masked_rows - .iter() - .map(|&wire| { - let row = node_codec - .decode(codec::WireRow::pinned(wire), atlas.node_universe()) - .expect("delivered wire ids decode"); - position_of[&row] - }) - .collect(); - let expected = schedule.delivery(0, root_cell, mode); - assert_eq!( - delivered, expected.positions, - "the {mode:?} root delivers the scope cascade's rows" - ); - - // The positions column carries each delivered row's own wire coordinates. - let masked_positions = section(&masked_bytes, POSITIONS).expect("POSITIONS is present"); - let expected_positions: Vec = expected - .positions - .iter() - .flat_map(|&position| { - let point = wire_points[BasePosition::from_u32(position)]; - let mut bytes = point.x().to_le_bytes().to_vec(); - bytes.extend_from_slice(&point.y().to_le_bytes()); - bytes - }) - .collect(); - assert_eq!(masked_positions, expected_positions); - } -} - -/// A fully masked populated tile answers byte-identically to a tile that never had rows. -/// -/// Empty is empty: the runs and children bits match a never-populated cell's, and the head's -/// occupancy fields carry no evidence of hidden points. -#[tokio::test] -async fn fully_masked_tile_vs_empty_tile() { - let (generation, atlas) = publish("masked-tile-empty").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - - let Artifacts { - quad, coordinates, .. - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let root_cell = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - let mut nodes = Vec::new(); - walk(&quad, 0, root_cell, &mut nodes); - let (_, populated_cell) = nodes[1..] - .iter() - .copied() - .find(|&(node, _)| { - let run = quad.nodes()[node as usize].run(); - run.end > run.start - }) - .expect("the fixture quadtree has a populated non-root node"); - let coordinate = coordinate_of(populated_cell); - let nothing = mask_hiding(&atlas, &(0..universe).collect::>()); - let expected = TileResponse { - head: TileHead { - generation: generation.id().digest(), - variant: 0, - coordinate, - mode: Mode::Delta, - first_bucket: coordinate.z + FIXTURE_LOD.span.get(), - runs: &[0], - global: None, - children: 0, - }, - delivered: crate::salt::wire::tile::DeliveredSet::Ranges(&[]), - positions: IdSlice::from_raw(points), - rows: IdSlice::from_raw(&[]), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - } - .encode(); - assert_eq!( - atlas - .tile( - &TileRequest { - coordinate, - query: TileQuery::default(), - }, - TileLimits::default(), - Bound::new(&atlas, ¬hing, CutOffset::ZERO).view(&atlas), - ) - .expect("the fully masked tile serves"), - expected, - "a fully masked tile is a tile that never had rows", - ); -} - -/// The edges path inherits the mask through its endpoints. -/// -/// Hiding one node removes exactly the edges incident to it - the delivered sets intersect the -/// proof before edges qualify, so the response is byte-identical to the qualifying computation over -/// the visible row set. -#[tokio::test] -async fn masked_edges_inherit_endpoint_visibility() { - let (generation, atlas) = publish("masked-edges").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - - // Row 5 is an endpoint of the reciprocal fixture pair (edge rows - // 3 and 4). Hiding it must remove exactly those two edges. - let hidden = 5_u32; - let proof = mask_hiding(&atlas, &[hidden]); - let endpoints: Vec<[u64; 2]> = FIXTURE_EDGES - .iter() - .map(|&(_, source, target)| [source, target]) - .collect(); - let delivered: HashSet = (0..universe).filter(|&row| row != hidden).collect(); - let (sources, targets, rows) = qualifying_columns(&endpoints, &delivered); - assert_eq!( - rows.len(), - FIXTURE_EDGES.len() - 2, - "two edges hide with row 5" - ); - let columns = wire_columns(&atlas, &sources, &targets, &rows); - - assert_eq!( - atlas - .edges( - &edges_request(full_grid()), - EdgesLimits::default(), - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the masked grid serves"), - expected_edges_bytes(&generation, true, &columns), - ); -} - -/// Translate answers missing for denied, in both identity domains. -/// -/// A hidden node's id is an absent key exactly like a nonexistent id. An edge is absent when either -/// endpoint hides (edge visibility derives) and present while both endpoints show. -#[tokio::test] -async fn translate_answers_missing_for_denied() { - use crate::serve::translate::{TranslateLimits, TranslateRequest}; - - let (_generation, atlas) = publish("masked-translate").await; - - // Row 5 endpoints fixture edge rows 3 and 4; edge row 0 joins - // rows 0 and 1, untouched by the mask. - let proof = mask_hiding(&atlas, &[5]); - let request = TranslateRequest { - entity_ids: vec![ - entity_string_of(5), // hidden node: absent - entity_string_of(6), // visible node: present - entity_string_of(EDGE_SEED + 3), // edge with hidden endpoint: absent - entity_string_of(EDGE_SEED), // edge with visible endpoints: present - ], - }; - let masked = atlas - .translate( - request.clone(), - TranslateLimits::default(), - &proof, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - assert!(!masked.nodes.contains_key(&entity_string_of(5))); - assert!(masked.nodes.contains_key(&entity_string_of(6))); - assert!(!masked.edges.contains_key(&entity_string_of(EDGE_SEED + 3))); - assert!(masked.edges.contains_key(&entity_string_of(EDGE_SEED))); - - // The full proof answers all four, so the absences above come from - // the mask rather than from the identity tables. - let full = atlas - .translate( - request, - TranslateLimits::default(), - &FULL, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - assert_eq!(full.nodes.len(), 2); - assert_eq!(full.edges.len(), 2); -} - -/// Locate filters partners under the mask and hides its source like a missing one. -/// -/// A hidden source answers the same `UnknownEntity` in both ingress domains; a hidden partner -/// drops with its edges BEFORE the cap selects - `complete` stays `true`, so the response never -/// discloses that the mask withheld anything. -#[tokio::test] -async fn locate_filters_partners_under_the_mask() { - let (_generation, atlas) = publish("masked-locate").await; - let limits = ServeLimits::default(); - let full_bound = Bound::of(&atlas, &FULL); - let full_view = full_bound.view(&atlas); - - // Ground truth: ego(5) = partner 40 over the reciprocal pair, - // edge rows 3 and 4. - let source = atlas - .resolve_source(&full_view, &entity_string_of(5)) - .expect("row 5 resolves"); - let full = atlas.locate_subgraph(source, limits.locate, &full_view); - assert_eq!( - super::delivered_row_ids(&atlas, &full), - [NodeRowId::new(5), NodeRowId::new(40)], - "the source and its one partner" - ); - assert_eq!(full.edges.len(), 2); - - // Hiding the partner removes it and both its edges: the source stands alone, complete - a - // masked ego-graph answers exactly like one where the partner never existed. - let proof = mask_hiding(&atlas, &[40]); - let masked_bound = Bound::of(&atlas, &proof); - let masked_view = masked_bound.view(&atlas); - let masked = atlas.locate_subgraph(source, limits.locate, &masked_view); - assert_eq!( - super::delivered_row_ids(&atlas, &masked), - [NodeRowId::new(5)], - "the hidden partner is not delivered" - ); - assert!(masked.edges.is_empty(), "its edges leave with it"); - assert!(masked.complete, "visibility is not truncation"); - - // Hidden partners drop BEFORE selection: under a cap of one, the - // masked response still answers from visible edges alone. - let capped = crate::serve::locate::LocateLimits { - edges: 1, - ..limits.locate - }; - let capped_masked = atlas.locate_subgraph(source, capped, &masked_view); - assert!(capped_masked.complete, "zero visible edges fit any cap"); - assert!(capped_masked.edges.is_empty()); - - // A hidden source is a missing source, in both ingress domains. - let hidden_source = mask_hiding(&atlas, &[0]); - assert_matches!( - atlas.locate( - &locate_request(entity_string_of(0)), - limits, - Bound::new(&atlas, &hidden_source, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(crate::serve::LocateError::UnknownEntity), - "the hidden source rejects", - ); - let node_codec = test_codec(&atlas); - let by_row = crate::serve::LocateRequest { - entity_id: None, - row: Some(node_codec.encode(NodeRowId::new(0), atlas.node_universe())), - colored_type_ids: Vec::new(), - }; - - assert_matches!( - atlas.locate( - &by_row, - limits, - Bound::new(&atlas, &hidden_source, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(crate::serve::LocateError::UnknownEntity), - "the hidden source rejects by row too", - ); -} - -/// The edges grid withholds a hidden link row though both its endpoints are visible. -/// -/// An edge carries its link entity's own authorization, which its endpoints do not imply. The -/// fixture pins the distinction exactly: edge rows 3 and 4 are the reciprocal pair over the same -/// endpoint pair, so hiding row 3 alone must leave an edge over those endpoints delivered. No -/// rule stated over endpoints can answer this response - it must drop both or neither - and a rule -/// that over-drops loses row 4 with it. -#[tokio::test] -async fn hidden_link_row_is_withheld_from_the_edges_grid() { - let (generation, atlas) = publish("masked-link-edges").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let hidden_link = 3_u32; - let proof = mask_hiding_rows(&atlas, &[], &[hidden_link]); - - let endpoints: Vec<[u64; 2]> = FIXTURE_EDGES - .iter() - .map(|&(_, source, target)| [source, target]) - .collect(); - let visible: HashSet = (0..universe).collect(); - let (sources, targets, rows) = qualifying_columns(&endpoints, &visible); - assert_eq!( - rows.len(), - FIXTURE_EDGES.len(), - "every node is visible, so every edge qualifies on endpoints" - ); - - // The expectation is the qualifying computation over the visible - // LINK set: every edge but row 3, its reciprocal included. - let mut kept_sources = Vec::new(); - let mut kept_targets = Vec::new(); - let mut kept_rows = Vec::new(); - for ((&row, &source), &target) in rows.iter().zip(&sources).zip(&targets) { - if row != hidden_link { - kept_sources.push(source); - kept_targets.push(target); - kept_rows.push(row); - } - } - assert!( - kept_rows.contains(&4), - "the reciprocal edge over the same endpoints stays delivered" - ); - let columns = wire_columns(&atlas, &kept_sources, &kept_targets, &kept_rows); - - assert_eq!( - atlas - .edges( - &edges_request(full_grid()), - EdgesLimits::default(), - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the masked grid serves"), - expected_edges_bytes(&generation, true, &columns), - ); -} - -/// A hidden link row leaves the locate ego-graph's partner delivered by its other edge. -/// -/// `ego(5)` is the reciprocal pair: partner 40 over edge rows 3 and 4. Hiding link row 3 withholds -/// one direction while row 4 still delivers the partner, so the response keeps both rows and -/// exactly one edge - a shape an endpoint-level rule cannot produce. `complete` stays true: -/// visibility is not truncation. -#[tokio::test] -async fn hidden_link_row_leaves_the_locate_partner_delivered() { - let (_generation, atlas) = publish("masked-link-locate").await; - let limits = ServeLimits::default(); - let full_bound = Bound::of(&atlas, &FULL); - let full_view = full_bound.view(&atlas); - let source = atlas - .resolve_source(&full_view, &entity_string_of(5)) - .expect("row 5 resolves"); - - // As a control, both directions deliver under full visibility. - let full = atlas.locate_subgraph(source, limits.locate, &full_view); - assert_eq!(full.edges.len(), 2, "the reciprocal pair, both directions"); - - let proof = mask_hiding_rows(&atlas, &[], &[3]); - let masked = viewing(&atlas, &proof, |view| { - atlas.locate_subgraph(source, limits.locate, view) - }); - assert_eq!( - super::delivered_row_ids(&atlas, &masked), - [NodeRowId::new(5), NodeRowId::new(40)], - "the partner stays: its other edge still delivers it" - ); - assert_eq!( - masked - .edges - .iter() - .map(|&(edge, _)| narrow_usize(super::fitted(edge).row.get().as_usize())) - .collect::>(), - [4], - "exactly the withheld link row leaves" - ); - assert!(masked.complete, "visibility is not truncation"); -} - -/// A hidden link row is an absent key in translate, beside a delivered edge on the same endpoints. -/// -/// Hidden and nonexistent are the same answer at this ingress, and the reciprocal link row proves -/// the absence is the link row's own: the two edges name the same two entities, and only the hidden -/// one is missing from the response. -#[tokio::test] -async fn hidden_link_row_is_an_absent_key_in_translate() { - use crate::serve::translate::{TranslateLimits, TranslateRequest}; - - let (_generation, atlas) = publish("masked-link-translate").await; - let hidden = entity_string_of(EDGE_SEED + 3); - let reciprocal = entity_string_of(EDGE_SEED + 4); - let endpoint = entity_string_of(5); - let ids = vec![hidden.clone(), reciprocal.clone(), endpoint.clone()]; - - let translate = |proof: &VisibilityProof| { - atlas - .translate( - TranslateRequest { - entity_ids: ids.clone(), - }, - TranslateLimits::default(), - proof, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap") - }; - - // As a control, under full visibility both link ids resolve, so the - // absence below is the link mask's. - let control = translate(&FULL); - assert!(control.edges.contains_key(&hidden)); - assert!(control.edges.contains_key(&reciprocal)); - - let masked = translate(&mask_hiding_rows(&atlas, &[], &[3])); - assert!( - !masked.edges.contains_key(&hidden), - "the hidden link row is an absent key" - ); - assert!( - masked.edges.contains_key(&reciprocal), - "the same endpoints still deliver their other edge" - ); - assert!( - masked.nodes.contains_key(&endpoint), - "the endpoints themselves stay visible" - ); -} - -/// The proof's membership algebra is fail-closed at every boundary. -/// -/// Rows beyond a mask's domain read hidden in both domains, an edge delivers only with its own link -/// row and both endpoints, and the intersection removes exactly the hidden rows. -#[test] -fn visibility_proof_is_fail_closed() { - use crate::{identity::EdgeRowId, serve::visibility::VisibleEdge}; - - // Node rows 1 and 2 visible of four; link rows 0 and 2 visible of - // three. - let proof = VisibilityProof::from_masks( - domain_mask(4, &[0, 3]), - domain_mask(3, &[1]), - hashql_core::collections::fast_hash_set(), - ); - - assert!(!proof.contains(NodeRowId::new(0))); - assert!(proof.contains(NodeRowId::new(1))); - assert!(!proof.contains(NodeRowId::new(3))); - // Beyond the mask's domain: hidden, never a panic. - assert!(!proof.contains(NodeRowId::new(4))); - assert!(!proof.contains(NodeRowId::from_u32(u32::MAX))); - - let visible_endpoints = [NodeRowId::new(1), NodeRowId::new(2)]; - let [source, target] = visible_endpoints; - assert_eq!( - proof - .verify_edge(EdgeRowId::new(0), source, target) - .map(VisibleEdge::get), - Some(EdgeRowId::new(0)), - "a visible link row over visible endpoints delivers, and the witness names it" - ); - assert!( - proof - .verify_edge(EdgeRowId::new(1), source, target) - .is_none(), - "a hidden link row withholds its edge over visible endpoints" - ); - assert!( - proof - .verify_edge(EdgeRowId::new(0), source, NodeRowId::new(3)) - .is_none(), - "a hidden target withholds a visible link row" - ); - assert!( - proof - .verify_edge(EdgeRowId::new(0), NodeRowId::new(0), target) - .is_none(), - "a hidden source withholds a visible link row" - ); - assert!( - proof - .verify_edge(EdgeRowId::new(3), source, target) - .is_none(), - "a link row beyond the mask's domain is hidden, never a panic" - ); - - let mut set = hashql_core::id::bit_vec::DenseBitSet::new_filled(6); - proof.intersect(&mut set); - assert_eq!( - set.iter().collect::>(), - [1, 2].map(NodeRowId::new), - "the intersection removes exactly the hidden rows" - ); - - assert_eq!(proof.visible_below(4), 2); - assert_eq!(FULL.visible_below(48), 48); - assert!(FULL.contains(NodeRowId::from_u32(u32::MAX))); - - // The delta-link arm is fail-closed on its identity domain: a masked proof admits exactly - // its captured set, and the full proof admits every identity. - let admitted = crate::postgres::id::ArchivedEntityId { - web_id: uuid::Uuid::from_bytes([0xC0; 16]).into(), - entity_uuid: crate::postgres::id::ArchivedEntityUuid::from_bytes([0x3F; 16]), - }; - let stranger = crate::postgres::id::ArchivedEntityId { - web_id: uuid::Uuid::from_bytes([0xC1; 16]).into(), - entity_uuid: crate::postgres::id::ArchivedEntityUuid::from_bytes([0x3E; 16]), - }; - let capturing = VisibilityProof::from_masks( - domain_mask(4, &[]), - domain_mask(3, &[]), - core::iter::once(admitted).collect(), - ); - assert!(capturing.admits_delta_link(admitted)); - assert!( - !capturing.admits_delta_link(stranger), - "an identity outside the captured set never serves" - ); - assert!(FULL.admits_delta_link(stranger)); -} - -/// Masking commutes with delivery on every endpoint. -/// -/// Exactness is per endpoint, because each one derives its rows differently. Tiles deliver the -/// scope cascade over exactly the visible rows, so a visible row may claim a shallower cell than it -/// held under the corpus schedule. Edges, translate, and locate equal the unmasked response with -/// the hidden rows' entries removed, so the mask never leaks and never over-drops. The fixture -/// serves without capacity pressure on the non-tile endpoints, so their filtered-full comparison is -/// the law verbatim. Locate's ground truth is the fixture edge list itself, the visible ego-graph -/// derived edge by edge. -#[tokio::test] -async fn composition_law_holds_under_random_masks() { - let (generation, atlas) = publish("composition-sweep").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let mut rng = Xoshiro256PlusPlus::seed_from_u64(0x51CA); - - for _ in 0..8 { - let hidden: Vec = (0..universe).filter(|_| rng.random_ratio(1, 4)).collect(); - let proof = mask_hiding(&atlas, &hidden); - - assert_tiles_mask_by_intersection(&atlas, &proof, &hidden); - assert_edges_mask_by_intersection(&generation, &atlas, &proof, &hidden); - assert_translate_masks_by_visibility(&atlas, &proof, &hidden); - assert_locate_delivers_the_visible_ego_graph(&atlas, &proof, &hidden); - } -} - -/// The masked rows are the scope cascade's rows at every tile coordinate in both modes. -/// -/// The delivered set, its order, and the run recounts equal the independent reference over -/// exactly the visible rows, and no hidden row shows up - the intersection with the corpus -/// schedule is not the law, because a visible row may claim a shallower cell once its hidden -/// competitor is out of its view. -#[track_caller] -fn assert_tiles_mask_by_intersection(atlas: &Atlas, proof: &VisibilityProof, hidden: &[u32]) { - let node_codec = test_codec(atlas); - let hidden_wire: HashSet = hidden - .iter() - .map(|&row| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }) - .collect(); - let position_of: HashMap = atlas - .row_ids() - .iter() - .enumerate() - .map(|(position, &row)| (row, u32::try_from(position).expect("positions fit u32"))) - .collect(); - - let schedule = super::schedule::reference::Schedule::new( - super::schedule::reference::rows(atlas, proof), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - 0, - ); - - for z in 0..=FIXTURE_LOD.max_tile_depth { - let cells = 1_u32 << z; - for (x, y) in (0..cells).flat_map(|x| (0..cells).map(move |y| (x, y))) { - let cell = MortonCell::new(Depth::new(z).expect("zooms are depths"), x, y) - .expect("the sweep stays on each zoom's grid"); - - for mode in [Mode::Delta, Mode::Total] { - let at = format!("the {mode:?} tile {z}/{x}/{y}"); - let masked_bytes = atlas - .tile( - &request(z, x, y, mode), - TileLimits::default(), - Bound::new(atlas, proof, CutOffset::ZERO).view(atlas), - ) - .expect("the masked tile serves"); - let masked_rows = - decode_rows(section(&masked_bytes, ROW_IDS).expect("ROW_IDS is present")); - - for wire in &masked_rows { - assert!(!hidden_wire.contains(wire), "{at} keeps hidden rows hidden"); - } - - let positions: Vec = masked_rows - .iter() - .map(|&wire| { - let row = node_codec - .decode(codec::WireRow::pinned(wire), atlas.node_universe()) - .expect("delivered wire ids decode"); - position_of[&row] - }) - .collect(); - assert_eq!( - positions, - schedule.delivery(z, cell, mode).positions, - "{at} delivers the scope cascade's rows in order", - ); - } - } - } -} - -/// The masked grid answers the qualifying computation over the visible row set, byte for byte. -#[track_caller] -fn assert_edges_mask_by_intersection( - generation: &Generation, - atlas: &Atlas, - proof: &VisibilityProof, - hidden: &[u32], -) { - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let endpoints: Vec<[u64; 2]> = FIXTURE_EDGES - .iter() - .map(|&(_, source, target)| [source, target]) - .collect(); - let delivered: HashSet = (0..universe).filter(|row| !hidden.contains(row)).collect(); - let (sources, targets, rows) = qualifying_columns(&endpoints, &delivered); - let columns = wire_columns(atlas, &sources, &targets, &rows); - assert_eq!( - atlas - .edges( - &edges_request(full_grid()), - EdgesLimits::default(), - Bound::new(atlas, proof, CutOffset::ZERO).view(atlas), - UntouchedStore, - ) - .expect("the masked grid serves"), - expected_edges_bytes(generation, true, &columns), - ); -} - -/// Every fixture identity translates exactly when visible (nodes) or when both endpoints are -/// (edges). -#[track_caller] -fn assert_translate_masks_by_visibility(atlas: &Atlas, proof: &VisibilityProof, hidden: &[u32]) { - use crate::serve::translate::{TranslateLimits, TranslateRequest}; - - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let every_identity: Vec = (0..universe) - .map(|row| entity_string_of(u8::try_from(row).expect("fixture rows fit u8"))) - .chain((0..FIXTURE_EDGES.len()).map(|row| { - entity_string_of(EDGE_SEED + u8::try_from(row).expect("fixture edge rows fit u8")) - })) - .collect(); - let translated = atlas - .translate( - TranslateRequest { - entity_ids: every_identity, - }, - TranslateLimits::default(), - proof, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - for row in 0..universe { - let id = entity_string_of(u8::try_from(row).expect("fixture rows fit u8")); - assert_eq!( - translated.nodes.contains_key(&id), - !hidden.contains(&row), - "node {row} translates exactly when visible" - ); - } - for (row, &(_, source, target)) in FIXTURE_EDGES.iter().enumerate() { - let id = entity_string_of(EDGE_SEED + u8::try_from(row).expect("edge rows fit u8")); - let visible = !hidden.contains(&u32::try_from(source).expect("fixture rows fit u32")) - && !hidden.contains(&u32::try_from(target).expect("fixture rows fit u32")); - assert_eq!( - translated.edges.contains_key(&id), - visible, - "edge {row} translates exactly when both endpoints show" - ); - } -} - -/// Every visible source's masked ego-graph is the fixture edge list filtered to visible partners. -/// -/// Edges ascend by link-entity identity bytes (for the fixture, edge row), partners derive from -/// the delivered edges ascending wire row id, and `complete` stays `true` because visibility is not -/// truncation. Wherever the mask shrinks a source's incident set, a second probe caps the query at -/// exactly the visible cardinality. Hidden partners drop before selection, so the tight cap -/// truncates nothing and delivers the whole visible set, complete and independent of the truncation -/// key. Selecting first and masking after would come up short in exactly these configurations. -#[track_caller] -fn assert_locate_delivers_the_visible_ego_graph( - atlas: &Atlas, - proof: &VisibilityProof, - hidden: &[u32], -) { - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let limits = ServeLimits::default(); - let node_codec = test_codec(atlas); - let wire_of = |row: u32| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }; - let bound = Bound::of(atlas, proof); - let view = bound.view(atlas); - - for source_row in (0..universe).filter(|row| !hidden.contains(row)) { - let source_id = entity_string_of(u8::try_from(source_row).expect("fixture rows fit u8")); - let masked = atlas.locate_subgraph( - atlas - .resolve_source(&view, &source_id) - .expect("a visible source resolves under the mask"), - limits.locate, - &view, - ); - assert!(masked.complete, "visibility is not truncation"); - - // Ground truth off the fixture edge list, one entry per edge - // incident to the source whose partner is visible. - let mut expected_edges: Vec = FIXTURE_EDGES - .iter() - .enumerate() - .filter(|&(_, &(_, edge_source, edge_target))| { - let incident = - edge_source == u64::from(source_row) || edge_target == u64::from(source_row); - let partner = if edge_source == u64::from(source_row) { - edge_target - } else { - edge_source - }; - incident && !hidden.contains(&u32::try_from(partner).expect("fixture rows fit u32")) - }) - .map(|(row, _)| narrow_usize(row)) - .collect(); - expected_edges.sort_unstable(); - let delivered: Vec = masked - .edges - .iter() - .map(|&(edge, _)| narrow_usize(super::fitted(edge).row.get().as_usize())) - .collect(); - assert_eq!(delivered, expected_edges, "ego({source_row}) edges"); - - let mut partner_keys: Vec<(u32, u32)> = expected_edges - .iter() - .flat_map(|&row| { - let (_, edge_source, edge_target) = FIXTURE_EDGES[row as usize]; - [ - u32::try_from(edge_source).expect("fixture rows fit u32"), - u32::try_from(edge_target).expect("fixture rows fit u32"), - ] - }) - .filter(|&row| row != source_row) - .map(|row| (wire_of(row), row)) - .collect(); - partner_keys.sort_unstable(); - partner_keys.dedup(); - let mut expected_rows = vec![source_row]; - expected_rows.extend(partner_keys.iter().map(|&(_, row)| row)); - let delivered_rows: Vec = super::delivered_row_ids(atlas, &masked) - .iter() - .map(|row| narrow_usize(row.as_usize())) - .collect(); - assert_eq!(delivered_rows, expected_rows, "ego({source_row}) rows"); - for row in &delivered_rows { - assert!(!hidden.contains(row), "every delivered row is visible"); - } - - // Drop-before-cap, key-independent: whenever the mask shrank - // this source's incident set, a cap of exactly the visible - // cardinality still delivers every visible edge. - let incident = FIXTURE_EDGES - .iter() - .filter(|&&(_, edge_source, edge_target)| { - edge_source == u64::from(source_row) || edge_target == u64::from(source_row) - }) - .count(); - if !expected_edges.is_empty() && expected_edges.len() < incident { - let tight = crate::serve::locate::LocateLimits { - edges: u32::try_from(expected_edges.len()).expect("the fixture edge count fits"), - ..limits.locate - }; - let capped = atlas.locate_subgraph( - atlas - .resolve_source(&view, &source_id) - .expect("a visible source resolves under the mask"), - tight, - &view, - ); - assert!( - capped.complete, - "a cap at the visible cardinality truncates nothing" - ); - let capped_edges: Vec = capped - .edges - .iter() - .map(|&(edge, _)| narrow_usize(super::fitted(edge).row.get().as_usize())) - .collect(); - assert_eq!( - capped_edges, expected_edges, - "ego({source_row}) under the tight cap" - ); - } - } -} - -/// Hidden and nonexistent answer identically at every id-bearing ingress, under any mask. -/// -/// The sweep drives eight seeded random proofs over every hidden row, through each of the three -/// ingresses that accept an identifier. Those are locate by entity id, locate by wire row id, and -/// translate. The case compares each denied request with the same request naming something that -/// never existed - an unknown entity seed, a wire value outside the codec's image - and the answers -/// are equal values at the resolution. The renderers downstream are deterministic functions of -/// those values, so equal values are equal response bytes: the collapse law, swept rather than -/// sampled. -#[tokio::test] -async fn hidden_and_nonexistent_collapse_at_every_id_bearing_ingress() { - use crate::serve::translate::{TranslateLimits, TranslateRequest}; - - let (_generation, atlas) = publish("p8-collapse").await; - let universe = u32::try_from(atlas.row_ids().len()).expect("the fixture universe fits u32"); - let node_codec = test_codec(&atlas); - let limits = ServeLimits::default(); - - // Identifiers that never existed: an entity seed no fixture row - // or edge carries, and the first wire value outside the image. - let ghost_id = entity_string_of(203); - let ghost_wire = (0..=u32::MAX) - .find(|&wire| { - atlas - .resolve(&FULL, PlacementCohort::EMPTY, codec::WireRow::pinned(wire)) - .is_none() - }) - .expect("the image has forty-eight values; almost everything is outside it"); - let by_row = |wire: u32| crate::serve::LocateRequest { - entity_id: None, - row: Some(codec::WireRow::pinned(wire)), - colored_type_ids: Vec::new(), - }; - let mut rng = Xoshiro256PlusPlus::seed_from_u64(0x9A08); - - for _ in 0..8 { - let hidden: Vec = (0..universe).filter(|_| rng.random_ratio(1, 4)).collect(); - assert!(!hidden.is_empty(), "the seeded masks hide at least one row"); - let proof = mask_hiding(&atlas, &hidden); - - // The nonexistent baselines, once per proof: both ingress - // domains answer unknown-entity for ids that never existed. - assert_matches!( - atlas.locate( - &locate_request(ghost_id.clone()), - limits, - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(crate::serve::LocateError::UnknownEntity), - "an unknown entity rejects", - ); - assert_matches!( - atlas.locate( - &by_row(ghost_wire), - limits, - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(crate::serve::LocateError::UnknownEntity), - "an out-of-image wire value rejects", - ); - let missing_translated = atlas - .translate( - TranslateRequest { - entity_ids: vec![ghost_id.clone()], - }, - TranslateLimits::default(), - &proof, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - - for &row in &hidden { - let id = entity_string_of(u8::try_from(row).expect("fixture rows fit u8")); - - // Denied and missing are one error: a hidden source - // answers exactly the variant the ghost baselines did, - // in both ingress domains. - assert_matches!( - atlas.locate( - &locate_request(id.clone()), - limits, - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(crate::serve::LocateError::UnknownEntity), - "a hidden source rejects", - ); - assert_matches!( - atlas.locate( - &by_row( - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get(), - ), - limits, - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(crate::serve::LocateError::UnknownEntity), - "the row ingress collapses the same way", - ); - - let denied = atlas - .translate( - TranslateRequest { - entity_ids: vec![id], - }, - TranslateLimits::default(), - &proof, - None, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - assert_eq!( - denied, missing_translated, - "a denied id translates exactly like one that never existed" - ); - } - } -} - -/// The masked root publishes the visible view's own census, not the generation's. -/// -/// The root tile's global map carries three corpus-wide aggregates, and each one resolves once -/// per scope rather than per request. This pins all three against an independent derivation over -/// the generation's own columns, under a mask chosen so that a census read off the artifacts -/// instead of off the view fails on every one of them: -/// -/// The hidden set is the rows attaining an extreme coordinate, so the visible extent is strictly -/// inside the generation's extent on all four edges, plus the rest of one root cell's rows, so the -/// root cut delivers strictly fewer of them - the fixture asserts both strictnesses rather than -/// assuming them, because a mask that vacates no edge and empties no cell could not fail on the -/// defect. -#[tokio::test] -async fn masked_root_publishes_the_visible_views_own_census() { - let (generation, atlas) = publish("masked-census").await; - let Artifacts { - coordinates, rows, .. - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - let (corpus, hidden) = extremes_vacating_a_root_cell(&atlas, points, &row_ids); - assert!( - !hidden.is_empty() && hidden.len() < points.len(), - "the mask hides the extremes and leaves a non-empty view" - ); - - let proof = mask_hiding(&atlas, &hidden); - let visible = |position: usize| !hidden.contains(&row_ids[position]); - - // The expectations come from the columns rather than the serve path, and they are the rows of - // the view's own cascade at or below the root cut, the tight extent of the whole visible set, - // and the deepest occupied scope bucket. - let (expected_visible, expected_deepest) = super::schedule::reference::Schedule::new( - super::schedule::reference::rows(&atlas, &proof), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - 0, - ) - .global(); - let expected_extent = Bounds2::from_points( - (0..points.len()) - .filter(|&position| visible(position)) - .map(|position| points[position]), - ) - .expect("the masked view holds points"); - - let edges = |bounds: &Bounds2| { - [ - bounds.min().x(), - bounds.min().y(), - bounds.max().x(), - bounds.max().y(), - ] - }; - - // The witness must be able to fail on the defect it names: every edge of the visible extent - // moved inward, so publishing the generation's extent here is a detectable answer. - assert!( - expected_extent.min().x() > corpus.min().x() - && expected_extent.min().y() > corpus.min().y() - && expected_extent.max().x() < corpus.max().x() - && expected_extent.max().y() < corpus.max().y(), - "the mask vacates all four extremes, so the view's extent is strictly inside the corpus's" - ); - - let masked_bytes = atlas - .tile( - &request(0, 0, 0, Mode::Delta), - TileLimits::default(), - Bound::new(&atlas, &proof, CutOffset::ZERO).view(&atlas), - ) - .expect("the masked root serves"); - let (visible_count, extent, min_resolution) = - head_global(section(&masked_bytes, HEAD).expect("HEAD is present")) - .expect("the root publishes its global map"); - - assert_eq!( - visible_count, expected_visible, - "the published count is the root schedule of the view's own cascade" - ); - assert_eq!( - extent, - Some(edges(&expected_extent)), - "the published extent is the visible set's own" - ); - assert_eq!( - min_resolution, expected_deepest, - "the published depth is the deepest occupied scope bucket" - ); - - // And the unmasked root over the same generation publishes the corpus's own numbers, so the - // three assertions above distinguish the view from the artifacts rather than restating them. - let full_bytes = atlas - .tile( - &request(0, 0, 0, Mode::Delta), - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the unmasked root serves"); - let (full_count, full_extent, _) = - head_global(section(&full_bytes, HEAD).expect("HEAD is present")) - .expect("the root publishes its global map"); - - assert_eq!( - full_extent, - Some(edges(&corpus)), - "the unmasked root publishes the generation's extent" - ); - assert_ne!(extent, full_extent, "the census follows the view"); - assert!( - visible_count < full_count, - "the mask removed delivered points from the root's schedule" - ); -} - -/// The census's unmasked fast path answers the extent the walk answers. -/// -/// [`Atlas::census`] reads the artifacts for a proof built as the full-visibility value and walks -/// the base column for a mask. A mask admitting *every* row of the generation is the one input both -/// regimes must agree on. The root count and depth of a masked view belong to its own cascade, and -/// the saturated memo's agreement with the corpus root is the arrival parity test's claim. -#[tokio::test] -async fn census_regimes_extent() { - let (_generation, atlas) = publish("census-regimes").await; - - let admits_everything = mask_hiding(&atlas, &[]); - assert_ne!( - FULL, admits_everything, - "the two proofs are distinct values, so the agreement below is not an identity" - ); - assert_eq!( - atlas.census(&FULL).bounds(), - atlas.census(&admits_everything).bounds(), - "the artifact-read census and the walked census answer the same extent" - ); -} - -/// The masked root publishes the view's own depth, not the generation's. -/// -/// This case accompanies the census witness above, which cannot fail on this clause. Hiding the -/// extreme coordinates leaves the deepest occupied bucket populated, so the visible depth and the -/// corpus depth coincide there and a census ignoring the mask would answer correctly by accident. -/// This case hides exactly the deepest bucket's rows, so the two must part. -#[tokio::test] -async fn masked_root_publishes_the_views_own_depth() { - let (generation, atlas) = publish("masked-depth").await; - let Artifacts { morton, rows, .. } = open_artifacts(&generation); - let row_ids = fixture_row_ids(&rows); - let lengths = morton.fenceposts().lengths(); - - // The generation's deepest occupied bucket, and the positions inside it. - let (deepest, _) = lengths - .iter() - .enumerate() - .rfind(|&(_, &length)| length > 0) - .expect("the fixture occupies a bucket"); - let start: u64 = lengths[..deepest].iter().sum(); - let start = usize::try_from(start).expect("fixture counts fit usize"); - let end = start + usize::try_from(lengths[deepest]).expect("fixture counts fit usize"); - - let mut hidden: Vec = (start..end).map(|position| row_ids[position]).collect(); - hidden.sort_unstable(); - hidden.dedup(); - - // The next occupied bucket below is where the view's depth must land. - let expected = lengths[..deepest] - .iter() - .enumerate() - .rfind(|&(_, &length)| length > 0) - .map_or(0, |(bucket, _)| bucket as u64); - assert!( - expected < deepest as u64, - "the witness must be able to fail: the mask has to vacate the deepest bucket" - ); - - let bytes = atlas - .tile( - &request(0, 0, 0, Mode::Delta), - TileLimits::default(), - Bound::new(&atlas, &mask_hiding(&atlas, &hidden), CutOffset::ZERO).view(&atlas), - ) - .expect("the masked root serves"); - let (_, _, min_resolution) = head_global(section(&bytes, HEAD).expect("HEAD is present")) - .expect("the root publishes its global map"); - - assert_eq!( - min_resolution, expected, - "the published depth is the deepest bucket holding a VISIBLE point" - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/metadata_channel.rs b/libs/@local/graph/atlas/src/serve/tests/metadata_channel.rs deleted file mode 100644 index 914ada79ca9..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/metadata_channel.rs +++ /dev/null @@ -1,152 +0,0 @@ -//! Controlled schedule proof for hidden occupancy in restricted response metadata. -//! -//! A restricted root response publishes two scalars the corpus cascade derives, the visible count -//! of the root's cumulative schedule and the deepest bucket holding a visible row. Both read the -//! bucket assignment, and that assignment is a function of every corpus row, so a hidden row moves -//! them while the visible view stays fixed. -//! -//! The comparison keeps the visible view fixed in the sense the delivery contract names, with -//! identical visible row identity, identical visible Morton keys, and identical relative importance -//! order among visible rows. Both worlds differ only in rows the proof hides. -//! -//! The witness needs two corpora. One corpus under two proofs shares one bucket assignment, so the -//! channel is invisible to any comparison that varies the mask alone. - -use hashql_core::id::{Id as _, IdSlice}; - -use crate::{ - identity::{ImportanceRank, NodeRowId}, - morton::{Depth, MortonKey}, - salt::lod::{cascade, rank::Ranking}, -}; - -/// The fixture's span exponent. -/// -/// The root's cumulative schedule is buckets `0..=2`. -const SPAN: u8 = 2; - -/// The fixture's deepest cascade grid. -const DEEPEST: u8 = 4; - -/// One world's cascade reading over the rows a proof admits. -#[derive(Debug, PartialEq, Eq)] -struct Reading { - /// The visible row's bucket. - bucket: u8, - /// Visible rows of the root's cumulative schedule: bucket at or below the cut. - visible_at_cut: usize, - /// The deepest bucket holding a visible row. - deepest_visible: u8, -} - -fn depth(value: u8) -> Depth { - Depth::new(value).expect("test depths lie within the key width") -} - -/// Builds the ranking that ranks row `r` at position `r`. -/// -/// The fixture names its rows in rank order, so rank and row index coincide. -fn ranking_by_row(rows: usize) -> Ranking { - let rows = u32::try_from(rows).expect("test rows fit u32"); - let row_of_rank: Vec = (0..rows).map(NodeRowId::from_u32).collect(); - let rank_of_row: Vec = (0..rows).map(ImportanceRank::from_u32).collect(); - - Ranking { - row_of_rank: IdSlice::from_boxed_slice(row_of_rank.into_boxed_slice()), - rank_of_row: IdSlice::from_boxed_slice(rank_of_row.into_boxed_slice()), - } -} - -/// Reads one world: the production cascade over `keys`, then the two published scalars over the -/// rows `visible` admits. -fn read(keys: &[MortonKey], visible: &[u32], subject: u32) -> Reading { - let ranking = ranking_by_row(keys.len()); - let keyed = IdSlice::::from_raw(keys); - let buckets = cascade::buckets(keyed, &ranking, depth(DEEPEST)); - - let bucket_of = |row: u32| buckets[NodeRowId::from_u32(row)].get(); - let visible_at_cut = visible - .iter() - .filter(|&&row| bucket_of(row) <= SPAN) - .count(); - let deepest_visible = visible - .iter() - .map(|&row| bucket_of(row)) - .max() - .expect("the fixture admits at least one row"); - - Reading { - bucket: bucket_of(subject), - visible_at_cut, - deepest_visible, - } -} - -/// A hidden row changes the authorized root response's published metadata with visible inputs -/// fixed. -/// -/// Row `V` keys to the origin in both worlds and is the only row the proof admits, so the visible -/// row identity, the visible key set, and the (single-element) visible importance order are equal -/// across the comparison. The sparse world holds `V` alone. The blocked world adds three -/// better-ranked hidden rows that claim `V`'s cell at depths 0, 1, and 2 in turn: -/// -/// | Row | Axis `x` | Claimed cell | Bucket | -/// | ---- | ------------- | ------------------- | -----: | -/// | `H0` | `0x8000_0000` | depth 1, `x` bit 1 | `0` | -/// | `H1` | `0x4000_0000` | depth 1, `x` bit 0 | `1` | -/// | `H2` | `0x2000_0000` | depth 2, `x` bits 0 | `2` | -/// | `V` | `0` | depth 3, `x` bits 0 | `3` | -/// -/// `V` therefore sits at bucket 0 in the sparse world and bucket 3 in the blocked world. The root -/// cut is bucket 2, so the visible count of the root's cumulative schedule reads 1 and then 0, and -/// the deepest visible bucket reads 0 and then 3. -/// -/// This is the negative control for the metadata arm of the fixed-view comparison: the corpus- -/// derived readings disagree, so a candidate derivation that agrees is doing work an inert -/// comparator would not. -/// -/// Scope: the witness establishes the property of the derivation the root response publishes. The -/// delivered row identities and their order are the selector's own arm of the comparison. -#[test] -fn hidden_row_changes_the_authorized_root_metadata() { - let subject = MortonKey::new(0, 0); - let claims_depth_one = MortonKey::new(0x8000_0000, 0); - let shares_depth_one = MortonKey::new(0x4000_0000, 0); - let shares_depth_two = MortonKey::new(0x2000_0000, 0); - - // The sparse world admits the visible row alone, the first occupant of the root cell. - let sparse = read(&[subject], &[0], 0); - assert_eq!( - sparse, - Reading { - bucket: 0, - visible_at_cut: 1, - deepest_visible: 0, - }, - ); - - // The blocked world names the same visible row last in rank order, behind three hidden rows. - let blocked = read( - &[ - claims_depth_one, - shares_depth_one, - shares_depth_two, - subject, - ], - &[3], - 3, - ); - assert_eq!( - blocked, - Reading { - bucket: 3, - visible_at_cut: 0, - deepest_visible: 3, - }, - ); - - assert_ne!( - sparse, blocked, - "the hidden rows moved the published metadata" - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/mod.rs b/libs/@local/graph/atlas/src/serve/tests/mod.rs deleted file mode 100644 index 9072478ca6f..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/mod.rs +++ /dev/null @@ -1,3602 +0,0 @@ -//! Serving reads over a real published generation. -//! -//! The fixture publishes through the production `fit`. Every artifact the serving surface maps is -//! therefore the pipeline's own output. Expectations derive from independently opened artifacts and -//! the schedule laws - fencepost sums, code-column scans, the quad walk - never from the assembly -//! under test. -#![expect( - clippy::little_endian_bytes, - reason = "the expectations spell out the wire contract's little-endian columns" -)] - -use alloc::borrow::Cow; -use core::{assert_matches, num::NonZero}; -use std::collections::{HashMap, HashSet}; - -use camino::Utf8PathBuf; -use futures::future::ready; -use hash_graph_postgres_store::store::{EntityEnd, EntityEvent}; -use hash_graph_temporal_versioning::Timestamp; -use hashql_core::{ - collections::fast_hash_set, - id::{Id, IdSlice, IdVec}, -}; -use rand::{RngExt as _, SeedableRng as _}; -use rand_xoshiro::Xoshiro256PlusPlus; -use smallvec::{SmallVec, smallvec}; -use type_system::{ - knowledge::entity::{EntityId, id::EntityUuid}, - ontology::id::{BaseUrl, VersionedUrl}, - principal::actor_group::WebId, -}; -use uuid::Uuid; -use zerocopy::{LE, U64}; - -use super::{ - Atlas, CutOffset, EdgesError, EdgesLimits, EdgesRequest, GenerationId, OpenOptions, - ServeLimits, TileError, TileLimits, TileQuery, TileRequest, View, ViewCensus, VisibilityLimits, - VisibilityProof, WireRow, WireSecret, codec, - delta::{DeltaEvent, DeltaRegister, DeltaRevision, DeltaSnapshot, PlacementCohort}, - edges::EdgesDetail, - error::OpenAtlasError, - hydrate::{ - DetailError, EdgeSlot, EdgesStore, LocateHydration, LocateLinkHydration, - LocateNodeHydration, LocateOrder, LocateStore, TypeSlot, - }, - locate::{LocateSubgraph, SourceSubject}, - neighbourhood::{EdgeColumns, ServedEdge}, - schedule::{ViewRow, ViewSchedule}, - tile::TileDetail, -}; -use crate::{ - bitset::{CompressedBitSet, DenseBitSlice}, - device::Device, - file::generation::GenerationRoot, - identity::{BasePosition, CardRow, EdgeRowId, ImportanceRank, NodeRowId, OntologyRowId}, - postgres::id::ArchivedOntologyTypeUuid, - salt::{ - embedding::{CardEmbedder, EmbedderFingerprint}, - landmark::select::SelectionOptions, - policy::classifier, - wire::{Mode, tile::TileCoordinate}, - }, -}; - -mod arrival; -mod authorization; -mod auxiliary; -mod delta_edges; -mod density; -mod frame_channel; -mod masking; -mod metadata_channel; -mod open; -mod row_codec; -mod schedule; -mod withdrawal; - -/// The tests' default authority. -/// -/// The operator proof is byte-identical to the pre-visibility serve. -pub(crate) static FULL: VisibilityProof = VisibilityProof::full_visibility(); -use crate::{ - dataset::{ - CANONICAL_DIMENSIONS, Edge, Node as CorpusNode, Ontology, PROJECTOR_DIMENSIONS, - auxiliary::Label, card::Card, memory::MemoryDataset, - }, - file::{ - WriteInto as _, - array::ArrayFile, - digest_file, - generation::{Generation, StagedGeneration}, - identity::{Key, Row}, - morton::read::MortonFile, - quad::read::QuadFile, - repository::FileName, - salt::SaltRepository, - }, - integrity::{Sha256, Update as _}, - math::{AffinityCurve, AlignedVecN, Bounds2, BoxedVecN, Log2, Vec2, VecN, positive}, - morton::{Depth, MortonCell, MortonKey}, - progress::NoProgress, - salt::{ - fit::{ClassifierInput, FitConfig, PlacementOptions, Supplies, fit}, - lod::stage::LodConfig, - wire::{ - edges::EdgesResponse, - tests::section, - tile::{TileHead, TileResponse}, - }, - }, -}; - -/// Corpus rows of the fixture fit. -const NODES: usize = 48; - -/// The fixture schedule. -/// -/// `span = 1`. The cut rule therefore reads `bucket = z + 1` and the root spans buckets `0..=1`. -pub(crate) const FIXTURE_LOD: LodConfig = LodConfig { - span: Log2::new(1).expect("1 lies below the shift width"), - max_tile_depth: 3, -}; - -/// The tile payload's pinned slot indexes. -const HEAD: usize = 0; -const POSITIONS: usize = 1; -pub(crate) const ROW_IDS: usize = 2; -const TYPE_MASK: usize = 3; -const MASS: usize = 4; - -/// The edges payload's pinned slot index. -const EDGE_IDS: usize = 3; - -/// The fixture edge list: `(id, source row, target row)`, edge row order. -/// -/// Row 2 carries a self-loop. Rows 3 and 4 are a reciprocal pair sharing both endpoints. -const FIXTURE_EDGES: [(u64, u64, u64); 6] = [ - (100, 0, 1), - (101, 1, 2), - (102, 2, 2), - (103, 5, 40), - (104, 40, 5), - (105, 3, 7), -]; - -fn scratch(name: &str) -> Utf8PathBuf { - let dir = Utf8PathBuf::from_path_buf(std::env::temp_dir()) - .expect("the temp directory is UTF-8") - .join(format!( - "hash-graph-atlas-serve-{}-{name}", - std::process::id() - )); - let _: Result<(), std::io::Error> = std::fs::remove_dir_all(&dir); - dir -} - -/// One fixture corpus row, owning its projector embedding under a plain integer id. -type FixtureNode = CorpusNode<'static, U64>; - -/// The fit-scale corpus rows. -/// -/// Unit-norm pseudo-random representations whose canonical embeddings extend them with zeros, -/// typed per row by `types`. Every fixture corpus shares this geometry. The placement -/// conditioning is therefore one derivation. -fn fixture_nodes( - types: impl Fn(usize) -> SmallVec, -) -> ( - Vec, - HashMap>, -) { - let mut rng = Xoshiro256PlusPlus::seed_from_u64(0x5E4E); - let mut canonical = HashMap::new(); - - let nodes: Vec<_> = (0..NODES) - .map(|row| { - let mut components = [0.0_f32; PROJECTOR_DIMENSIONS]; - for component in &mut components { - *component = rng.random::() - 0.5; - } - let norm = components - .iter() - .map(|&component| f64::from(component) * f64::from(component)) - .sum::() - .sqrt(); - #[expect( - clippy::cast_possible_truncation, - reason = "the normalization factor of a 512-component vector is far inside f32 \ - range" - )] - for component in &mut components { - *component = (f64::from(*component) / norm) as f32; - } - - let mut extended = BoxedVecN::::zero(); - extended.as_array_mut()[..PROJECTOR_DIMENSIONS].copy_from_slice(&components); - canonical.insert(row as u64, extended); - - CorpusNode { - id: U64::::new(row as u64), - ontology: types(row), - embedding: Cow::Owned(BoxedVecN::new(&VecN::new(components))), - confidence: None, - } - }) - .collect(); - - (nodes, canonical) -} - -/// A fit-scale corpus. -/// -/// The [`fixture_nodes`] geometry with one node type alternating between two ontology rows, and -/// one link type. -fn fixture_dataset() -> MemoryDataset { - let (nodes, canonical) = fixture_nodes(|row| smallvec![OntologyRowId::from_usize(row & 1)]); - - let edges = FIXTURE_EDGES - .into_iter() - .map(|(id, source, target)| Edge { - id: U64::::new(id), - source: NodeRowId::new(source), - target: NodeRowId::new(target), - ontology: smallvec![OntologyRowId::new(2)], - embedding: None, - confidence: None, - source_confidence: None, - target_confidence: None, - }) - .collect(); - - let ontology = vec![ - Ontology { - id: U64::::new(0), - parents: smallvec![], - }, - Ontology { - id: U64::::new(1), - parents: smallvec![], - }, - Ontology { - id: U64::::new(2), - parents: smallvec![], - }, - ]; - - let cards = HashMap::from([ - (0, Card::verbatim("Person entity card".to_owned())), - (1, Card::verbatim("Company entity card".to_owned())), - (2, Card::verbatim("Employment link card".to_owned())), - ]); - - MemoryDataset::new(nodes, edges, ontology, canonical, cards) -} - -/// A deterministic provider deriving each embedding from its text hash. -struct HashEmbedder; - -impl CardEmbedder for HashEmbedder { - type Error = !; - - fn fingerprint(&self) -> EmbedderFingerprint { - let mut hasher = Sha256::new(); - hasher.update(b"serve test embedder"); - EmbedderFingerprint::new(hasher.finalize()) - } - - fn embed<'text>( - &self, - texts: impl IntoIterator + Send, - ) -> impl Future>, Self::Error>> + Send - { - ready(Ok(texts - .into_iter() - .map(|text| { - let mut hasher = Sha256::new(); - hasher.update(text.as_bytes()); - let bytes = hasher.finalize().to_bytes(); - - let mut vector = BoxedVecN::zero(); - for (component, &byte) in vector.as_array_mut().iter_mut().zip(bytes.iter().cycle()) - { - *component = f32::from(byte) / 255.0; - } - vector - }) - .collect())) - } -} - -/// A deterministic classifier input fitted from a synthetic corpus. -/// -/// The supplied model input of the fixture fit. -fn fixture_classifier() -> ClassifierInput { - const ROWS: usize = 4; - // Coprime to the dimension, hence no two corpus rows repeat. - const PATTERN: [f32; 13] = [ - -0.75, -0.625, -0.5, -0.375, -0.25, -0.125, 0.0, 0.125, 0.25, 0.375, 0.5, 0.625, 0.75, - ]; - - let mut storage = BoxedVecN::<{ ROWS * CANONICAL_DIMENSIONS }>::zero(); - for (component, &value) in storage - .as_array_mut() - .iter_mut() - .zip(PATTERN.iter().cycle()) - { - *component = value; - } - let embeddings: &IdSlice> = IdSlice::from_raw( - AlignedVecN::from_slice(storage.as_array()).expect("boxed storage aligns"), - ); - - let rows: IdVec = [ - ([0.7, 0.2, 0.1], b"group-a" as &[u8]), - ([0.2, 0.6, 0.2], b"group-b"), - ([0.1, 0.2, 0.7], b"group-c"), - ([0.3, 0.4, 0.3], b"group-d"), - ] - .into_iter() - .map(|(target, group)| { - let mut hasher = Sha256::new(); - hasher.update(group); - classifier::TrainingRow { - target, - weight: 1.0, - group: hasher.finalize(), - } - }) - .collect(); - - let training = - classifier::TrainingSet::new(embeddings, &rows).expect("the fixture corpus validates"); - let classifier = classifier::fit( - training, - classifier::FitConfig { folds: 2, .. }, - &NoProgress, - ) - .expect("the fixture classifier fits") - .classifier; - - let mut hasher = Sha256::new(); - hasher.update(b"serve fixture classifier"); - ClassifierInput::Supplied { - classifier, - source: hasher.finalize(), - } -} - -fn fixture_config() -> FitConfig { - FitConfig { - seed: 11, - selection: SelectionOptions { - maximum_count: NonZero::new(8).expect("the fixture capacity is nonzero"), - .. - }, - curve: AffinityCurve::fit(positive!(1.0), positive!(0.1)) - .expect("the reference falloff is well-conditioned"), - neighbours: NonZero::new(4).expect("the fixture neighbour count is nonzero"), - // The serving fixture reads artifacts, not placement quality: - // it opts out of the default's training run. - placement: PlacementOptions::LandmarkBaseline, - lod: FIXTURE_LOD, - .. - } -} - -/// Fits and publishes one fixture generation, as the pipeline writes it. -/// -/// Identity artifacts carry the memory dataset's 8-byte positional ids, which the serving open -/// rejects. -async fn fit_fixture(name: &str) -> (GenerationRoot, Generation) { - fit_dataset(name, &fixture_dataset()).await -} - -/// Fits and publishes `dataset` as one fixture generation. -async fn fit_dataset(name: &str, dataset: &MemoryDataset) -> (GenerationRoot, Generation) { - let root = GenerationRoot::new(scratch(name)).expect("the root should open"); - let published = fit( - dataset, - &HashEmbedder, - &fixture_config(), - Supplies { - classifier: &fixture_classifier(), - .. - }, - &root, - Device::Cpu.pin(0).resolve(), - &NoProgress, - ) - .await - .expect("the fit should publish"); - - let generation = root - .open(published.id()) - .expect("the published generation should open"); - - (root, generation) -} - -/// The versioned type URL behind fixture ontology row `row`. -/// -/// The rewritten ontology identities key each row by the uuid its URL derives, exactly as the -/// store's identities would. -fn fixture_type_url(row: u64) -> String { - format!("https://example.com/types/fixture-{row}/v/1") -} - -/// The edge-domain seed offset. -/// -/// Link entities own ids disjoint from node ids, as the store's would be. -const EDGE_SEED: u8 = 64; - -/// Republishes `generation` under `root` with `edit` applied to a staged copy of its files. -/// -/// Every published file copies into a fresh staging, `edit` rewrites whichever staged files it -/// names, and the manifest re-binds each artifact to the digest of its staged bytes before the -/// staging seals as a new generation. The original generation stays on disk as it was. Open -/// verifies every file against the digest its manifest records, hence a published file edited in -/// place fails as corruption. A test that needs a generation whose files are structurally -/// inconsistent while each is exactly what its manifest records publishes it through here. -fn republish( - root: &GenerationRoot, - generation: &Generation, - edit: impl FnOnce(&StagedGeneration), -) -> Generation { - let published = { - let staging = root.stage().expect("the staging should create"); - for file in generation.repository().files.files() { - std::fs::copy(generation.path_of(&file.name), staging.path_of(&file.name)) - .expect("a published file should copy into the staging"); - } - edit(&staging); - - // The manifest re-binds through its serialized form. Every binding serializes as its - // entry's name and hash, an absent optional binding serializes as null, and the staged - // file of an entry's name supplies its hash. - let mut document = - serde_json::to_value(generation.repository()).expect("the manifest should serialize"); - let entries = document - .get_mut("files") - .and_then(serde_json::Value::as_object_mut) - .expect("the manifest holds its files object"); - for entry in entries.values_mut().filter(|entry| !entry.is_null()) { - let name: FileName = entry - .get("name") - .cloned() - .map(serde_json::from_value) - .expect("an entry names its file") - .expect("an entry's name is a file name"); - let digest = digest_file(staging.path_of(&name)).expect("a staged file should digest"); - entry["hash"] = serde_json::to_value(digest).expect("a digest should serialize"); - } - let repository: SaltRepository = - serde_json::from_value(document).expect("the rebound manifest should deserialize"); - - staging - .seal(&repository) - .expect("the edited staging should seal") - }; - - root.open(published.id()) - .expect("the republished generation should open") -} - -/// Republishes a fixture generation with its identity artifacts rewritten to store-width ids. -/// -/// Deterministic by row: ontology row `r` keys the uuid derived from [`fixture_type_url`] of `r`, -/// and node row `r` keys [`entity_id_of`] of `r`. Edge row `r` keys [`entity_id_of`] of -/// `EDGE_SEED + r`. -/// -/// The memory dataset speaks 8-byte positional ids, and the serving open rejects those. The -/// rewrite is the test-lane bridge that gives a fixture generation store-width ids. It goes -/// through [`republish`], hence the rewritten artifacts carry the digests their manifest records -/// and the returned generation opens verified. -/// -/// Display payloads copy through the rewrite row by row. A dataset's labels and icons therefore -/// survive the bridge and serve exactly as the production pipeline wrote them. -fn store_identities(root: &GenerationRoot, generation: &Generation) -> Generation { - use type_system::ontology::id::VersionedUrl; - - use crate::{ - file::identity::read::IdentityFile, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - salt::fit::prepare::identity::IdentityTable, - }; - - fn entity_table(rows: u64, seed: u8) -> IdentityTable { - let mut table = IdentityTable::new(); - for row in 0..rows { - let row = u8::try_from(row).expect("fixture row counts fit u8"); - table.push(entity_id_of(seed + row)); - } - table - } - - let files = &generation.repository().files; - let rows_of = |name: &FileName| { - IdentityFile::open(generation.path_of(name)) - .expect("the published identity artifact opens") - .rows() - }; - - let ontology_rows = rows_of(&files.ontology_identities.name()); - let mut ontology = IdentityTable::::new(); - for row in 0..ontology_rows { - let url: VersionedUrl = fixture_type_url(row) - .parse() - .expect("the fixture URL parses"); - ontology.push(ArchivedOntologyTypeUuid::from_url(&url)); - } - - let nodes = entity_table::(rows_of(&files.node_identities.name()), 0); - let edges = entity_table::(rows_of(&files.edge_identities.name()), EDGE_SEED); - - let ontology_payloads = payloads_of(&generation.path_of(&files.ontology_identities.name())); - let node_payloads = payloads_of(&generation.path_of(&files.node_identities.name())); - let edge_payloads = payloads_of(&generation.path_of(&files.edge_identities.name())); - - republish(root, generation, |staging| { - rewrite_identities( - &staging.path_of(&files.ontology_identities.name()), - &ontology, - &ontology_payloads, - ); - rewrite_identities( - &staging.path_of(&files.node_identities.name()), - &nodes, - &node_payloads, - ); - rewrite_identities( - &staging.path_of(&files.edge_identities.name()), - &edges, - &edge_payloads, - ); - }) -} - -/// Reads every row's payload bytes out of a published identity artifact. -/// -/// The mapping drops when this returns. The caller can therefore truncate and rewrite the file -/// afterwards. -fn payloads_of(path: &camino::Utf8Path) -> Vec> { - use crate::file::identity::read::IdentityFile; - - let file = IdentityFile::open(path).expect("the published identity artifact opens"); - let payload = file.payload(); - file.spans() - .iter() - .map(|span| { - let offset = usize::try_from(span.offset()).expect("payload regions fit usize"); - let length = usize::try_from(span.length()).expect("payload regions fit usize"); - payload[offset..offset + length].to_vec() - }) - .collect() -} - -/// Overwrites one identity artifact with a hand-built table, keeping the given payloads. -/// -/// `payloads` carries one byte string per row, in row order - [`payloads_of`] of the published -/// artifact when the rewrite preserves them. -fn rewrite_identities( - path: &camino::Utf8Path, - table: &crate::salt::fit::prepare::identity::IdentityTable, - payloads: &[Vec], -) where - R: Row, - I: Key, -{ - let rows = usize::try_from(table.len()).expect("fixture row counts fit the address space"); - assert_eq!(payloads.len(), rows, "one payload survives per row"); - let mut file = recreate_writable(path); - let payloads: Vec<&I::Payload> = payloads - .iter() - .map(|bytes| { - ::try_ref_from_bytes(bytes) - .expect("published payload bytes cast as the id type's payload") - }) - .collect(); - let _digest = table - .write_into(payloads, &mut file) - .expect("the identities should write"); -} - -/// Reopens a published artifact for rewriting. -/// -/// Sealing dropped the write permission. A tamper therefore lifts it before truncating the file. -fn recreate_writable(path: &camino::Utf8Path) -> std::fs::File { - let mut permissions = std::fs::metadata(path) - .expect("the published artifact should stat") - .permissions(); - #[expect( - clippy::permissions_set_readonly_false, - reason = "tests rewrite their own scratch files" - )] - permissions.set_readonly(false); - std::fs::set_permissions(path, permissions).expect("the permissions should set"); - std::fs::File::create(path).expect("the published artifact rewrites") -} - -/// The suite's wire secret, exactly the codec's key width, with an arbitrary value. -const TEST_WIRE_SECRET: [u8; 32] = *b"atlas-test-wire-secret-32-bytes!"; - -/// The open options every suite open uses. -fn test_open_options() -> OpenOptions { - OpenOptions { - wire_secret: WireSecret::new(TEST_WIRE_SECRET), - } -} - -/// Publishes one fixture generation with store-width identities and opens its serving surface. -pub(crate) async fn publish(name: &str) -> (Generation, Atlas) { - publish_dataset(name, &fixture_dataset()).await -} - -/// Publishes `dataset` with store-width identities and opens its serving surface. -async fn publish_dataset(name: &str, dataset: &MemoryDataset) -> (Generation, Atlas) { - let (root, generation) = fit_dataset(name, dataset).await; - let generation = store_identities(&root, &generation); - let atlas = - Atlas::open(&root, generation.id(), test_open_options()).expect("the atlas should open"); - - (generation, atlas) -} - -/// One proof's delivery inputs, owned so that a test can hand assembly a [`View`]. -/// -/// A view borrows the census and the schedule its scope resolved; production holds both in the -/// visibility cache entry, and a test holds them here. Binding through [`Bound::view`] runs the -/// same pairing check the request boundary runs. -#[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] -pub(crate) struct Bound<'proof> { - proof: &'proof VisibilityProof, - census: ViewCensus, - schedule: ViewSchedule, - k: CutOffset, - cohort: PlacementCohort<'proof>, - delta: Option<&'proof DeltaSnapshot>, -} - -impl<'proof> Bound<'proof> { - /// Resolves `proof`'s census and schedule, the way a scope resolution would. - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - fn new(atlas: &Atlas, proof: &'proof VisibilityProof, k: CutOffset) -> Self { - Self::resolved(atlas, proof, PlacementCohort::EMPTY, k) - } - - /// Resolves `proof`'s census and schedule against `cohort`, the way an arrival-bearing - /// scope resolution would. - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - pub(crate) fn resolved( - atlas: &Atlas, - proof: &'proof VisibilityProof, - cohort: PlacementCohort<'proof>, - k: CutOffset, - ) -> Self { - Self { - proof, - census: atlas.census(proof), - schedule: ViewSchedule::of(atlas, proof, cohort), - k, - cohort, - delta: None, - } - } - - /// Carries `delta` as the view's ingress capture, the way a data route would. - pub(crate) fn withdrawing(mut self, delta: &'proof DeltaSnapshot) -> Self { - self.delta = Some(delta); - self - } - - /// Resolves `proof` at the zero offset, the corpus-equivalent cut. - fn of(atlas: &Atlas, proof: &'proof VisibilityProof) -> Self { - Self::new(atlas, proof, CutOffset::ZERO) - } - - /// Binds the delivery view assembly reads. - pub(crate) fn view(&self, atlas: &Atlas) -> View<'_> { - View::bind( - atlas.grid, - self.proof, - self.census, - &self.schedule, - self.k, - self.cohort, - self.delta, - ) - .expect("the fixture's proof, schedule and offset pair") - } -} - -/// Binds `proof`'s delivery view at the zero offset and reads it once. -pub(crate) fn viewing( - atlas: &Atlas, - proof: &VisibilityProof, - body: impl FnOnce(&View<'_>) -> T, -) -> T { - let bound = Bound::of(atlas, proof); - - body(&bound.view(atlas)) -} - -pub(crate) fn request(z: u8, x: u32, y: u32, mode: Mode) -> TileRequest { - TileRequest { - coordinate: TileCoordinate { z, x, y }, - query: TileQuery { - mode, - ..TileQuery::default() - }, - } -} - -/// Returns the tile coordinate addressing a Morton cell. -fn coordinate_of(cell: MortonCell) -> TileCoordinate { - let z = cell.depth().get(); - if z == 0 { - return TileCoordinate { z, x: 0, y: 0 }; - } - - let [x, y] = cell.min_key().coordinates(); - TileCoordinate { - z, - x: x >> (32 - z), - y: y >> (32 - z), - } -} - -/// Collects every quad node with its cell, walking children in Morton child order from the root. -fn walk(quad: &QuadFile, node: u32, cell: MortonCell, out: &mut Vec<(u32, MortonCell)>) { - out.push((node, cell)); - let children = cell - .children() - .expect("fixture nodes stay above Depth::MAX"); - for (quadrant, child_cell) in children.into_iter().enumerate() { - if let Some(child) = quad.nodes()[node as usize].child(quadrant) { - walk(quad, child, child_cell, out); - } - } -} - -/// Counts the fixture codes inside one cell by scanning the column. -fn population(morton: &MortonFile, cell: MortonCell) -> u64 { - morton - .codes() - .iter() - .filter(|code| cell.contains(MortonKey::from_bits(code.get()))) - .count() as u64 -} - -/// Extracts a subgraph's delivered node rows, in delivered order. -/// -/// The subgraphs this resolves deliver fitted rows alone. An arrival vessel is therefore a -/// fixture defect rather than a case. -pub(crate) fn delivered_row_ids(atlas: &Atlas, subgraph: &LocateSubgraph) -> Vec { - let row_ids = atlas.row_ids(); - subgraph - .delivered - .iter() - .map(|&vessel| match vessel { - ViewRow::Base(position) => row_ids[position], - ViewRow::Arrival(index) => { - panic!("a fitted-only fixture delivered the arrival vessel {index:?}") - } - }) - .collect() -} - -/// Unwraps a fitted delivered edge. -/// -/// The subgraphs this resolves deliver fitted edges alone. A delta vessel is therefore a -/// fixture defect rather than a case. -pub(crate) fn fitted(edge: ServedEdge) -> crate::serve::neighbourhood::DeliveredEdge { - match edge { - ServedEdge::Fitted(edge) => edge, - ServedEdge::Delta(edge) => panic!("a fitted-only fixture delivered the delta {edge:?}"), - } -} - -/// Decodes a `ROW_IDS` section into row ids. -fn decode_rows(bytes: &[u8]) -> Vec { - let (chunks, remainder) = bytes.as_chunks::<4>(); - assert!(remainder.is_empty(), "row sections are whole u32 columns"); - chunks - .iter() - .map(|&chunk| u32::from_le_bytes(chunk)) - .collect() -} - -/// The independently opened serving artifacts of one generation. -pub(crate) struct Artifacts { - pub morton: MortonFile, - pub quad: QuadFile, - pub coordinates: ArrayFile, - pub rows: ArrayFile, -} - -pub(crate) fn open_artifacts(generation: &Generation) -> Artifacts { - let files = &generation.repository().files; - Artifacts { - morton: MortonFile::open(generation.path_of(&files.morton.name())) - .expect("the morton artifact should open"), - quad: QuadFile::open(generation.path_of(&files.quad.name())) - .expect("the quad artifact should open"), - coordinates: ArrayFile::open(generation.path_of(&files.wire_coordinates.name())) - .expect("the coordinate artifact should open"), - rows: ArrayFile::open(generation.path_of(&files.row_of_position.name())) - .expect("the row artifact should open"), - } -} - -/// Reads the gather column narrowed to the fixture tests' `u32` row vocabulary. -pub(crate) fn fixture_row_ids(rows: &ArrayFile) -> Vec { - rows.column::() - .expect("the row column is a node-row column") - .as_raw() - .iter() - .map(|row| row.as_u32()) - .collect() -} - -/// The generation's extent, and the rows attaining any of its four extremes. -/// -/// Removing exactly these rows from a view vacates every edge of the extent, which is what lets -/// an aggregate witness fail on an extent read off the artifacts rather than off the view: with -/// any edge still attained, the corpus extent and the view's extent agree there and the wrong -/// answer looks right. -fn extremes(points: &[Vec2], row_ids: &[u32]) -> (Bounds2, Vec) { - let corpus = Bounds2::from_points(points.iter().copied()).expect("the fixture holds points"); - - // Exact equality is the predicate: an extremum is one of this column's own values. A row - // therefore attains it bit-for-bit or does not attain it. - #[expect( - clippy::float_cmp, - reason = "the comparand comes from this column itself, hence bit equality is the intended \ - test" - )] - let mut attaining: Vec = points - .iter() - .enumerate() - .filter(|(_, point)| { - point.x() == corpus.min().x() - || point.x() == corpus.max().x() - || point.y() == corpus.min().y() - || point.y() == corpus.max().y() - }) - .map(|(position, _)| row_ids[position]) - .collect(); - attaining.sort_unstable(); - attaining.dedup(); - - (corpus, attaining) -} - -/// The generation's extent, and rows whose removal both vacates every edge and empties a root cell. -/// -/// The extreme-attaining rows alone move the extent. The root count moves only when the withdrawn -/// rows were some cell's whole population, since the root cut delivers one row per occupied cell -/// and a cell keeping one visible row keeps its representative. Whether any -/// cell holds nothing but extreme rows is a property of the fitted layout rather than of the -/// witness, and a layout is target-specific down to the last bit of each coordinate. A witness -/// resting on that coincidence therefore has teeth on one target and none on the next. -/// -/// Adding the rest of one extreme-bearing cell empties that cell whatever the layout, which takes -/// the occupied-cell count down by at least one and the delivered count with it. An aggregate read -/// off the artifacts instead of off the view is then a detectable answer on the count axis as well -/// as on the extent axis, and the construction asserts that outcome rather than trusting it. -pub(crate) fn extremes_vacating_a_root_cell( - atlas: &Atlas, - points: &[Vec2], - row_ids: &[u32], -) -> (Bounds2, Vec) { - let (corpus, attaining) = extremes(points, row_ids); - - // The root's cut, which is the depth whose cells the root tile delivers one row apiece from. - let cut = Depth::new(FIXTURE_LOD.span.get()).expect("the fixture span is a valid depth"); - let cell_of = |position: usize| { - atlas - .morton - .code(BasePosition::from_u32( - u32::try_from(position).expect("fixture positions fit u32"), - )) - .prefix(cut) - }; - - let hidden: HashSet = attaining.iter().copied().collect(); - let mut population: HashMap> = HashMap::new(); - for position in 0..points.len() { - population - .entry(cell_of(position)) - .or_default() - .push(position); - } - assert!( - population.len() > 1, - "the fixture occupies one root cell, hence no mask can vacate one and leave a view" - ); - - // Vacating the cell that keeps the fewest visible rows costs the view the least, and the cell - // key breaks ties rather than a map's iteration order. - let survivors = |positions: &[usize]| { - positions - .iter() - .filter(|&&position| !hidden.contains(&row_ids[position])) - .count() - }; - let (_, vacated) = population - .iter() - .filter(|(_, positions)| positions.iter().any(|&at| hidden.contains(&row_ids[at]))) - .map(|(&cell, positions)| ((survivors(positions), cell), positions)) - .min_by_key(|&(order, _)| order) - .expect("the extremes occupy a cell"); - - let mut hiding = attaining; - hiding.extend(vacated.iter().map(|&position| row_ids[position])); - hiding.sort_unstable(); - hiding.dedup(); - - // What the count witness rests on, checked here rather than left to the next layout to decide. - let remaining = population - .values() - .filter(|positions| { - positions - .iter() - .any(|&position| !hiding.contains(&row_ids[position])) - }) - .count(); - assert!( - remaining < population.len(), - "the hidden set empties no root cell, hence the count witness would have no teeth" - ); - assert!(remaining > 0, "the hidden set empties every root cell"); - - (corpus, hiding) -} - -/// Every operator head accounts for exactly the rows its response delivered. -/// -/// The wire law is `sum(runs) == delivered` in every response. A client paints from `runs`, reading -/// bucket `b0 + i` at column offset `sum(runs[..i])`. A head that overcounts therefore moves every -/// later bucket's points. The producer asserts the identity when it encodes. This reads the same -/// law back off the bytes, over the corpus schedule rather than a scope cascade. -/// -/// The scoped side of the sweep lives in `schedule.rs`, and both reach it through [`head_counts`]. -#[tokio::test] -async fn operator_head_accounting() { - let (generation, atlas) = publish("operator-head").await; - let Artifacts { quad, .. } = open_artifacts(&generation); - - let root_cell = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - let mut nodes = Vec::new(); - walk(&quad, 0, root_cell, &mut nodes); - assert!(nodes.len() > 1, "the fixture quadtree subdivides"); - - for mode in [Mode::Delta, Mode::Total] { - for &(_node, cell) in &nodes { - let TileCoordinate { z, x, y } = coordinate_of(cell); - let bytes = atlas - .tile( - &request(z, x, y, mode), - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the operator tile serves"); - let rows = decode_rows(section(&bytes, ROW_IDS).expect("ROW_IDS is present")); - - assert_head_delivers(&bytes, rows.len() as u64); - } - } -} - -#[tokio::test] -async fn serves_published_tiles() { - let (generation, atlas) = publish("serves").await; - assert_eq!(atlas.generation(), generation.id()); - - let Artifacts { - morton, - quad, - coordinates, - rows, - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - // The root delta delivers buckets 0..=m: the head of the base - // order, sized by the fencepost lengths. - let bytes = atlas - .tile( - &request(0, 0, 0, Mode::Delta), - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the root tile should serve"); - assert_eq!(&bytes[..8], b"SALTILET"); - - let delivered: u64 = morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - assert!(delivered > 0, "the fixture root delivers points"); - - let head = usize::try_from(delivered).expect("fixture counts fit usize"); - let positions_section = section(&bytes, POSITIONS).expect("POSITIONS is present"); - let rows_section = section(&bytes, ROW_IDS).expect("ROW_IDS is present"); - assert_eq!(positions_section.len() as u64, delivered * 8); - assert_eq!(rows_section.len() as u64, delivered * 4); - assert!( - section(&bytes, TYPE_MASK).is_none(), - "TYPE_MASK is absent without colouring", - ); - assert!(section(&bytes, MASS).is_none(), "MASS is a reserved key"); - - // The wire column carries the codec's ids: encode the head of - // the base order through the independent derivation. - let node_codec = test_codec(&atlas); - let expected_rows: Vec = row_ids[..head] - .iter() - .flat_map(|&row| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - .to_le_bytes() - }) - .collect(); - assert_eq!(rows_section, expected_rows); - let expected_positions: Vec = points[..head] - .iter() - .flat_map(|point| { - let [x, y] = [point.x().to_le_bytes(), point.y().to_le_bytes()]; - x.into_iter().chain(y) - }) - .collect(); - assert_eq!(positions_section, expected_positions); - - // The root's total delivery equals its delta delivery. - let total = atlas - .tile( - &request(0, 0, 0, Mode::Total), - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the root total should serve"); - assert_eq!( - section(&total, POSITIONS).expect("POSITIONS is present"), - positions_section, - ); - assert_eq!( - section(&total, ROW_IDS).expect("ROW_IDS is present"), - rows_section, - ); - - // Delta tiles partition the base order: walking every quad node - // delivers each row exactly once. - let root_cell = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - let mut nodes = Vec::new(); - walk(&quad, 0, root_cell, &mut nodes); - assert!(nodes.len() > 1, "the fixture quadtree subdivides"); - - let mut delivered_rows = decode_rows(rows_section); - for &(node, cell) in &nodes[1..] { - let coordinate = coordinate_of(cell); - let bytes = atlas - .tile( - &TileRequest { - coordinate, - query: TileQuery::default(), - }, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("every node tile should serve"); - let tile_rows = decode_rows(section(&bytes, ROW_IDS).expect("ROW_IDS is present")); - - let run = quad.nodes()[node as usize].run(); - assert_eq!(tile_rows.len() as u64, run.end - run.start); - delivered_rows.extend(tile_rows); - } - - let mut expected: Vec = row_ids - .iter() - .map(|&row| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }) - .collect(); - expected.sort_unstable(); - delivered_rows.sort_unstable(); - assert_eq!(delivered_rows, expected, "each row arrives exactly once"); -} - -#[tokio::test] -async fn serves_empty_and_deepest_cells() { - let (generation, atlas) = publish("cells").await; - let Artifacts { - morton, - quad, - coordinates, - .. - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - - // A valid coordinate with no quad node serves the honest empty - // tile, byte for byte. - let empty_cell = (1..=FIXTURE_LOD.max_tile_depth) - .flat_map(|z| { - let cells = 1_u32 << z; - (0..cells).flat_map(move |x| { - (0..cells).map(move |y| { - MortonCell::new(Depth::new(z).expect("fixture depths are valid"), x, y) - .expect("the coordinates lie on the grid") - }) - }) - }) - .find(|&cell| quad.locate(cell).is_none()) - .expect("the fixture grid has empty cells"); - - let coordinate = coordinate_of(empty_cell); - let expected = TileResponse { - head: TileHead { - generation: generation.id().digest(), - variant: 0, - coordinate, - mode: Mode::Delta, - first_bucket: coordinate.z + FIXTURE_LOD.span.get(), - runs: &[0], - global: None, - children: 0, - }, - delivered: crate::salt::wire::tile::DeliveredSet::Ranges(&[]), - positions: IdSlice::from_raw(points), - rows: IdSlice::from_raw(&[]), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - } - .encode(); - assert_eq!( - atlas - .tile( - &TileRequest { - coordinate, - query: TileQuery::default(), - }, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the empty tile should serve"), - expected, - ); - - // At the deepest zoom a total tile delivers its cell's whole - // population: the cut reaches the catch-all bucket. - let deep_cell = MortonKey::from_bits(morton.codes()[BasePosition::from_u32(0)].get()) - .cell(Depth::new(FIXTURE_LOD.max_tile_depth).expect("the deepest tile depth is valid")); - let bytes = atlas - .tile( - &TileRequest { - coordinate: coordinate_of(deep_cell), - query: TileQuery { - mode: Mode::Total, - ..TileQuery::default() - }, - }, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the deepest total tile should serve"); - let tile_rows = decode_rows(section(&bytes, ROW_IDS).expect("ROW_IDS is present")); - assert_eq!(tile_rows.len() as u64, population(&morton, deep_cell)); -} - -#[tokio::test] -async fn tile_contract_rejections() { - let (_generation, atlas) = publish("rejects").await; - - assert_eq!( - atlas.tile( - &request(4, 0, 0, Mode::Delta), - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas) - ), - Err(TileError::Depth { z: 4, maximum: 3 }), - ); - assert_eq!( - atlas.tile( - &request(2, 4, 0, Mode::Delta), - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas) - ), - Err(TileError::Grid { z: 2, x: 4, y: 0 }), - ); - - let mut colored = request(0, 0, 0, Mode::Delta); - colored.query.colored_type_ids = vec![ - "https://example.com/types/thing/v/1" - .parse() - .expect("the literal is a versioned url"); - TileLimits::default().colored_type_ids as usize + 1 - ]; - assert_eq!( - atlas.tile( - &colored, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas) - ), - Err(TileError::Types { - count: TileLimits::default().colored_type_ids as usize + 1, - maximum: TileLimits::default().colored_type_ids, - }), - ); - - // Bucket 3 first enters at zoom 3 - span(1) - k(0) = 2, inside the grid's bound of 3. - let manifest = serde_json::to_value(atlas.manifest( - ServeLimits::default().manifest_limits(VisibilityLimits::default()), - CutOffset::ZERO, - 3, - )) - .expect("the manifest serializes"); - assert_eq!( - manifest, - serde_json::json!({ - "generation": atlas.generation().to_string(), - "wireVersion": 1, - "variants": ["plain"], - "bucketSchedule": { "span": 2, "cut": "z+1", "maxZoom": 3 }, - "scopeSchedule": { "k": 0, "cut": "z+1", "maxZoom": 2 }, - "limits": { "coloredTypeIds": 32, "edgesTiles": 256, "locateEdges": 512, "locateProperties": 10, "locateLinkProperties": 10, "locateLinkTypeIds": 5, "translateEntityIds": 1024, "authorityRefreshSeconds": 480, "authorityHardSeconds": 600 }, - // No createdAt: the fixture dataset has no temporal axes. - }), - ); - - // The cut rule's three edges, each one field read: an offset deepens the subtrahend, the - // grid clamps a catch-all-deep bucket, and an empty view saturates at the root. - let limits = ServeLimits::default().manifest_limits(VisibilityLimits::default()); - let offset = serde_json::to_value(atlas.manifest(limits, CutOffset::new(1), 3)) - .expect("the manifest serializes"); - assert_eq!(offset["scopeSchedule"]["maxZoom"], 1, "3 - 1 - 1"); - let clamped = serde_json::to_value(atlas.manifest(limits, CutOffset::ZERO, 9)) - .expect("the manifest serializes"); - assert_eq!(clamped["scopeSchedule"]["maxZoom"], 3, "min(9 - 1, 3)"); - let empty = serde_json::to_value(atlas.manifest(limits, CutOffset::ZERO, 0)) - .expect("the manifest serializes"); - assert_eq!( - empty["scopeSchedule"]["maxZoom"], 0, - "an empty view saturates" - ); -} - -#[test] -fn tile_query_delta_default() { - let query: TileQuery = serde_json::from_str("{}").expect("the empty body parses"); - assert_eq!(query.mode, Mode::Delta); - assert!(query.colored_type_ids.is_empty()); - assert_eq!(query.detail, TileDetail::Minimal); - - let query: TileQuery = serde_json::from_str( - r#"{ - "mode": "total", - "coloredTypeIds": ["https://example.com/types/thing/v/1"], - "detail": "auxiliary" - }"#, - ) - .expect("the full body parses"); - assert_eq!(query.mode, Mode::Total); - assert_eq!(query.colored_type_ids.len(), 1); - assert_eq!(query.detail, TileDetail::Auxiliary); -} - -#[test] -fn atlas_shared_across_requests() { - const fn shared() {} - shared::(); -} - -#[test] -fn open_rejects_unpublished() { - let root = GenerationRoot::new(scratch("unpublished")).expect("the root should open"); - let id: GenerationId = "0000000000000000000000000000000000000000000000000000000000000000" - .parse() - .expect("the zero digest parses"); - assert_matches!( - Atlas::open(&root, id, test_open_options()), - Err(OpenAtlasError::Unpublished(unpublished)) if unpublished == id, - ); -} - -/// The edge-side serving artifacts of one generation, independently opened. -struct EdgeArtifacts { - endpoints: ArrayFile, - ranks: ArrayFile, - positions: ArrayFile, -} - -fn open_edge_artifacts(generation: &Generation) -> EdgeArtifacts { - let files = &generation.repository().files; - EdgeArtifacts { - endpoints: ArrayFile::open(generation.path_of(&files.edge_endpoints.name())) - .expect("the endpoint artifact should open"), - ranks: ArrayFile::open(generation.path_of(&files.rank_of_position.name())) - .expect("the rank artifact should open"), - positions: ArrayFile::open(generation.path_of(&files.position_of_row.name())) - .expect("the position artifact should open"), - } -} - -/// Every tile coordinate of the deepest zoom. -/// -/// The cut reaches the catch-all bucket. The grid therefore delivers the whole corpus. -fn full_grid() -> Vec { - let cells = 1_u32 << FIXTURE_LOD.max_tile_depth; - (0..cells) - .flat_map(|x| { - (0..cells).map(move |y| TileCoordinate { - z: FIXTURE_LOD.max_tile_depth, - x, - y, - }) - }) - .collect() -} - -fn edges_request(tiles: Vec) -> EdgesRequest { - EdgesRequest { - tiles, - detail: EdgesDetail::Minimal, - } -} - -/// A store capability the request under test must drop unused. -/// -/// A rejection never reaches hydration and a minimal request orders none. Dropping the capability -/// is therefore part of the contract under test, and consuming it panics. -struct UntouchedStore; - -impl LocateStore for UntouchedStore { - #[expect( - clippy::panic_in_result_fn, - reason = "consuming the capability is the failure under test, and the panic is its witness" - )] - fn hydrate(self, _: LocateOrder<'_>) -> Result { - panic!("the request under test must not hydrate") - } -} - -impl EdgesStore for UntouchedStore { - #[expect( - clippy::panic_in_result_fn, - reason = "consuming the capability is the failure under test, and the panic is its witness" - )] - fn hydrate( - self, - _: &IdSlice, - ) -> Result>, DetailError> { - panic!("the request under test must not hydrate") - } -} - -/// A store answering that nothing resolves. -/// -/// Every store-derived column reads empty and every completeness flag `false`. An expectation -/// built over it therefore pins the envelope and the in-process columns without store-derived -/// content. -struct UnresolvedStore; - -impl LocateStore for UnresolvedStore { - fn hydrate(self, order: LocateOrder<'_>) -> Result { - Ok(LocateHydration { - nodes: LocateNodeHydration::empty(order.nodes.count()), - links: LocateLinkHydration::empty(order.links.len()), - }) - } -} - -impl EdgesStore for UnresolvedStore { - fn hydrate( - self, - types: &IdSlice, - ) -> Result>, DetailError> { - Ok(IdVec::from_elem(None, types.len())) - } -} - -/// Derives the qualifying edge columns for a delivered row set. -/// -/// Both-endpoint edges in ascending edge-row order. -fn qualifying_columns( - endpoints: &[[u64; 2]], - delivered: &HashSet, -) -> (Vec, Vec, Vec) { - let mut sources = Vec::new(); - let mut targets = Vec::new(); - let mut rows = Vec::new(); - for (row, &[source, target]) in endpoints.iter().enumerate() { - let source = u32::try_from(source).expect("fixture rows fit u32"); - let target = u32::try_from(target).expect("fixture rows fit u32"); - if delivered.contains(&source) && delivered.contains(&target) { - sources.push(source); - targets.push(target); - rows.push(u32::try_from(row).expect("fixture edge rows fit u32")); - } - } - - (sources, targets, rows) -} - -/// Maps a derivation's internal edge columns onto the wire's. -/// -/// Node ids encode through an independently derived codec, and edge identities come from the -/// fixture's seeding rule. Delivery order ascends by identity bytes, which for the fixture is -/// ascending internal edge row - the input order `qualifying_columns` already produces. -fn wire_columns(atlas: &Atlas, sources: &[u32], targets: &[u32], rows: &[u32]) -> EdgeColumns { - let node_codec = test_codec(atlas); - assert!(rows.is_sorted(), "the derivation supplies ascending rows"); - - EdgeColumns::pinned( - sources - .iter() - .zip(targets) - .zip(rows) - .map(|((&source, &target), &row)| { - ( - node_codec - .encode(NodeRowId::from_u32(source), atlas.node_universe()) - .get(), - node_codec - .encode(NodeRowId::from_u32(target), atlas.node_universe()) - .get(), - edge_identity_of(row), - ) - }), - ) -} - -fn expected_edges_bytes(generation: &Generation, complete: bool, edges: &EdgeColumns) -> Vec { - EdgesResponse { - generation: generation.id().digest(), - variant: 0, - complete, - edges, - trailer: None, - } - .encode() -} - -#[tokio::test] -async fn edges_full_coverage() { - let (generation, atlas) = publish("edges-full").await; - let artifacts = open_edge_artifacts(&generation); - let endpoints = artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let endpoints = endpoints.as_slice(); - - // The endpoint artifact follows the dataset stream order, which - // the derivations below lean on. - assert_eq!(endpoints.len(), FIXTURE_EDGES.len()); - for (&(_, source, target), &actual) in FIXTURE_EDGES.iter().zip(endpoints) { - assert_eq!([source, target], actual); - } - - let request = edges_request(full_grid()); - let bytes = atlas - .edges( - &request, - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the full grid should serve"); - assert_eq!(&bytes[..8], b"SALTILEE"); - - let delivered: HashSet = - (0..u32::try_from(NODES).expect("the fixture count fits u32")).collect(); - let (sources, targets, rows) = qualifying_columns(endpoints, &delivered); - assert_eq!( - rows.len(), - FIXTURE_EDGES.len(), - "every fixture edge qualifies" - ); - let columns = wire_columns(&atlas, &sources, &targets, &rows); - assert_eq!(bytes, expected_edges_bytes(&generation, true, &columns)); - - // Identical requests yield identical bytes. - assert_eq!( - atlas - .edges( - &request, - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the repeat should serve"), - bytes, - ); -} - -#[tokio::test] -async fn edges_root_visible_subgraph() { - let (generation, atlas) = publish("edges-root").await; - let artifacts = open_artifacts(&generation); - let row_ids = fixture_row_ids(&artifacts.rows); - let edge_artifacts = open_edge_artifacts(&generation); - let endpoints = edge_artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let endpoints = endpoints.as_slice(); - - // The root delivers buckets 0..=m: the head of the base order. - let head: u64 = artifacts.morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let head = usize::try_from(head).expect("fixture counts fit usize"); - let delivered: HashSet = row_ids[..head].iter().copied().collect(); - - let (sources, targets, edge_rows) = qualifying_columns(endpoints, &delivered); - let columns = wire_columns(&atlas, &sources, &targets, &edge_rows); - let root = TileCoordinate { z: 0, x: 0, y: 0 }; - let bytes = atlas - .edges( - &edges_request(vec![root]), - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the root should serve"); - assert_eq!(bytes, expected_edges_bytes(&generation, true, &columns)); - - // Listing a tile twice changes nothing: the delivered union - // deduplicates before the outgoing walk. - let doubled = atlas - .edges( - &edges_request(vec![root, root]), - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the doubled root should serve"); - assert_eq!(doubled, bytes); -} - -#[tokio::test] -async fn edges_exclude_partially_delivered_pairs() { - let (generation, atlas) = publish("edges-cross").await; - let artifacts = open_artifacts(&generation); - let row_ids = fixture_row_ids(&artifacts.rows); - let codes = artifacts.morton.codes(); - let edge_artifacts = open_edge_artifacts(&generation); - let endpoints = edge_artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let endpoints = endpoints.as_slice(); - let positions: Vec = edge_artifacts - .positions - .column::() - .expect("the position column holds base positions") - .as_raw() - .iter() - .map(|position| position.as_u32()) - .collect(); - - let depth = Depth::new(FIXTURE_LOD.max_tile_depth).expect("the deepest tile depth is valid"); - let cell_of_row = |row: u64| { - let position = positions[usize::try_from(row).expect("fixture rows fit usize")]; - MortonKey::from_bits(codes[BasePosition::from_u32(position)].get()).cell(depth) - }; - - // An edge whose endpoints occupy different deepest-zoom cells: - // its source tile alone delivers the source but not the target. - let (crossing, &[source, _]) = endpoints - .iter() - .enumerate() - .find(|&(_, &[source, target])| cell_of_row(source) != cell_of_row(target)) - .expect("the fixture spreads endpoints across deepest cells"); - let crossing = u32::try_from(crossing).expect("fixture edge rows fit u32"); - - let cell = cell_of_row(source); - let bytes = atlas - .edges( - &edges_request(vec![coordinate_of(cell)]), - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the source tile should serve"); - - // A deepest tile delivers exactly its cell's population. - let delivered: HashSet = codes - .iter() - .enumerate() - .filter(|&(_, code)| cell.contains(MortonKey::from_bits(code.get()))) - .map(|(position, _)| row_ids[position]) - .collect(); - let (sources, targets, edge_rows) = qualifying_columns(endpoints, &delivered); - assert!( - !edge_rows.contains(&crossing), - "the derivation excludes the crossing edge", - ); - let columns = wire_columns(&atlas, &sources, &targets, &edge_rows); - assert_eq!(bytes, expected_edges_bytes(&generation, true, &columns)); -} - -#[tokio::test] -async fn edges_cap_truncates_by_worse_endpoint_rank() { - let (generation, atlas) = publish("edges-cap").await; - let edge_artifacts = open_edge_artifacts(&generation); - let endpoints = edge_artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let endpoints = endpoints.as_slice(); - let ranks: Vec = edge_artifacts - .ranks - .column::() - .expect("the rank column holds importance ranks") - .as_raw() - .iter() - .map(|rank| rank.as_u32()) - .collect(); - let positions: Vec = edge_artifacts - .positions - .column::() - .expect("the position column holds base positions") - .as_raw() - .iter() - .map(|position| position.as_u32()) - .collect(); - let rank_of_row = - |row: u64| ranks[positions[usize::try_from(row).expect("fixture rows fit usize")] as usize]; - - // Under full coverage every edge qualifies; the cap keeps the - // two whose worse endpoint ranks best - ties on identity bytes, - // which for the fixture ascend with the edge row - emitted in - // ascending identity order. - let mut ranked: Vec<(u32, crate::postgres::id::ArchivedEntityId, u32)> = endpoints - .iter() - .enumerate() - .map(|(row, &[source, target])| { - let row = u32::try_from(row).expect("fixture edge rows fit u32"); - ( - rank_of_row(source).max(rank_of_row(target)), - edge_identity_of(row), - row, - ) - }) - .collect(); - ranked.sort_unstable(); - ranked.truncate(2); - let mut kept: Vec = ranked.into_iter().map(|(.., row)| row).collect(); - kept.sort_unstable(); - let sources: Vec = kept - .iter() - .map(|&row| u32::try_from(endpoints[row as usize][0]).expect("fixture rows fit u32")) - .collect(); - let targets: Vec = kept - .iter() - .map(|&row| u32::try_from(endpoints[row as usize][1]).expect("fixture rows fit u32")) - .collect(); - let columns = wire_columns(&atlas, &sources, &targets, &kept); - - let capped = EdgesLimits { - edges: 2, - ..EdgesLimits::default() - }; - let bytes = atlas - .edges( - &edges_request(full_grid()), - capped, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the capped request should serve"); - assert_eq!(bytes, expected_edges_bytes(&generation, false, &columns)); - - // A zero cap serves the honest empty truncation. - let empty = atlas - .edges( - &edges_request(full_grid()), - EdgesLimits { - edges: 0, - ..EdgesLimits::default() - }, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the zero cap should serve"); - assert_eq!( - empty, - expected_edges_bytes(&generation, false, &EdgeColumns::pinned([])) - ); -} - -#[tokio::test] -async fn edges_contract_rejections() { - let (generation, atlas) = publish("edges-rejects").await; - let root = TileCoordinate { z: 0, x: 0, y: 0 }; - - assert_matches!( - atlas.edges( - &edges_request(vec![root, root]), - EdgesLimits { - tiles: 1, - ..EdgesLimits::default() - }, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(EdgesError::Tiles { - count: 2, - maximum: 1, - }), - ); - assert_matches!( - atlas.edges( - &edges_request(vec![TileCoordinate { z: 4, x: 0, y: 0 }]), - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(EdgesError::Depth { z: 4, maximum: 3 }), - ); - assert_matches!( - atlas.edges( - &edges_request(vec![TileCoordinate { z: 2, x: 4, y: 0 }]), - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(EdgesError::Grid { z: 2, x: 4, y: 0 }), - ); - - // An empty tile list serves the honest empty response, with every - // column present-empty. - let bytes = atlas - .edges( - &edges_request(Vec::new()), - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ) - .expect("the empty request should serve"); - assert_eq!( - bytes, - expected_edges_bytes(&generation, true, &EdgeColumns::pinned([])) - ); - assert!( - section(&bytes, EDGE_IDS) - .expect("EDGE_IDS is present") - .is_empty(), - ); -} - -/// A detail request hydrates through its store capability and encodes the interned trailer. -/// -/// The one-call path answers byte-exactly against the directly built wire document. Labels -/// resolve in process from the generation's payloads - empty under the fixture, whose identity -/// rewrite persists empty payloads - and the store capability answers the type column, all-`null` -/// here (G6 pins the non-null trailer bytes). -#[tokio::test] -async fn detailed_edges_trailer() { - use crate::salt::wire::edges::EdgesTrailer; - - let (generation, atlas) = publish("detailed-edges").await; - let artifacts = open_artifacts(&generation); - let row_ids = fixture_row_ids(&artifacts.rows); - let edge_artifacts = open_edge_artifacts(&generation); - let endpoints = edge_artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let endpoints = endpoints.as_slice(); - - let root = TileCoordinate { z: 0, x: 0, y: 0 }; - let mut request = edges_request(vec![root]); - request.detail = EdgesDetail::Auxiliary; - - let bytes = atlas - .edges( - &request, - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UnresolvedStore, - ) - .expect("the detail request should serve"); - - let head: u64 = artifacts.morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let head = usize::try_from(head).expect("fixture counts fit usize"); - let delivered: HashSet = row_ids[..head].iter().copied().collect(); - let (sources, targets, internal_edges) = qualifying_columns(endpoints, &delivered); - let columns = wire_columns(&atlas, &sources, &targets, &internal_edges); - - let no_labels: Vec<&Label> = vec![Label::EMPTY; columns.count()]; - let no_types: Vec>> = vec![None; columns.count()]; - let expected = EdgesResponse { - generation: generation.id().digest(), - variant: 0, - complete: true, - edges: &columns, - trailer: Some(EdgesTrailer { - type_table: IdSlice::from_raw(&[]), - link_labels: IdSlice::from_raw(&no_labels), - link_type_ids: IdSlice::from_raw(&no_types), - }), - } - .encode(); - assert_eq!(bytes, expected, "the trailer rides the pinned envelope"); -} - -/// Derives the type-resolution expectations from the published artifacts alone. -/// -/// Each delivered edge's legend payload names its representative ontology row, the rewritten -/// ontology table pins that row's uuid to a fixture URL, and the returned resolution map answers -/// a synthetic URL per uuid. Returns that map, the per-edge expected URLs, and the distinct -/// uuids in first-occurrence order over the delivered edges. -fn type_expectations( - generation: &Generation, - internal_edges: &[u32], -) -> ( - hashql_core::collections::FastHashMap< - crate::postgres::id::ArchivedOntologyTypeUuid, - VersionedUrl, - >, - Vec>, - Vec, -) { - use crate::postgres::id::ArchivedOntologyTypeUuid; - - let files = &generation.repository().files; - let ontology_rows = payloads_of(&generation.path_of(&files.ontology_identities.name())).len(); - let uuid_of = |row: usize| { - let url: VersionedUrl = fixture_type_url(row as u64) - .parse() - .expect("the fixture URL parses"); - ArchivedOntologyTypeUuid::from_url(&url) - }; - let resolved_of = |row: usize| -> VersionedUrl { - format!("https://example.com/types/resolved-{row}/v/1") - .parse() - .expect("the synthetic URL parses") - }; - let urls = (0..ontology_rows) - .map(|row| (uuid_of(row), resolved_of(row))) - .collect(); - - let edge_payloads = payloads_of(&generation.path_of(&files.edge_identities.name())); - let representative_of = |edge: u32| { - let payload = &edge_payloads[usize::try_from(edge).expect("fixture rows fit usize")]; - let representative = u64::from_le_bytes( - payload[..8] - .try_into() - .expect("a legend leads with its representative"), - ); - usize::try_from(representative).expect("fixture rows fit usize") - }; - let expected_urls = internal_edges - .iter() - .map(|&edge| { - let representative = representative_of(edge); - (representative < ontology_rows).then(|| resolved_of(representative)) - }) - .collect(); - let expected_asked = { - let mut seen: Vec = Vec::new(); - for &edge in internal_edges { - let uuid = uuid_of(representative_of(edge)); - if !seen.contains(&uuid) { - seen.push(uuid); - } - } - seen - }; - - (urls, expected_urls, expected_asked) -} - -/// The trailer's type column resolves each edge's legend representative through the store. -/// -/// Expected values derive from the published artifacts alone, through [`type_expectations`]. -/// The store receives the distinct uuids in first-occurrence order over the delivered edges, -/// which is the deduplication this path exists to buy. -#[tokio::test] -async fn detailed_edges_types() { - use alloc::rc::Rc; - use core::cell::RefCell; - - use hashql_core::collections::FastHashMap; - - use crate::{postgres::id::ArchivedOntologyTypeUuid, salt::wire::edges::EdgesTrailer}; - - /// Answers `urls` per known uuid and records every uuid set that reaches it. - struct RecordingTypeStore { - urls: FastHashMap, - asked: Rc>>, - } - - impl EdgesStore for RecordingTypeStore { - fn hydrate( - self, - types: &IdSlice, - ) -> Result>, DetailError> { - self.asked.borrow_mut().extend(types.iter().copied()); - - Ok(types - .iter() - .map(|uuid| self.urls.get(uuid).cloned()) - .collect()) - } - } - - let (generation, atlas) = publish("edge-type-resolution").await; - let artifacts = open_artifacts(&generation); - let row_ids = fixture_row_ids(&artifacts.rows); - let edge_artifacts = open_edge_artifacts(&generation); - let endpoints = edge_artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - - // The delivered edge set, exactly as the pinned-envelope test derives it. - let head: u64 = artifacts.morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let head = usize::try_from(head).expect("fixture counts fit usize"); - let delivered: HashSet = row_ids[..head].iter().copied().collect(); - let (sources, targets, internal_edges) = qualifying_columns(&endpoints, &delivered); - let columns = wire_columns(&atlas, &sources, &targets, &internal_edges); - - let (urls, expected_urls, expected_asked) = type_expectations(&generation, &internal_edges); - - let root = TileCoordinate { z: 0, x: 0, y: 0 }; - let mut request = edges_request(vec![root]); - request.detail = EdgesDetail::Auxiliary; - - let asked = Rc::new(RefCell::new(Vec::new())); - let bytes = atlas - .edges( - &request, - EdgesLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - RecordingTypeStore { - urls, - asked: Rc::clone(&asked), - }, - ) - .expect("the detail request should serve"); - - assert_eq!( - *asked.borrow(), - expected_asked, - "the order carries the distinct uuids in first-occurrence order" - ); - - let table = super::intern::Table::new(expected_urls.iter().flatten()); - let link_type_ids: Vec>> = expected_urls - .iter() - .map(|url| url.as_ref().map(|url| table.index_of(url))) - .collect(); - let no_labels: Vec<&Label> = vec![Label::EMPTY; columns.count()]; - let expected = EdgesResponse { - generation: generation.id().digest(), - variant: 0, - complete: true, - edges: &columns, - trailer: Some(EdgesTrailer { - type_table: table.entries(), - link_labels: IdSlice::from_raw(&no_labels), - link_type_ids: IdSlice::from_raw(&link_type_ids), - }), - } - .encode(); - assert_eq!( - bytes, expected, - "each edge's type id points at its representative's resolved URL" - ); -} - -/// Source resolution answers the delivery contract rather than a formula alone. -/// -/// The resolved (zoom, cell) tile delivers the row under the cumulative schedule, and at zoom > 0 -/// the parent tile's schedule does not. `zoom` is therefore the first visible zoom. -#[tokio::test] -async fn locate_first_visible_tile() { - let (_generation, atlas) = publish("locate-resolve").await; - let node_codec = test_codec(&atlas); - let wire_of = |row: u32| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }; - let bound = Bound::of(&atlas, &FULL); - let view = bound.view(&atlas); - - let row_of = |bytes: &[u8]| { - let rows = section(bytes, ROW_IDS).expect("ROW_IDS is present"); - rows.as_chunks::<4>() - .0 - .iter() - .map(|&chunk| u32::from_le_bytes(chunk)) - .collect::>() - }; - - let mut resolved = 0; - for row in 0..4_u8 { - let Some(source) = atlas.resolve_source(&view, &entity_string_of(row)) else { - panic!("fixture node ids resolve"); - }; - let SourceSubject::Base { - row: source_row, - position: source_position, - } = source.subject - else { - panic!("a fitted fixture source resolves in the base domain"); - }; - assert_eq!(source_row.get(), NodeRowId::from_u32(u32::from(row))); - assert_eq!(source_position, atlas.positions_of_row()[source_row.get()]); - - // The resolved tile delivers the row. - let request = TileRequest { - coordinate: source.cell, - query: TileQuery { - mode: Mode::Total, - ..TileQuery::default() - }, - }; - let bytes = atlas - .tile( - &request, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the resolved tile serves"); - assert!( - row_of(&bytes).contains(&wire_of(source_row.get().as_u32())), - "the resolved tile delivers its source", - ); - - // The parent's cumulative schedule does not: zoom is first. - if source.zoom > 0 { - let parent = TileRequest { - coordinate: TileCoordinate { - z: source.zoom - 1, - // The parent tile halves each grid index: one - // right-shift, the quadtree's own arithmetic. - x: source.cell.x >> 1_u32, - y: source.cell.y >> 1_u32, - }, - query: TileQuery { - mode: Mode::Total, - ..TileQuery::default() - }, - }; - let bytes = atlas - .tile( - &parent, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("the parent tile serves"); - assert!( - !row_of(&bytes).contains(&wire_of(source_row.get().as_u32())), - "the parent zoom does not deliver the source yet", - ); - resolved += 1; - } - } - // The fixture spreads buckets, so at least one probed row sits - // below the root cut and exercises the parent assertion. - assert!(resolved > 0, "at least one source resolves below zoom 0"); - - // Non-node shapes read absent: an edge id, an unknown id, junk. - assert_eq!( - atlas.resolve_source(&view, &entity_string_of(EDGE_SEED)), - None - ); - assert_eq!( - atlas.resolve_source( - &view, - &format!("{}~{}", uuid::Uuid::nil(), uuid::Uuid::nil()) - ), - None, - ); - assert_eq!(atlas.resolve_source(&view, "not an id"), None); -} - -/// The locate delivered set answers the wire pin over hand-derived fixture ego-graphs. -/// -/// Source first, then the delivered edges' partners ascending wire row id; edges are the source's -/// incident set - both directions, a self-loop exactly once - ascending link-entity identity -/// bytes (for the fixture, ascending edge row). -#[tokio::test] -async fn locate_ego_graph() { - use super::locate::LocateLimits; - - let (_generation, atlas) = publish("locate-ego").await; - let node_codec = test_codec(&atlas); - let bound = Bound::of(&atlas, &FULL); - let view = bound.view(&atlas); - - // The fixture edge list by row: 0 = (0 → 1), 1 = (1 → 2), - // 2 = (2 → 2) self-loop, 3 = (5 → 40), 4 = (40 → 5), - // 5 = (3 → 7). Hand-derived ego-graphs, (source, partners, - // edge rows): - // ego(0) = partner 1 over edge 0; - // ego(1) = partners {0, 2} over edges {0, 1}; - // ego(2) = partner 1 over edges {1, 2} - the self-loop - // delivers exactly once and adds no partner; - // ego(4) = alone, zero edges - the honest-empty case; - // ego(5) = partner 40 over edges {3, 4} - the reciprocal - // pair shares one partner, delivered once. - let cases: [(u8, &[u32], &[u32]); 5] = [ - (0, &[1], &[0]), - (1, &[0, 2], &[0, 1]), - (2, &[1], &[1, 2]), - (4, &[], &[]), - (5, &[40], &[3, 4]), - ]; - - for (source_row, partners, edge_rows) in cases { - let source = atlas - .resolve_source(&view, &entity_string_of(source_row)) - .expect("fixture node ids resolve"); - let subgraph = atlas.locate_subgraph(source, LocateLimits::default(), &view); - assert!(subgraph.complete, "ego({source_row}) is under the cap"); - - // Partners deliver ascending by wire row id; the expectation - // recomputes the order through an independently constructed - // codec. - let mut expected_rows: Vec = partners.to_vec(); - expected_rows.sort_unstable_by_key(|&row| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }); - expected_rows.insert(0, u32::from(source_row)); - let delivered_rows: Vec = delivered_row_ids(&atlas, &subgraph) - .iter() - .map(|row| row.as_u32()) - .collect(); - assert_eq!(delivered_rows, expected_rows, "ego({source_row}) rows"); - let expected_vessels: Vec = expected_rows - .iter() - .map(|&row| ViewRow::Base(atlas.positions_of_row()[NodeRowId::from_u32(row)])) - .collect(); - assert_eq!( - subgraph.delivered.as_raw(), - expected_vessels, - "ego({source_row}) positions", - ); - - // Edges deliver ascending by identity bytes - for the - // fixture, ascending edge row - endpoints straight off the - // fixture edge list. - let delivered: Vec = subgraph - .edges - .iter() - .map(|&(edge, _)| narrow_usize(fitted(edge).row.get().as_usize())) - .collect(); - assert_eq!(delivered, edge_rows, "ego({source_row}) edges"); - for &(edge, id) in &subgraph.edges { - let edge = fitted(edge); - let (_, edge_source, edge_target) = FIXTURE_EDGES[edge.row.get().as_usize()]; - assert_eq!(edge.source.as_u64(), edge_source); - assert_eq!(edge.target.as_u64(), edge_target); - assert_eq!( - id, - edge_identity_of(narrow_usize(edge.row.get().as_usize())) - ); - } - } -} - -/// The squared wire-frame distance between two fitted rows, as selection-key bits. -/// -/// The derivation must mirror the selection key bit for bit, because a fused `mul_add` rounds -/// differently and reorders near-ties. -fn wire_distance_bits(atlas: &Atlas, from: u32, to: u32) -> u32 { - let positions = atlas.positions(); - let origin = positions[atlas.positions_of_row()[NodeRowId::from_u32(from)]]; - let point = positions[atlas.positions_of_row()[NodeRowId::from_u32(to)]]; - let (dx, dy) = (point.x() - origin.x(), point.y() - origin.y()); - #[expect( - clippy::suboptimal_flops, - reason = "unfused arithmetic mirrors the selection key exactly" - )] - (dx * dx + dy * dy).to_bits() -} - -/// The locate edge cap keeps the nearest partner, proven by hand on the self-loop. -/// -/// Row 2's self-loop partner is the source itself at distance zero. It therefore survives every -/// nonzero cap, and partner 1's only edge truncates - the partner leaves with its edge and the -/// source stands alone. -#[tokio::test] -async fn locate_cap_self_loop() { - use super::locate::LocateLimits; - - let (_generation, atlas) = publish("locate-truncation").await; - let bound = Bound::of(&atlas, &FULL); - let view = bound.view(&atlas); - - // The rows occupy distinct wire coordinates, which the case asserts so that the hand - // derivation cannot degenerate into an unnoticed tie. - assert_ne!( - wire_distance_bits(&atlas, 2, 1), - 0, - "rows 1 and 2 are not co-located" - ); - let source = atlas - .resolve_source(&view, &entity_string_of(2)) - .expect("fixture node ids resolve"); - let capped = atlas.locate_subgraph( - source, - LocateLimits { - edges: 1, - ..LocateLimits::default() - }, - &view, - ); - assert!(!capped.complete, "one of two incident edges truncated"); - assert_eq!( - capped - .edges - .iter() - .map(|&(edge, _)| narrow_usize(fitted(edge).row.get().as_usize())) - .collect::>(), - [2], - "the self-loop is the nearest edge", - ); - // Partner 1's only edge truncated, so partner 1 is not delivered: the source stands alone. - assert_eq!(delivered_row_ids(&atlas, &capped), [NodeRowId::new(2)]); - assert_eq!( - capped.delivered.as_raw(), - [ViewRow::Base(atlas.positions_of_row()[NodeRowId::new(2)])] - ); -} - -/// The locate edge cap's survivors match an independent key derivation at every cap. -/// -/// The selection key is ascending (squared wire-frame distance to the partner, partner -/// first-visible zoom, link-entity identity bytes). Presentation stays ascending identity -/// bytes, the delivered nodes are exactly the survivors' partners, and assembly repeats -/// identically. -#[tokio::test] -async fn locate_cap_vs_independent_key() { - use super::locate::LocateLimits; - - let (_generation, atlas) = publish("locate-truncation-sweep").await; - let node_codec = test_codec(&atlas); - let accepted = atlas.node_universe(); - let bound = Bound::of(&atlas, &FULL); - let view = bound.view(&atlas); - - for source_row in [0_u8, 1, 2, 3, 4, 5, 7, 40] { - let source = atlas - .resolve_source(&view, &entity_string_of(source_row)) - .expect("fixture node ids resolve"); - let full = atlas.locate_subgraph(source, LocateLimits::default(), &view); - - for cap in 0..=full.edges.len() { - let limits = LocateLimits { - edges: u32::try_from(cap).expect("fixture edge counts are small"), - ..LocateLimits::default() - }; - let subgraph = atlas.locate_subgraph(source, limits, &view); - assert_eq!( - subgraph.complete, - full.edges.len() <= cap, - "ego({source_row}) cap {cap}", - ); - - // The independent key runs distance bits, then the partner's first visible zoom - // through the public resolve path (the HEAD fly-to derivation), then the identity - // bytes. - let mut expected = full.edges.clone(); - expected.sort_unstable_by_key(|&(edge, id)| { - let edge = fitted(edge); - let partner = if edge.source.as_u32() == u32::from(source_row) { - edge.target.as_u32() - } else { - edge.source.as_u32() - }; - let zoom = atlas - .resolve_source( - &view, - &entity_string_of(u8::try_from(partner).expect("fixture rows fit u8")), - ) - .expect("fixture partners resolve") - .zoom; - ( - wire_distance_bits(&atlas, u32::from(source_row), partner), - zoom, - id, - ) - }); - expected.truncate(cap); - expected.sort_unstable_by_key(|&(_, id)| id); - assert_eq!(subgraph.edges, expected, "ego({source_row}) cap {cap}"); - - let mut partner_keys: Vec<(u32, u32)> = expected - .iter() - .flat_map(|&(edge, _)| { - let edge = fitted(edge); - - [edge.source.as_u32(), edge.target.as_u32()] - }) - .filter(|&row| row != u32::from(source_row)) - .map(|row| { - ( - node_codec.encode(NodeRowId::from_u32(row), accepted).get(), - row, - ) - }) - .collect(); - partner_keys.sort_unstable(); - partner_keys.dedup(); - let mut expected_rows = vec![u32::from(source_row)]; - expected_rows.extend(partner_keys.iter().map(|&(_, row)| row)); - let delivered_rows: Vec = delivered_row_ids(&atlas, &subgraph) - .iter() - .map(|row| row.as_u32()) - .collect(); - assert_eq!(delivered_rows, expected_rows, "ego({source_row}) cap {cap}"); - - // Determinism pair: identical assembly on repeat. - assert_eq!( - subgraph, - atlas.locate_subgraph(source, limits, &view), - "ego({source_row}) cap {cap}", - ); - } - } -} - -#[test] -fn edges_body_contract() { - let request: EdgesRequest = - serde_json::from_str(r#"{ "tiles": [{ "z": 1, "x": 0, "y": 1 }] }"#) - .expect("the minimal body parses"); - assert_eq!(request.tiles, vec![TileCoordinate { z: 1, x: 0, y: 1 }]); - assert_eq!(request.detail, EdgesDetail::Minimal); - - let request: EdgesRequest = serde_json::from_str( - r#"{ - "tiles": [], - "detail": "auxiliary" - }"#, - ) - .expect("the full body parses"); - assert!(request.tiles.is_empty()); - assert_eq!(request.detail, EdgesDetail::Auxiliary); -} - -/// A colored request mixing resolvable and unresolvable ids over the published fixture. -/// -/// `TYPE_MASK` rides the request at full shape, unresolvable ids read 0 in every mask, and a -/// fixture type URL resolves to real bits. -#[tokio::test] -async fn colored_types_zero_unknowns() { - let (_generation, atlas) = publish("colored-masks-e2e").await; - - let mut colored = request(0, 0, 0, Mode::Delta); - colored.query.colored_type_ids = vec![ - fixture_type_url(0).parse().expect("fixture urls parse"), - "https://example.com/types/unknown/v/1" - .parse() - .expect("the literal is a versioned url"), - "https://example.com/types/unknown/v/2" - .parse() - .expect("the literal is a versioned url"), - ]; - let bytes = atlas - .tile( - &colored, - TileLimits::default(), - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - ) - .expect("a colored request serves"); - - let rows = section(&bytes, ROW_IDS).expect("ROW_IDS is present"); - let mask = section(&bytes, TYPE_MASK).expect("TYPE_MASK rides colored requests"); - // With three requested ids the stride is ceil(3/8) = 1 byte per - // point, so four row-id bytes stand behind every mask byte. - assert_eq!(mask.len() * 4, rows.len()); - assert!( - mask.iter().all(|&byte| byte & !0b1 == 0), - "unresolvable ids read 0 in every mask", - ); - assert!( - mask.iter().any(|&byte| byte & 0b1 != 0), - "the fixture type URL resolves to real membership bits", - ); -} - -/// A generation carrying the memory dataset's positional ids does not serve. -/// -/// The open fails on the key kind. -#[tokio::test] -async fn foreign_key_kinds_fail_open() { - use super::error::{IdentityDomain, OpenAtlasError}; - use crate::salt::fit::prepare::identity::InvalidIdentityFile; - - let (root, generation) = fit_fixture("foreign-kind-fails").await; - let error = Atlas::open(&root, generation.id(), test_open_options()) - .expect_err("a foreign key kind must fail the open"); - assert!( - matches!( - error, - OpenAtlasError::Identity { - domain: IdentityDomain::Ontology, - error: InvalidIdentityFile::KeyKind { .. }, - }, - ), - "the open names the first identity table it rejects: {error}", - ); -} - -/// Resolution and descendant expansion against hand-built artifacts. -/// -/// Eight points, four types (`1 <- 0`, `2 <- 0`, `3 <- {1, 2}`), the postings fixture's dense/list -/// split. -#[test] -fn colored_masks_expand_descendants() { - use type_system::ontology::id::VersionedUrl; - - use super::colour; - use crate::{ - file::{identity::read::IdentityFile, postings::read::PostingsFile}, - postgres::id::ArchivedOntologyTypeUuid, - salt::{ - fit::prepare::identity::{IdentityTable, IdentityTableArchive}, - postings::{artifact::PostingsArchive, build::Postings, closure::ClosureMap}, - }, - }; - - let dir = scratch("colored-masks"); - std::fs::create_dir_all(&dir).expect("the scratch directory creates"); - - // Row-order direct types and the gather permutation, copied from - // the postings fixture; member positions per type, hand-derived: - // type 0 [1, 2, 3, 6], type 1 [5, 7], type 2 [0, 1, 7], type 3 []. - let types: IdVec> = - [&[0_u64][..], &[0, 2], &[1], &[2], &[0], &[1, 2], &[], &[0]] - .iter() - .map(|list| list.iter().copied().map(OntologyRowId::new).collect()) - .collect(); - let parents: IdVec> = - [&[][..], &[0_u64], &[0], &[1, 2]] - .iter() - .map(|list| list.iter().copied().map(OntologyRowId::new).collect()) - .collect(); - let row_of_position: [u32; 8] = [3, 1, 4, 0, 6, 2, 7, 5]; - let row_of_position = row_of_position.map(NodeRowId::from_u32); - - let postings = Postings::build(&types, IdSlice::from_raw(&row_of_position), &parents) - .expect("the fixture stays in domain"); - let postings_path = dir.join("fixture.post"); - let mut file = std::fs::File::create(&postings_path).expect("the postings file creates"); - postings - .write_into(&mut file) - .expect("the postings should write"); - drop(file); - let postings = - PostingsArchive::new(PostingsFile::open(&postings_path).expect("the postings file opens")) - .expect("the postings validate"); - let closure = ClosureMap::new(&postings, []).expect("the parent graph is acyclic"); - - // One versioned URL per type row; the table keys each row by the - // uuid its URL derives, exactly as the store's identities would. - let urls: Vec = (0..4) - .map(|row| format!("https://example.com/types/fixture-{row}/v/1")) - .collect(); - let mut table = IdentityTable::::new(); - for url in &urls { - let parsed: VersionedUrl = url.parse().expect("the fixture URL parses"); - table.push(ArchivedOntologyTypeUuid::from_url(&parsed)); - } - let identity_path = dir.join("fixture.idnt"); - let mut file = std::fs::File::create(&identity_path).expect("the identity file creates"); - let rows = usize::try_from(table.len()).expect("fixture row counts fit the address space"); - let empty = - ::try_ref_from_bytes(&[]) - .expect("every payload type admits the empty byte string"); - let _digest = table - .write_into(core::iter::repeat_n(empty, rows), &mut file) - .expect("the identities should write"); - drop(file); - let table = IdentityTableArchive::::new( - IdentityFile::open(&identity_path).expect("the identity file opens"), - ) - .expect("the identity table validates"); - - let members = |ids: &[String]| -> Vec> { - let urls: Vec = ids - .iter() - .map(|id| id.parse().expect("test urls parse")) - .collect(); - - let set = colour::resolve_masks(&postings, &closure, &table, &colour::Palette::of(&urls)); - set.memberships(&postings) - .iter() - .map(|membership| { - membership - .positions_in(BasePosition::from_u32(0)..BasePosition::from_u32(8)) - .map(BasePosition::as_u32) - .collect() - }) - .collect() - }; - - // Type 0's descendants are every type: the union covers every - // typed position. Type 1 folds type 3's empty membership in. - // Type 3 has no proper descendant and serves its stored (empty) - // membership. A URL this generation never ingested reads empty. - assert_eq!(members(&[urls[0].clone()]), vec![vec![0, 1, 2, 3, 5, 6, 7]],); - assert_eq!( - members(&[ - urls[1].clone(), - urls[2].clone(), - urls[3].clone(), - "https://example.com/types/unknown/v/1".to_owned(), - ]), - vec![vec![5, 7], vec![0, 1, 7], vec![], vec![]], - ); -} - -/// One synthetic entity identity per seed byte, plus its upstream string form. -fn entity_id_of(seed: u8) -> crate::postgres::id::ArchivedEntityId { - crate::postgres::id::ArchivedEntityId { - web_id: uuid::Uuid::from_bytes([seed; 16]).into(), - entity_uuid: uuid::Uuid::from_bytes([seed ^ 0xFF; 16]).into(), - } -} - -/// The `webId~entityUuid` string form of [`entity_id_of`]'s identity. -/// -/// Narrows a fixture-sized index into the wire's `u32` row domain. -pub(crate) fn narrow_usize(value: usize) -> u32 { - u32::try_from(value).expect("fixture indexes fit u32") -} - -/// Derives the node wire codec of an atlas opened with the suite's secret. -/// -/// The independent derivation the assembly's egress must agree with. -pub(crate) fn test_codec(atlas: &Atlas) -> codec::RowCodec { - codec::RowCodec::derive( - &WireSecret::new(TEST_WIRE_SECRET), - atlas.generation(), - codec::NODE_LABEL, - ) -} - -/// Derives a fixture edge row's link-entity identity from the seeding rule. -/// -/// Identity bytes ascend with the edge row, because `entity_id_of` leads with its seed byte. -/// Ascending internal row order is therefore the wire's ascending-identity delivery order for the -/// fixture. -fn edge_identity_of(row: u32) -> crate::postgres::id::ArchivedEntityId { - entity_id_of(EDGE_SEED + u8::try_from(row).expect("fixture edge rows fit u8")) -} - -fn entity_string_of(seed: u8) -> String { - format!( - "{}~{}", - uuid::Uuid::from_bytes([seed; 16]), - uuid::Uuid::from_bytes([seed ^ 0xFF; 16]), - ) -} - -/// Writes and reopens one hand-built entity identity table. -fn entity_identity_table( - path: &camino::Utf8PathBuf, - ids: &[crate::postgres::id::ArchivedEntityId], -) -> crate::salt::fit::prepare::identity::IdentityTableArchive< - crate::postgres::id::ArchivedEntityId, - R, -> { - use crate::{ - file::identity::read::IdentityFile, postgres::id::ArchivedEntityId, - salt::fit::prepare::identity::IdentityTable, - }; - - let mut table = IdentityTable::::new(); - for &id in ids { - table.push(id); - } - let mut file = std::fs::File::create(path).expect("the identity file creates"); - let empty = crate::dataset::auxiliary::OwnedLegend::new( - crate::identity::OntologyRowId::new(0), - Label::EMPTY, - ); - let _digest = table - .write_into(core::iter::repeat_n(empty.as_ref(), ids.len()), &mut file) - .expect("the identities should write"); - drop(file); - crate::salt::fit::prepare::identity::IdentityTableArchive::new( - IdentityFile::open(path).expect("the identity file opens"), - ) - .expect("the identity table validates") -} - -/// Translate resolution against hand-built identity tables. -/// -/// Nodes answer row and wire position, edges answer their endpoints' node rows, and every -/// non-resolving shape - draft-suffixed, unparsable, unknown - reads as an absent key. -#[test] -fn translate_by_identity() { - use super::translate::{ - TranslateColumns, TranslateLimits, TranslateRequest, TranslatedEdge, TranslatedNode, - translate, - }; - use crate::math::Vec2; - - let dir = scratch("translate-identity"); - std::fs::create_dir_all(&dir).expect("the scratch directory creates"); - - // Three nodes, two edges. Node row 1 sits at base position 2. - let nodes = entity_identity_table( - &dir.join("nodes.idnt"), - &[entity_id_of(1), entity_id_of(2), entity_id_of(3)], - ); - let edges = entity_identity_table( - &dir.join("edges.idnt"), - &[entity_id_of(10), entity_id_of(11)], - ); - let positions = [ - Vec2::new(0.0, 0.5), - Vec2::new(-0.25, 1.0), - Vec2::new(0.75, -0.5), - ]; - let position_of_row = [1_u32, 2, 0].map(BasePosition::from_u32); - - let request = TranslateRequest { - entity_ids: vec![ - entity_string_of(2), // node row 1 - entity_string_of(10), // edge row 0 - entity_string_of(2), // duplicate: collapses - format!("{}~draft-tail", entity_string_of(3)), // draft-suffixed: absent - "not an entity id".to_owned(), // unparsable: absent - entity_string_of(0xAB), // unknown: absent - ], - }; - // The table's universe is three node rows, and the expectations - // below encode through the same derivation. - let universe = codec::Universe::new(NodeRowId::new(3)); - let node_codec = codec::RowCodec::derive( - &WireSecret::new(TEST_WIRE_SECRET), - codec_generation(), - codec::NODE_LABEL, - ); - // Edge row 0 joins nodes 0 and 1, edge row 1 joins 1 and 2: - // arbitrary but in-universe, visible under the full proof. - let endpoints = [ - [NodeRowId::new(0), NodeRowId::new(1)], - [NodeRowId::new(1), NodeRowId::new(2)], - ]; - let response = translate( - request, - TranslateLimits::default(), - &FULL, - None, - crate::serve::delta::PlacementCohort::EMPTY, - &TranslateColumns { - node_ids: &nodes, - edge_ids: &edges, - positions: IdSlice::from_raw(&positions), - position_of_row: IdSlice::from_raw(&position_of_row), - endpoints: IdSlice::from_raw(&endpoints), - node_codec: &node_codec, - universe, - fitted: universe, - }, - ) - .expect("the request is under the cap"); - - assert_eq!( - response.nodes.into_iter().collect::>(), - vec![( - entity_string_of(2), - TranslatedNode { - id: node_codec.encode(NodeRowId::new(1), universe), - x: 0.75, - y: -0.5, - }, - )], - ); - assert_eq!( - response.edges.into_iter().collect::>(), - vec![( - entity_string_of(10), - TranslatedEdge { - source: node_codec.encode(NodeRowId::new(0), universe), - target: node_codec.encode(NodeRowId::new(1), universe), - }, - )], - ); -} - -/// The cap rejects by count before any id lookup. -#[test] -fn translate_rejects_over_cap() { - use super::translate::{ - TranslateColumns, TranslateError, TranslateLimits, TranslateRequest, translate, - }; - - let dir = scratch("translate-cap"); - std::fs::create_dir_all(&dir).expect("the scratch directory creates"); - let nodes = entity_identity_table(&dir.join("nodes.idnt"), &[entity_id_of(1)]); - let edges = entity_identity_table(&dir.join("edges.idnt"), &[entity_id_of(10)]); - - let over = TranslateRequest { - entity_ids: vec![String::new(); TranslateLimits::default().entity_ids as usize + 1], - }; - let node_codec = codec::RowCodec::derive( - &WireSecret::new(TEST_WIRE_SECRET), - codec_generation(), - codec::NODE_LABEL, - ); - assert_eq!( - translate( - over, - TranslateLimits::default(), - &FULL, - None, - crate::serve::delta::PlacementCohort::EMPTY, - &TranslateColumns { - node_ids: &nodes, - edge_ids: &edges, - positions: IdSlice::from_raw(&[]), - position_of_row: IdSlice::from_raw(&[]), - endpoints: IdSlice::from_raw(&[]), - node_codec: &node_codec, - universe: codec::Universe::new(NodeRowId::new(1)), - fitted: codec::Universe::new(NodeRowId::new(1)), - }, - ), - Err(TranslateError::Ids { - count: TranslateLimits::default().entity_ids as usize + 1, - maximum: TranslateLimits::default().entity_ids, - }), - ); -} - -/// Translate over the published fixture. -/// -/// The rewritten store-width identities resolve end to end. Node row and wire position agree with -/// the serving columns. An edge id answers its row, and an unknown id reads absent. -#[tokio::test] -async fn translate_store_identities() { - use super::translate::{TranslateLimits, TranslateRequest, TranslatedEdge, TranslatedNode}; - - let (_generation, atlas) = publish("translate-e2e").await; - - let response = atlas - .translate( - TranslateRequest { - entity_ids: vec![ - entity_string_of(0), - entity_string_of(EDGE_SEED), - format!("{}~{}", uuid::Uuid::nil(), uuid::Uuid::nil()), - ], - }, - TranslateLimits::default(), - &FULL, - None, - crate::serve::delta::PlacementCohort::EMPTY, - ) - .expect("the request is under the cap"); - - let position = atlas.positions_of_row()[NodeRowId::new(0)]; - let point = atlas.positions()[position]; - let node_codec = test_codec(&atlas); - assert_eq!( - response.nodes.into_iter().collect::>(), - vec![( - entity_string_of(0), - TranslatedNode { - id: node_codec.encode(NodeRowId::new(0), atlas.node_universe()), - x: point.x(), - y: point.y(), - }, - )], - ); - // Fixture edge row 0 joins node rows 0 and 1. - assert_eq!( - response.edges.into_iter().collect::>(), - vec![( - entity_string_of(EDGE_SEED), - TranslatedEdge { - source: node_codec.encode(NodeRowId::new(0), atlas.node_universe()), - target: node_codec.encode(NodeRowId::new(1), atlas.node_universe()), - }, - )], - ); -} - -/// Shorthand for a null-valued property entry. -fn property(name: &str) -> (BaseUrl, super::hydrate::ScalarValue) { - ( - BaseUrl::new(name.to_owned()).expect("fixture keys are base URLs"), - super::hydrate::ScalarValue::Null, - ) -} - -#[test] -fn scalar_shapes_parse() { - use super::hydrate::ScalarValue; - - // The store renders 2.5 and 1.0 with their points. Both therefore - // read as doubles; a number beyond i64 falls back to f64 (the wire's - // integer is i64 - the scalar-value shapes carry no wider integral form). - let object = serde_json::from_str( - r#"{ - "https://x.test/f/": 2.5, - "https://x.test/g/": 1.0, - "https://x.test/i/": 7, - "https://x.test/j/": -3, - "https://x.test/n/": null, - "https://x.test/t/": "text", - "https://x.test/u/": 18446744073709551615, - "https://x.test/y/": true - }"#, - ) - .expect("the fixture is JSON"); - let mut entries = super::hydrate::select::scalar_properties(object); - entries.sort_by(|left, right| left.0.cmp(&right.0)); - - let expected = [ - ("https://x.test/f/", ScalarValue::Float(2.5)), - ("https://x.test/g/", ScalarValue::Float(1.0)), - ("https://x.test/i/", ScalarValue::Integer(7)), - ("https://x.test/j/", ScalarValue::Integer(-3)), - ("https://x.test/n/", ScalarValue::Null), - ("https://x.test/t/", ScalarValue::String("text".to_owned())), - // u64::MAX itself is not an f64. The fallback therefore rounds to - // the nearest double. - ( - "https://x.test/u/", - ScalarValue::Float(1.844_674_407_370_955_2e19), - ), - ("https://x.test/y/", ScalarValue::Bool(true)), - ]; - assert_eq!( - entries, - expected - .into_iter() - .map(|(name, value)| { - ( - BaseUrl::new(name.to_owned()).expect("fixture keys are base URLs"), - value, - ) - }) - .collect::>(), - ); -} - -#[test] -#[should_panic(expected = "the store aggregates a JSON object")] -fn scalar_properties_reject_non_objects() { - let _entries = super::hydrate::select::scalar_properties(serde_json::json!([1, 2])); -} - -#[test] -fn scalar_properties_skip_non_url_keys() { - let entries = super::hydrate::select::scalar_properties( - serde_json::json!({"not a url": null, "https://x.test/a/": true}), - ); - - // The malformed key skips its entry alone. The well-keyed sibling survives. - assert_eq!( - entries, - vec![( - BaseUrl::new("https://x.test/a/".to_owned()).expect("fixture keys are base URLs"), - super::hydrate::ScalarValue::Bool(true), - )], - ); -} - -#[test] -fn select_properties_drop_reverse_lexicographically() { - let entries = vec![ - property("https://x.test/b/"), - property("https://x.test/d/"), - property("https://x.test/a/"), - property("https://x.test/c/"), - ]; - - // Under the cap: nothing drops, output ascends by name. - assert_eq!( - super::hydrate::select::select_properties(entries.clone(), None, 4), - vec![ - property("https://x.test/a/"), - property("https://x.test/b/"), - property("https://x.test/c/"), - property("https://x.test/d/") - ], - ); - - // Over the cap: d drops first, then c - the largest names go. - assert_eq!( - super::hydrate::select::select_properties(entries, None, 2), - vec![property("https://x.test/a/"), property("https://x.test/b/")], - ); -} - -#[test] -fn select_properties_protect_label() { - let entries = vec![ - property("https://x.test/a/"), - property("https://x.test/b/"), - property("https://x.test/z/"), - ]; - let label = BaseUrl::new("https://x.test/z/".to_owned()).expect("fixture keys are base URLs"); - - // z is reverse-lexicographically first to drop, but it is the - // label property: it survives every cap that admits at least - // one property, and the survivors still emit ascending. - assert_eq!( - super::hydrate::select::select_properties(entries.clone(), Some(&label), 2), - vec![property("https://x.test/a/"), property("https://x.test/z/")], - ); - assert_eq!( - super::hydrate::select::select_properties(entries.clone(), Some(&label), 1), - vec![property("https://x.test/z/")], - ); - - // A cap of zero admits nothing - even the label drops. - assert_eq!( - super::hydrate::select::select_properties(entries, Some(&label), 0), - vec![], - ); -} - -/// One locate request built directly. -fn locate_request(entity_id: String) -> super::LocateRequest { - super::LocateRequest { - entity_id: Some(entity_id), - row: None, - colored_type_ids: Vec::new(), - } -} - -/// A locate source names one subject in one of two identity domains. -/// -/// A by-`row` request resolves through the wire codec's ingress, pure arithmetic with no store, and -/// answers the same response bytes as the by-`entityId` request for that node. A wire value outside -/// the encoded image collapses into `unknown-entity`, and assembly rejects a body carrying -/// both or neither source field by name with its count. -#[tokio::test] -async fn locate_by_wire_row_matches_by_entity() { - let (_generation, atlas) = publish("locate-by-row").await; - let limits = ServeLimits::default(); - - // Row 7's wire id round-trips by construction (the codec is a - // bijection); the equivalence under test is the two request - // paths, whose agreement is the whole contract in one assertion. - let node_codec = test_codec(&atlas); - let wire = node_codec.encode(NodeRowId::new(7), atlas.node_universe()); - let by_entity = locate_request(entity_string_of(7)); - let mut by_row: super::LocateRequest = - serde_json::from_value(serde_json::json!({ "row": wire.get() })) - .expect("a by-row body deserializes"); - assert_eq!(by_row.row, Some(wire)); - assert_eq!(by_row.entity_id, None); - let by_entity_bytes = atlas - .locate( - &by_entity, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UnresolvedStore, - ) - .expect("the entity resolves"); - let by_row_bytes = atlas - .locate( - &by_row, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UnresolvedStore, - ) - .expect("the wire row resolves"); - assert_eq!( - by_entity_bytes, by_row_bytes, - "one node, two source domains, identical bytes", - ); - - // Wire ids are sparse in the u32 range: a value outside the - // encoded image collapses into the entity path's own rejection. - // Neither probe collides with the 48-value image under the - // fixture key - pinned here, not left to runtime luck. - let image: HashSet = (0..48) - .map(|row| { - node_codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }) - .collect(); - assert!( - !image.contains(&48) && !image.contains(&u32::MAX), - "the probes lie outside the image", - ); - for garbage in [48, u32::MAX] { - by_row.row = Some(codec::WireRow::pinned(garbage)); - assert_matches!( - atlas.locate( - &by_row, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore - ), - Err(super::LocateError::UnknownEntity), - "{garbage}", - ); - } - - // Both sources or none: rejected by name, with the count. - by_row.row = Some(wire); - by_row.entity_id = Some(entity_string_of(7)); - assert_matches!( - atlas.locate( - &by_row, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore - ), - Err(super::LocateError::Source { carried: 2 }), - ); - by_row.row = None; - by_row.entity_id = None; - assert_matches!( - atlas.locate( - &by_row, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore - ), - Err(super::LocateError::Source { carried: 0 }), - ); -} - -/// The locate one-call path encodes byte-exactly against the derived wire document. -/// -/// Assembly and encoding against the groundwork layers' own outputs, with a store answering that -/// nothing resolves: the mandatory trailer rides empty tables and null columns, and both source -/// completeness flags read `false` - an unresolved source can attest nothing. -#[tokio::test] -async fn locate_pinned_envelope() { - use crate::salt::{ - postings::artifact::Membership, - wire::locate::{LocateResponse, LocateTrailer}, - }; - - let (generation, atlas) = publish("locate-endpoint").await; - let limits = ServeLimits::default(); - let bound = Bound::of(&atlas, &FULL); - let view = bound.view(&atlas); - - let mut request = locate_request(entity_string_of(0)); - let bytes = atlas - .locate( - &request, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UnresolvedStore, - ) - .expect("the request is well-formed"); - assert_eq!(bytes[0..8], *b"SALTILEL"); - - // The groundwork layers derive the expectation independently; - // their own tests pin their behaviour. - let source = atlas - .resolve_source(&view, &entity_string_of(0)) - .expect("row 0 is a node"); - let subgraph = atlas.locate_subgraph(source, limits.locate, &view); - let node_codec = test_codec(&atlas); - let wire_of = |row: NodeRowId| node_codec.encode(row, atlas.node_universe()); - let columns = EdgeColumns::pinned(subgraph.edges.iter().map(|&(edge, id)| { - let edge = fitted(edge); - - (wire_of(edge.source).get(), wire_of(edge.target).get(), id) - })); - let wire_rows: Vec> = - atlas.row_ids().iter().map(|&row| wire_of(row)).collect(); - let nodes = subgraph.delivered.len(); - let edges = subgraph.edges.len(); - let no_labels: Vec<&Label> = vec![Label::EMPTY; nodes]; - let no_types: Vec>> = vec![None; nodes]; - let no_link_labels: Vec<&Label> = vec![Label::EMPTY; edges]; - let no_lists: Vec>> = vec![Vec::new(); edges]; - let no_flags: Box> = DenseBitSlice::new_empty(edges); - let no_maps: Vec>> = vec![None; edges]; - let response = |masks: Option<&[Membership<'_>]>| { - LocateResponse { - generation: generation.id().digest(), - variant: 0, - cell: source.cell, - complete: subgraph.complete, - entity_id: entity_id_of(0), - type_ids_complete: false, - properties_complete: false, - delivered: &subgraph.delivered, - arrivals: IdSlice::from_raw(&[]), - positions: atlas.positions(), - rows: IdSlice::from_raw(&wire_rows), - masks, - edges: &columns, - trailer: LocateTrailer { - type_table: IdSlice::from_raw(&[]), - property_table: IdSlice::from_raw(&[]), - labels: IdSlice::from_raw(&no_labels), - type_ids: IdSlice::from_raw(&no_types), - properties: None, - link_labels: IdSlice::from_raw(&no_link_labels), - link_type_ids: IdSlice::from_raw(&no_lists), - link_type_ids_complete: &no_flags, - link_properties: IdSlice::from_raw(&no_maps), - link_properties_complete: &no_flags, - }, - } - .encode() - }; - assert_eq!(bytes, response(None), "the envelope matches the derivation"); - - // An id resolving to no type is legal and reads zero bits; the - // TYPE_MASK slot rides exactly the requests that colour. - request.colored_type_ids = vec![ - "https://unknown.test/t/v/1" - .parse() - .expect("the literal is a versioned url"), - ]; - let colored_bytes = atlas - .locate( - &request, - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UnresolvedStore, - ) - .expect("unresolvable colored ids are legal"); - assert_eq!(colored_bytes, response(Some(&[Membership::List(&[])]))); -} - -/// Every locate rejection carries its name. -/// -/// The unknown-entity doctrine treats unparsable, unknown, and wrong-domain ids identically. -#[tokio::test] -async fn locate_rejection_names() { - let (_generation, atlas) = publish("locate-rejects").await; - let limits = ServeLimits::default(); - - // Unparsable, unknown, and an EDGE id (wrong identity domain) - // are one rejection: an id that cannot name a visible node. - for id in [ - "not an entity id".to_owned(), - entity_string_of(50), - entity_string_of(EDGE_SEED), - ] { - assert_matches!( - atlas.locate( - &locate_request(id.clone()), - limits, - Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), - UntouchedStore, - ), - Err(super::LocateError::UnknownEntity), - "{id}", - ); - } - - // The coloredTypeIds cap is the tile endpoint's own. - let mut colored = locate_request(entity_string_of(0)); - colored.colored_type_ids = vec![ - "https://example.com/types/thing/v/1" - .parse() - .expect("the literal is a versioned url"); - limits.tile.colored_type_ids as usize + 1 - ]; - assert_matches!( - atlas.locate(&colored, limits, Bound::new(&atlas, &FULL, CutOffset::ZERO).view(&atlas), UntouchedStore), - Err(super::LocateError::Types { count, maximum }) - if count == limits.tile.colored_type_ids as usize + 1 - && maximum == limits.tile.colored_type_ids, - ); -} - -/// The intern tables are the sorted, deduplicated unions of every reference. -/// -/// The property maps lead with the source's and keep each entity's ascending-name order as -/// ascending indexes; node type references take the representative type, link references keep -/// canonical order. `None` marks an unresolved entity, an empty list a resolved one without -/// surviving entries. -#[test] -fn intern_table_references() { - use super::{ - TableIndex, - hydrate::ScalarValue, - locate::{intern_properties, intern_types}, - }; - use crate::salt::wire::locate::{PropertyMap, PropertyValue}; - - let owned = |name: &str, value: ScalarValue| { - ( - BaseUrl::new(format!("https://x.test/{name}/")).expect("fixture keys are base URLs"), - value, - ) - }; - let source = vec![ - owned("b", ScalarValue::String("t".to_owned())), - owned("d", ScalarValue::Integer(7)), - ]; - let links = vec![ - None, - Some(vec![ - owned("a", ScalarValue::Null), - owned("b", ScalarValue::Bool(true)), - ]), - Some(vec![]), - ]; - - let (names, maps) = intern_properties(Some(&source), IdSlice::from_raw(&links)); - assert_eq!( - names.entries().as_raw(), - [ - "https://x.test/a/", - "https://x.test/b/", - "https://x.test/d/" - ], - ); - assert_eq!( - maps, - vec![ - Some(PropertyMap::new_unchecked(vec![ - (TableIndex::new(1), PropertyValue::Text("t")), - (TableIndex::new(2), PropertyValue::Integer(7)), - ])), - None, - Some(PropertyMap::new_unchecked(vec![ - (TableIndex::new(0), PropertyValue::Null), - (TableIndex::new(1), PropertyValue::Boolean(true)), - ])), - Some(PropertyMap::new_unchecked(vec![])), - ], - ); - - // An absent source stays the leading entry. - let (names, maps) = intern_properties(None, IdSlice::from_raw(&links[..2])); - assert_eq!( - names.entries().as_raw(), - ["https://x.test/a/", "https://x.test/b/"] - ); - assert_eq!( - maps, - vec![ - None, - None, - Some(PropertyMap::new_unchecked(vec![ - (TableIndex::new(0), PropertyValue::Null), - (TableIndex::new(1), PropertyValue::Boolean(true)), - ])), - ], - ); - - // Types: nodes contribute their FIRST direct type, links their - // whole capped lists in canonical (unsorted) order. - let url = |name: &str| -> VersionedUrl { - format!("https://t.test/{name}/v/1") - .parse() - .expect("test urls parse") - }; - let nodes = vec![vec![url("m"), url("z")], Vec::new(), vec![url("a")]]; - let link_types = vec![vec![url("z"), url("a")], Vec::new()]; - let (table, type_ids, link_type_ids) = - intern_types(IdSlice::from_raw(&nodes), IdSlice::from_raw(&link_types)); - // The node-only type "m" interns; the node-second "z" also - // interns through the link list. - assert_eq!( - table.entries().as_raw(), - [ - "https://t.test/a/v/1", - "https://t.test/m/v/1", - "https://t.test/z/v/1" - ], - ); - assert_eq!( - type_ids.as_raw(), - [Some(TableIndex::new(1)), None, Some(TableIndex::new(0))], - ); - assert_eq!( - link_type_ids.as_raw(), - [vec![TableIndex::new(2), TableIndex::new(0)], Vec::new()], - ); -} - -/// The source coverage predicate reads exactly the ratified rule. -/// -/// `directTypes \u{2286} coloredTypeIds`, with `false` for a store-absent source, an unrecorded -/// type list, and an empty palette. -#[test] -fn source_type_subset_rule() { - use super::{colour::Palette, locate::covers_source_types}; - - let url = |name: &str| -> VersionedUrl { - format!("https://t.test/{name}/v/1") - .parse() - .expect("test urls parse") - }; - let colored = Palette::of(&[url("a"), url("b")]); - - // In the ratified example the source carries {a, c} and the - // request colours {a, b}, so coverage fails because c is outside - // the set. - assert!(!covers_source_types(true, &[url("a"), url("c")], &colored)); - assert!(covers_source_types(true, &[url("a")], &colored)); - assert!(covers_source_types(true, &[url("b"), url("a")], &colored)); - - // An empty palette covers nothing, and an unreadable or - // unrecorded type list attests nothing. An unparsable direct - // type cannot reach coverage: the store boundary parses every - // URL before a hydration column exists. - assert!(!covers_source_types(true, &[url("a")], &Palette::of(&[]))); - assert!(!covers_source_types(true, &[], &colored)); - assert!(!covers_source_types(false, &[url("a")], &colored)); -} - -/// A proof hiding exactly `hidden` among the atlas's node rows, and no link rows. -/// -/// The link mask admits every link row of the generation. A battery built on this helper therefore -/// varies the node axis alone. -pub(crate) fn mask_hiding(atlas: &Atlas, hidden: &[u32]) -> VisibilityProof { - mask_hiding_rows(atlas, hidden, &[]) -} - -/// A proof hiding `hidden_nodes` among the atlas's node rows and `hidden_edges` among its link -/// rows. -/// -/// Exclusion over the generation's own domains builds both masks. The proof therefore differs from -/// the full-visibility proof in exactly the listed rows. -fn mask_hiding_rows(atlas: &Atlas, hidden_nodes: &[u32], hidden_edges: &[u32]) -> VisibilityProof { - VisibilityProof::from_masks( - domain_mask(atlas.row_ids().len(), hidden_nodes), - domain_mask(atlas.endpoints.view().len(), hidden_edges), - fast_hash_set(), - ) -} - -/// A mask over the domain `[0, rows)` admitting every row except `hidden`. -fn domain_mask(rows: usize, hidden: &[u32]) -> CompressedBitSet { - let rows = u32::try_from(rows).expect("fixture domains fit u32"); - for &row in hidden { - assert!(row < rows, "a fixture hides rows of the domain it masks"); - } - - CompressedBitSet::from_rows( - (0..rows) - .filter(|row| !hidden.contains(row)) - .map(T::from_u32), - ) -} - -/// Folds one `Ended` feed event per seed into a published snapshot over `atlas`. -/// -/// The events travel the consumer's own conversion. The snapshot is therefore the one publication -/// serving would read rather than a hand-assembled equivalent. Fixture node row `r` owns seed -/// `r`. Withdrawing a row is therefore withdrawing its seed. -pub(crate) fn withdrawing(atlas: &Atlas, seeds: &[u8]) -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - for &seed in seeds { - let event = EntityEvent::Ended(EntityEnd { - entity: EntityId { - web_id: WebId::new(Uuid::from_bytes([seed; 16])), - entity_uuid: EntityUuid::new(Uuid::from_bytes([seed ^ 0xFF; 16])), - draft_id: None, - }, - ended_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - } - - register.snapshot( - atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(1), - ) -} - -/// Returns a fixed generation identity for codec derivation. -fn codec_generation() -> GenerationId { - "1111111111111111111111111111111111111111111111111111111111111111" - .parse() - .expect("the literal is 64 hex digits") -} - -/// Reads the delivered count and the per-bucket runs from a tile `HEAD`. -/// -/// Panics on the retired fill key and on a head whose runs do not account for its delivered -/// count, the two laws the head owns for every caller of this helper. -fn head_counts(head: &[u8]) -> (u64, Vec) { - let mut reader = CborReader { bytes: head, at: 0 }; - let entries = reader.head(5); - let (mut delivered, mut runs) = (0, Vec::new()); - for _ in 0..entries { - let key = reader.uint(); - match key { - 4 => delivered = reader.uint(), - 7 => { - let count = usize::try_from(reader.head(4)).expect("run counts fit usize"); - runs = core::iter::repeat_with(|| reader.uint()) - .take(count) - .collect(); - } - 11 => panic!("HEAD key 11 belongs to a retired field: no response carries a fill tail"), - _ => reader.skip(), - } - } - - assert_eq!( - runs.iter().sum::(), - delivered, - "every delivered row belongs to a bucket the runs count", - ); - - (delivered, runs) -} - -/// Asserts one tile head accounts for `delivered` points. -/// -/// The runs-against-delivered law lives in [`head_counts`], which every head reader calls. This -/// adds the caller's own count. A head agreeing with itself but not with the response therefore -/// still fails. -#[track_caller] -fn assert_head_delivers(bytes: &[u8], delivered: u64) { - let (counted, _runs) = head_counts(section(bytes, HEAD).expect("HEAD is present")); - assert_eq!(counted, delivered, "the head counts the delivered rows"); -} - -/// Reads the occupied-child bitmask from a tile `HEAD`. -fn children_of(head: &[u8]) -> u64 { - let mut reader = CborReader { bytes: head, at: 0 }; - let entries = reader.head(5); - for _ in 0..entries { - let key = reader.uint(); - if key == 9 { - return reader.uint(); - } - reader.skip(); - } - - 0 -} - -/// Reads the root's global map from a tile `HEAD`, [`None`] when the head carries none. -/// -/// The visible count of the root's schedule, the visible extent as `[minX, minY, maxX, maxY]`, and -/// the deepest visible bucket - `HEAD` key 8, whose value is the census the scope resolved. -fn head_global(head: &[u8]) -> Option<(u64, Option<[f32; 4]>, u64)> { - let mut reader = CborReader { bytes: head, at: 0 }; - let entries = reader.head(5); - for _ in 0..entries { - if reader.uint() != 8 { - reader.skip(); - continue; - } - - let fields = reader.head(5); - let (mut visible, mut bounds, mut min_resolution) = (0, None, 0); - for _ in 0..fields { - match reader.uint() { - 0 => visible = reader.uint(), - 1 => { - assert_eq!(reader.head(4), 4, "the extent is four wire floats"); - bounds = Some([reader.f32(), reader.f32(), reader.f32(), reader.f32()]); - } - 2 => min_resolution = reader.uint(), - _ => reader.skip(), - } - } - - return Some((visible, bounds, min_resolution)); - } - - None -} - -/// A minimal CBOR reader over the deterministic profile the tile `HEAD` uses. -struct CborReader<'bytes> { - bytes: &'bytes [u8], - at: usize, -} - -impl CborReader<'_> { - /// Reads one head, asserting the major type, and returns its argument. - fn head(&mut self, major: u8) -> u64 { - let byte = self.bytes[self.at]; - assert_eq!( - byte >> 5, - major, - "the HEAD item at {} has major {major}", - self.at - ); - self.read_head().1 - } - - /// Reads one head, returning the major type and the argument. - fn read_head(&mut self) -> (u8, u64) { - let byte = self.bytes[self.at]; - self.at += 1; - let (major, argument) = (byte >> 5, byte & 0x1F); - let argument = match argument { - 0..24 => u64::from(argument), - 24..28 => { - let width = 1 << (argument - 24); - let mut value = 0_u64; - for _ in 0..width { - value = (value << 8) | u64::from(self.bytes[self.at]); - self.at += 1; - } - value - } - _ => panic!("the deterministic profile emits definite lengths alone"), - }; - - (major, argument) - } - - /// Reads one unsigned integer. - fn uint(&mut self) -> u64 { - let (major, value) = self.read_head(); - assert_eq!(major, 0, "expected a uint at {}", self.at); - value - } - - /// Reads one wire float. - /// - /// The profile emits coordinates as CBOR single-precision floats. The head read therefore - /// consumed the four payload bytes as the argument. - fn f32(&mut self) -> f32 { - let (major, bits) = self.read_head(); - assert_eq!(major, 7, "expected a float at {}", self.at); - f32::from_bits(u32::try_from(bits).expect("wire floats are single precision")) - } - - /// Skips one item of any shape the tile `HEAD` carries. - fn skip(&mut self) { - let (major, argument) = self.read_head(); - match major { - // Uints and simple values / floats: the head read already consumed the argument or - // its trailing bytes. - 0 | 7 => {} - // Byte and text strings carry their content inline. - 2 | 3 => self.at += usize::try_from(argument).expect("section lengths fit usize"), - // Arrays and maps recurse per element. - 4 => { - for _ in 0..argument { - self.skip(); - } - } - 5 => { - for _ in 0..argument * 2 { - self.skip(); - } - } - _ => panic!("the tile HEAD carries no major-{major} items"), - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/tests/open.rs b/libs/@local/graph/atlas/src/serve/tests/open.rs deleted file mode 100644 index 0841ffb230c..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/open.rs +++ /dev/null @@ -1,697 +0,0 @@ -//! Cross-artifact agreement at open: each shared-domain disagreement names its own variant. -//! -//! `Atlas::open` is the only place that checks the artifacts against each other. Every read path -//! below it indexes across them without re-validating. A generation whose artifacts disagree -//! therefore gets one chance at refusal, and refusing it under the wrong name leaves an operator -//! little better off than serving it would: the variant is what an operator repairs from. -//! -//! Open verifies every published file against the digest its manifest records before any -//! structural check runs. A structural tamper therefore travels through [`republish`], which seals -//! the edited file into a new generation with its digest recorded, and the check under test is the -//! first one that can refuse it. The corruption tests edit a published generation in place instead, -//! and the digest check refuses them before anything else looks. - -use core::assert_matches; -use std::io::Write as _; - -use camino::Utf8PathBuf; -use hashql_core::id::{Id as _, IdVec}; -use type_system::ontology::id::VersionedUrl; -use zerocopy::{IntoBytes as _, LE, U64}; - -use super::{ - Atlas, EDGE_SEED, OpenAtlasError, entity_id_of, fit_fixture, fixture_type_url, - recreate_writable, republish, rewrite_identities, store_identities, test_open_options, -}; -use crate::{ - file::{ - WriteInto as _, - array::{ArrayVariant, Dim, SizedArrayWriter}, - generation::{Generation, GenerationRoot}, - identity::{Row, read::IdentityFile}, - postings::{read::PostingsFile, write::Regions}, - quad::{Node, TypeSets, read::QuadFile}, - repository::{FileName, IntegrityVerificationError}, - }, - identity::{BasePosition, EdgeRowId, ImportanceRank, NodeRowId, OntologyRowId}, - postgres::id::{ArchivedEntityId, ArchivedOntologyTypeUuid}, - salt::{adjacency::Adjacency, fit::prepare::identity::IdentityTable}, -}; - -/// Rewrites an entity identity artifact with `rows` sequential fixture ids from `seed`. -fn shorten_entities(path: &Utf8PathBuf, rows: u64, seed: u8) { - let mut table = IdentityTable::::new(); - for row in 0..rows { - let row = u8::try_from(row).expect("fixture row counts fit u8"); - table.push(entity_id_of(seed + row)); - } - // The all-zero legend payload names ontology row 0 under the empty label. - let empty = vec![vec![0_u8; 8]; usize::try_from(rows).expect("fixture row counts fit usize")]; - rewrite_identities(path, &table, &empty); -} - -/// Rewrites the ontology identity artifact with `rows` fixture type uuids. -fn shorten_ontology(path: &Utf8PathBuf, rows: u64) { - let mut table = IdentityTable::::new(); - for row in 0..rows { - let url: VersionedUrl = fixture_type_url(row) - .parse() - .expect("the fixture URL parses"); - table.push(ArchivedOntologyTypeUuid::from_url(&url)); - } - let empty = vec![Vec::new(); usize::try_from(rows).expect("fixture row counts fit usize")]; - rewrite_identities(path, &table, &empty); -} - -/// Rewrites the endpoint column with `pairs`, dropping whatever the fixture published beyond it. -fn shorten_endpoints(path: &Utf8PathBuf, pairs: &[[NodeRowId; 2]]) { - let file = recreate_writable(path); - let mut writer = SizedArrayWriter::new( - file, - ArrayVariant::U64Le, - &[Dim::new(pairs.len() as u64), Dim::new(2)], - ) - .expect("the header writes"); - for &[source, target] in pairs { - let row = [ - U64::::new(source.as_u64()), - U64::::new(target.as_u64()), - ]; - writer.write_row(row.as_bytes()).expect("the row writes"); - } - writer.finish().expect("the column seals"); -} - -/// Rewrites the adjacency artifact over the same edges, spanning `rows` node rows. -/// -/// The production builder writes it. The file therefore keeps every property the incident-list -/// contract checks (paired runs, the domain-bound column dimension, one slot per edge per -/// direction) and disagrees with the columns on the node domain alone. -fn respan_adjacency(path: &Utf8PathBuf, rows: usize, endpoints: &[[NodeRowId; 2]]) { - let file = recreate_writable(path); - let _digest = Adjacency::build(rows, endpoints) - .write_into(file) - .expect("the adjacency should write"); -} - -/// Rewrites the quad artifact with the root's subtree count set to `points`. -/// -/// The topology, the runs, and the type sets are the published ones. The quad format validates the -/// header, the fenceposts, and the child indexes, and never the subtree counts, which is why `open` -/// must. -fn retarget_quad_root(path: &Utf8PathBuf, points: u32) { - // The mapping ends before the rewrite: the file backs the slices read here. - let (mut nodes, sets) = { - let quad = QuadFile::open(path).expect("the published quad artifact opens"); - let nodes = quad.nodes().to_vec(); - let sets: Vec> = (0..nodes.len()) - .map(|node| { - let node = u32::try_from(node).expect("fixture node tables fit u32"); - quad.type_set(node).iter().map(|id| id.get()).collect() - }) - .collect(); - (nodes, TypeSets::from_sets(&sets)) - }; - - let root = *nodes.first().expect("the fixture quad holds a root"); - let run = root.run(); - let length = u32::try_from(run.end - run.start).expect("fixture runs fit u32"); - nodes[0] = Node::new(root.children(), run.start, length, points); - - let file = recreate_writable(path); - let mut file = std::io::BufWriter::new(file); - crate::file::quad::write::write_regions(&nodes, &sets, &mut file) - .expect("the quad regions write"); - file.flush().expect("the quad artifact flushes"); -} - -/// Rewrites the postings artifact with its point domain set to `points`. -/// -/// Every other region restates the published one. The dense sets rebuild over the new domain, -/// because every frame's own domain count restates the header's and open checks the agreement, -/// and the direct fenceposts resize to cover it - truncating drops the stranded runs' ids, -/// growing appends empty runs. Only the header's point count and the bound every list position -/// must clear actually move. -fn retarget_postings_points(path: &Utf8PathBuf, points: u64) { - // The mapping ends before the rewrite. The file backs the slices read here, hence every region - // copies into build vocabulary first. - let (flags, lists, dense_sets, parents, direct) = { - let postings = PostingsFile::open(path).expect("the published postings artifact opens"); - - let published = postings.flags(); - let mut flags = crate::bitset::DenseBitSlice::new_empty( - usize::try_from(postings.types()).expect("fixture type domains fit usize"), - ); - let mut dense_sets = crate::bitset::DenseBitSliceArray::new_empty( - usize::try_from(points).expect("fixture point domains fit usize"), - usize::try_from(published.count()).expect("fixture dense counts fit usize"), - ); - for (rank, type_row) in published.iter().enumerate() { - flags.insert(type_row); - for member in postings.dense_sets()[rank].iter() { - dense_sets[rank].insert(member); - } - } - - let posts_len = usize::try_from(points).expect("fixture point domains fit usize") + 1; - let mut direct_posts = postings - .direct_posts() - .iter() - .map(|post| usize::try_from(post.get()).expect("fixture posts fit usize")) - .collect::>(); - let mut direct_ids = postings.direct_ids().to_vec(); - if direct_posts.len() > posts_len { - direct_posts.truncate(posts_len); - let close = *direct_posts - .last() - .expect("the fencepost region anchors at zero"); - direct_ids.truncate(close); - } else { - let close = *direct_posts - .last() - .expect("the fencepost region anchors at zero"); - direct_posts.resize(posts_len, close); - } - - ( - flags, - crate::runs::Runs::from_parts( - IdVec::from_raw(postings.list_posts().to_vec()), - postings.list_entries().to_vec(), - ) - .expect("the published list columns satisfy the fencepost law"), - dense_sets, - crate::runs::Runs::from_parts( - IdVec::from_raw(postings.parent_posts().to_vec()), - postings.parent_ids().to_vec(), - ) - .expect("the published parent columns satisfy the fencepost law"), - crate::runs::Runs::from_parts( - IdVec::from_raw( - direct_posts - .iter() - .map(|&post| U64::new(post as u64)) - .collect(), - ), - direct_ids, - ) - .expect("the resized direct columns satisfy the fencepost law"), - ) - }; - - let file = recreate_writable(path); - let mut file = std::io::BufWriter::new(file); - crate::file::postings::write::write_regions( - Regions { - flags: &flags, - lists: &lists, - dense_sets: &dense_sets, - parents: &parents, - direct: &direct, - }, - &mut file, - ) - .expect("the postings regions write"); - file.flush().expect("the postings artifact flushes"); -} - -/// Rewrites a little-endian `u32` column with `rows` ascending values. -/// -/// The values do not matter to the check under test - `open` compares lengths - and ascending keeps -/// the file a plausible permutation prefix rather than a shape no producer would write. -fn shorten_u32_column(path: &Utf8PathBuf, rows: u64) { - let file = recreate_writable(path); - let mut writer = SizedArrayWriter::new(file, ArrayVariant::U32Le, &[Dim::new(rows)]) - .expect("the header writes"); - for row in 0..rows { - let value = u32::try_from(row).expect("fixture rows fit u32"); - writer - .write_row(&value.to_le_bytes()) - .expect("the row writes"); - } - writer.finish().expect("the column seals"); -} - -/// Rewrites a little-endian `u32` column with `rows` copies of `value`. -/// -/// A constant column keeps its length and its format and cannot be a permutation. The roundtrip -/// sample refuses exactly that. -fn constant_u32_column(path: &Utf8PathBuf, rows: u64, value: u32) { - let file = recreate_writable(path); - let mut writer = SizedArrayWriter::new(file, ArrayVariant::U32Le, &[Dim::new(rows)]) - .expect("the header writes"); - for _ in 0..rows { - writer - .write_row(&value.to_le_bytes()) - .expect("the row writes"); - } - writer.finish().expect("the column seals"); -} - -/// Rewrites a little-endian `u64` column with `rows` copies of `value`. -/// -/// The row column is the one `u64` column of the base order, and the same constant-column argument -/// holds for it. -fn constant_u64_column(path: &Utf8PathBuf, rows: u64, value: u64) { - let file = recreate_writable(path); - let mut writer = SizedArrayWriter::new(file, ArrayVariant::U64Le, &[Dim::new(rows)]) - .expect("the header writes"); - for _ in 0..rows { - writer - .write_row(&value.to_le_bytes()) - .expect("the row writes"); - } - writer.finish().expect("the column seals"); -} - -/// One published generation with the counts its tamper witnesses compare against. -/// -/// Every tamper test publishes its own fixture and moves a single domain by one row, leaving -/// the artifact valid at its own format. The tamper republishes the generation with the edited -/// file sealed under its own digest, and the rejection must name the tamper's own variant. The -/// untampered generation is never written to, and reopening it after the rejection is the -/// negative control: each rejection belongs to its tamper rather than to a fixture that had -/// stopped opening unnoticed. -/// -/// The artifact structure forces which direction a tamper moves a domain. Dropping a node row -/// from the adjacency would drop that node's edge slots with it and move the edge domain in the -/// same tamper; narrowing the postings' point domain can strand a membership position outside -/// it, which the postings contract refuses first and under its own name. Both therefore add a -/// row, and widening is as much a producer bug as truncation is. -/// -/// Both universe variants are absent and cannot be present: `Universe` and `EdgeUniverse` fire -/// above `u32::MAX` rows, which no fixture constructs. They guard arithmetic beyond any fixture's -/// reach. Every other variant of the cross-artifact pass has a tamper test over this fixture. -struct TamperFixture { - root: GenerationRoot, - generation: Generation, - nodes: u64, - endpoints: Vec<[NodeRowId; 2]>, -} - -impl TamperFixture { - /// Publishes `name`'s generation and reads the untampered counts, proving it opens. - async fn publish(name: &str) -> Self { - let (root, generation) = fit_fixture(name).await; - let generation = store_identities(&root, &generation); - - let atlas = Atlas::open(&root, generation.id(), test_open_options()) - .expect("the published fixture opens"); - let nodes = atlas.row_ids().len() as u64; - let endpoints = atlas.endpoint_pairs().as_raw().to_vec(); - drop(atlas); - - Self { - root, - generation, - nodes, - endpoints, - } - } - - /// Opens the untampered generation. - fn open(&self) -> Result { - Atlas::open(&self.root, self.generation.id(), test_open_options()) - } - - /// Republishes the generation with `edit` applied to the artifact `name` and opens the result. - fn open_tampered( - &self, - name: &FileName, - edit: impl FnOnce(&Utf8PathBuf), - ) -> Result { - let tampered = republish(&self.root, &self.generation, |staging| { - edit(&staging.path_of(name)); - }); - Atlas::open(&self.root, tampered.id(), test_open_options()) - } - - /// Returns the path of `name`'s artifact in the untampered generation. - fn path_of(&self, name: &FileName) -> Utf8PathBuf { - self.generation.path_of(name) - } - - /// The fixture's edge count, the endpoint pairs' own length. - fn edges(&self) -> u64 { - self.endpoints.len() as u64 - } -} - -/// Open refuses a node identity table short of the code column, under `Identities`. -#[tokio::test] -async fn node_identities_short() { - let fixture = TamperFixture::publish("open-node-identities").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.node_identities.name(), |path| { - shorten_entities::(path, nodes - 1, 0); - }) - .expect_err("a short node identity table is refused"), - OpenAtlasError::Identities { identities, codes } - if identities == nodes - 1 && codes == nodes, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses an edge identity table short of the adjacency's edge domain, under -/// `EdgeIdentities`. -#[tokio::test] -async fn edge_identities_short() { - let fixture = TamperFixture::publish("open-edge-identities").await; - let edges = fixture.edges(); - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.edge_identities.name(), |path| { - shorten_entities::(path, edges - 1, EDGE_SEED); - }) - .expect_err("a short edge identity table is refused"), - OpenAtlasError::EdgeIdentities { identities, edges: spanned } - if identities == edges - 1 && spanned == edges, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses an ontology identity table short of the postings' type domain, under `Types`. -#[tokio::test] -async fn ontology_identities_short() { - let fixture = TamperFixture::publish("open-ontology-identities").await; - let files = &fixture.generation.repository().files; - let types = IdentityFile::open(fixture.path_of(&files.ontology_identities.name())) - .expect("the published ontology identities open") - .rows(); - - assert_matches!( - fixture - .open_tampered(&files.ontology_identities.name(), |path| { - shorten_ontology(path, types - 1); - }) - .expect_err("a short ontology identity table is refused"), - OpenAtlasError::Types { postings, identities } - if postings == types && identities == types - 1, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a rank column short of the code column, under `Columns`. -#[tokio::test] -async fn rank_column_short() { - let fixture = TamperFixture::publish("open-rank-column").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.rank_of_position.name(), |path| { - shorten_u32_column(path, nodes - 1); - }) - .expect_err("a short rank column is refused"), - OpenAtlasError::Columns { - codes, - coordinates, - rows, - ranks: ranked, - positions, - rank_positions, - } if codes == nodes - && coordinates == nodes - && rows == nodes - && ranked == nodes - 1 - && positions == nodes - && rank_positions == nodes, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a reverse rank permutation short of the code column, under `Columns`. -#[tokio::test] -async fn rank_positions_short() { - let fixture = TamperFixture::publish("open-rank-positions-short").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.position_of_rank.name(), |path| { - shorten_u32_column(path, nodes - 1); - }) - .expect_err("a short reverse rank permutation is refused"), - OpenAtlasError::Columns { - codes, - rank_positions: reversed, - .. - } if codes == nodes && reversed == nodes - 1, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a reverse row permutation short of the code column, under `Columns`. -#[tokio::test] -async fn row_positions_short() { - let fixture = TamperFixture::publish("open-row-positions-short").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.position_of_row.name(), |path| { - shorten_u32_column(path, nodes - 1); - }) - .expect_err("a short reverse row permutation is refused"), - OpenAtlasError::Columns { - codes, - positions: reversed, - .. - } if codes == nodes && reversed == nodes - 1, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a reverse rank permutation that is no permutation, under `RankInverse`. -/// -/// Every rank claiming position zero keeps the length and the format. The roundtrip sample -/// therefore refuses the pairing at the first sampled position past zero. -#[tokio::test] -async fn rank_positions_constant() { - let fixture = TamperFixture::publish("open-rank-positions-constant").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.position_of_rank.name(), |path| { - constant_u32_column(path, nodes, 0); - }) - .expect_err("a non-inverse reverse rank permutation is refused"), - OpenAtlasError::RankInverse { - position, - rank: _, - roundtrip: Some(roundtrip), - } if position > BasePosition::MIN && roundtrip == BasePosition::MIN, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a rank outside the reverse permutation's domain, under `RankInverse`. -/// -/// The sample reports the roundtrip as absent at the first sampled position. -#[tokio::test] -async fn ranks_out_of_domain() { - let fixture = TamperFixture::publish("open-ranks-out-of-domain").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.rank_of_position.name(), |path| { - constant_u32_column(path, nodes, u32::MAX); - }) - .expect_err("an out-of-domain rank is refused"), - OpenAtlasError::RankInverse { - position, - rank, - roundtrip: None, - } if position == BasePosition::MIN && rank == ImportanceRank::MAX, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a reverse row permutation that is no permutation, under `RowInverse`. -/// -/// Every node claiming position zero keeps the length and the format. Position zero's own node -/// roundtrips, and the first sampled position past it does not. -#[tokio::test] -async fn row_positions_constant() { - let fixture = TamperFixture::publish("open-row-positions-constant").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.position_of_row.name(), |path| { - constant_u32_column(path, nodes, 0); - }) - .expect_err("a non-inverse reverse row permutation is refused"), - OpenAtlasError::RowInverse { - position, - node: _, - roundtrip: Some(roundtrip), - } if position > BasePosition::MIN && roundtrip == BasePosition::MIN, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a node row outside the reverse permutation's domain, under `RowInverse`. -/// -/// The sample reports the roundtrip as absent at the first sampled position. -#[tokio::test] -async fn rows_out_of_domain() { - let fixture = TamperFixture::publish("open-rows-out-of-domain").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.row_of_position.name(), |path| { - constant_u64_column(path, nodes, u64::MAX); - }) - .expect_err("an out-of-domain node row is refused"), - OpenAtlasError::RowInverse { - position, - node, - roundtrip: None, - } if position == BasePosition::MIN && node == NodeRowId::MAX, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses an adjacency spanning an extra node row, under `Nodes`. -#[tokio::test] -async fn adjacency_extra_node_row() { - let fixture = TamperFixture::publish("open-adjacency-span").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.adjacency.name(), |path| { - respan_adjacency( - path, - usize::try_from(nodes + 1).expect("fixture node counts fit usize"), - &fixture.endpoints, - ); - }) - .expect_err("an adjacency spanning an extra node row is refused"), - OpenAtlasError::Nodes { adjacency: spanned, codes } - if spanned == nodes + 1 && codes == nodes, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses an endpoint column short of the adjacency's edge domain, under `Edges`. -#[tokio::test] -async fn endpoint_column_short() { - let fixture = TamperFixture::publish("open-endpoint-column").await; - let edges = fixture.edges(); - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.edge_endpoints.name(), |path| { - shorten_endpoints(path, &fixture.endpoints[..fixture.endpoints.len() - 1]); - }) - .expect_err("a short endpoint column is refused"), - OpenAtlasError::Edges { adjacency: spanned, endpoints: paired } - if spanned == edges && paired == edges - 1, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a quadtree root whose subtree count lies below the point count, under -/// `Subtree`. -#[tokio::test] -async fn quad_root_subtree_short() { - let fixture = TamperFixture::publish("open-quad-root").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.quad.name(), |path| { - retarget_quad_root( - path, - u32::try_from(nodes - 1).expect("fixture point counts fit u32"), - ); - }) - .expect_err("a root subtree count below the point count is refused"), - OpenAtlasError::Subtree { quad: counted, codes } - if counted == nodes - 1 && codes == nodes, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a postings point domain above the code column's count, under `Points`. -#[tokio::test] -async fn postings_points_wide() { - let fixture = TamperFixture::publish("open-postings-points").await; - let nodes = fixture.nodes; - let files = &fixture.generation.repository().files; - - assert_matches!( - fixture - .open_tampered(&files.postings.name(), |path| { - retarget_postings_points(path, nodes + 1); - }) - .expect_err("a postings point domain above the point count is refused"), - OpenAtlasError::Points { postings: spanned, codes } - if spanned == nodes + 1 && codes == nodes, - ); - fixture.open().expect("the untampered generation opens"); -} - -/// Open refuses a published file rewritten in place, under `Corruption`. -/// -/// The rewrite keeps the column's length and format and would fail the roundtrip sample if it -/// reached it. The digest check runs first and names the file with both digests. -#[tokio::test] -async fn corruption_rewritten_file() { - let fixture = TamperFixture::publish("open-corruption-rewritten").await; - let files = &fixture.generation.repository().files; - let name = files.rank_of_position.name(); - - constant_u32_column(&fixture.path_of(&name), fixture.nodes, 0); - assert_matches!( - fixture - .open() - .expect_err("a published file rewritten in place is refused"), - OpenAtlasError::Corruption(IntegrityVerificationError::Checksum { file, received }) - if file.name == name - && file.hash == files.rank_of_position.hash() - && received != file.hash, - ); -} - -/// Open refuses a generation missing a published file, under `Corruption`. -#[tokio::test] -async fn corruption_missing_file() { - let fixture = TamperFixture::publish("open-corruption-missing").await; - let name = fixture - .generation - .repository() - .files - .rank_of_position - .name(); - - std::fs::remove_file(fixture.path_of(&name)).expect("the published file removes"); - assert_matches!( - fixture - .open() - .expect_err("a generation missing a published file is refused"), - OpenAtlasError::Corruption(IntegrityVerificationError::Io { name: missing, error }) - if missing == name && error.kind() == std::io::ErrorKind::NotFound, - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/row_codec.rs b/libs/@local/graph/atlas/src/serve/tests/row_codec.rs deleted file mode 100644 index ff98010f63b..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/row_codec.rs +++ /dev/null @@ -1,476 +0,0 @@ -//! Tests for the wire row-id codec. -//! -//! The battery covers round trips, exactness of the image, key separation, and agreement with the -//! spec reference. - -use super::*; - -hashql_core::id::newtype!( - /// The row domain the codec battery drives. - struct ProbeRow(u32) -); - -/// Builds a test secret: `seed` left-aligned in a zeroed 32-byte key. -fn secret(seed: &[u8]) -> WireSecret { - let mut bytes = [0_u8; WireSecret::BYTES]; - bytes[..seed.len()].copy_from_slice(seed); - WireSecret::new(bytes) -} - -#[test] -fn codec_round_trips_every_small_universe() { - let codec = - codec::RowCodec::::derive(&secret(b"secret"), codec_generation(), b"test"); - - for universe in [ - 1_u32, 2, 3, 4, 5, 7, 8, 9, 15, 16, 17, 31, 32, 33, 48, 100, 257, 1000, - ] { - let bound = codec::Universe::new(ProbeRow::new(universe)); - - let image: Vec = (0..universe) - .map(|row| codec.encode(ProbeRow::new(row), bound).get()) - .collect(); - for (row, &wire) in (0..universe).zip(&image) { - assert_eq!( - codec.decode(codec::WireRow::pinned(wire), bound), - Some(ProbeRow::new(row)), - "decode inverts encode at N={universe}", - ); - } - let distinct: HashSet = image.iter().copied().collect(); - assert_eq!( - u32::try_from(distinct.len()).expect("the universe fits u32"), - universe, - "encoded ids stay distinct at N={universe}", - ); - assert!( - image.iter().any(|&wire| wire >= universe), - "the image escapes [0, {universe}): ids no longer bound the universe", - ); - } -} - -#[test] -fn codec_round_trips_a_large_universe_sample() { - let universe = 500_000; - let bound = codec::Universe::new(ProbeRow::new(universe)); - let codec = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - codec::NODE_LABEL, - ); - - let mut seen = HashSet::new(); - for row in (0..universe).step_by(631) { - let wire = codec.encode(ProbeRow::new(row), bound).get(); - assert!(seen.insert(wire), "sampled wire values stay distinct"); - assert_eq!( - codec.decode(codec::WireRow::pinned(wire), bound), - Some(ProbeRow::new(row)) - ); - } -} - -#[test] -fn codec_decodes_only_the_encoded_image() { - let universe = 48_u32; - let bound = codec::Universe::new(ProbeRow::new(universe)); - let codec = - codec::RowCodec::::derive(&secret(b"secret"), codec_generation(), b"test"); - let image: HashSet = (0..universe) - .map(|row| codec.encode(ProbeRow::new(row), bound).get()) - .collect(); - - // A probe sweep outside the image answers None - including the - // low dense range the retired [0, N) codec would have occupied. - for wire in (0..10_000).chain([1 << 31, u32::MAX - 1, u32::MAX]) { - match codec.decode(codec::WireRow::pinned(wire), bound) { - Some(row) => { - let row = row.as_u32(); - assert!(row < universe, "decoded rows lie in the universe"); - assert_eq!( - codec.encode(ProbeRow::new(row), bound).get(), - wire, - "a decoding wire value is its row's encoding", - ); - assert!(image.contains(&wire), "decoding values lie in the image"); - } - None => assert!(!image.contains(&wire), "image values decode"), - } - } -} - -#[test] -fn codec_degenerate_universes_stay_closed() { - let codec = - codec::RowCodec::::derive(&secret(b"secret"), codec_generation(), b"test"); - - let empty = codec::Universe::new(ProbeRow::new(0)); - for wire in [0, 1, u32::MAX] { - assert_eq!( - codec.decode(codec::WireRow::pinned(wire), empty), - None, - "an empty universe decodes nothing" - ); - } - - let single = codec::Universe::new(ProbeRow::new(1)); - let wire = codec.encode(ProbeRow::new(0), single); - assert_eq!(codec.decode(wire, single), Some(ProbeRow::new(0))); - assert_eq!( - codec.decode(codec::WireRow::pinned(wire.get().wrapping_add(1)), single), - None, - ); -} - -#[test] -#[should_panic(expected = "the codec encodes rows of the caller's universe")] -fn codec_encode_rejects_out_of_universe_rows() { - let codec = - codec::RowCodec::::derive(&secret(b"secret"), codec_generation(), b"test"); - _ = codec.encode(ProbeRow::new(48), codec::Universe::new(ProbeRow::new(48))); -} - -#[test] -fn codec_separates_secrets_generations_and_labels() { - let universe = 4096; - let bound = codec::Universe::new(ProbeRow::new(universe)); - let base = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - codec::NODE_LABEL, - ); - let other_secret = codec::RowCodec::::derive( - &secret(b"another"), - codec_generation(), - codec::NODE_LABEL, - ); - let other_generation = codec::RowCodec::::derive( - &secret(b"secret"), - "2222222222222222222222222222222222222222222222222222222222222222" - .parse() - .expect("the literal is 64 hex digits"), - codec::NODE_LABEL, - ); - let other_label = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - b"another-label", - ); - - for (name, other) in [ - ("secret", &other_secret), - ("generation", &other_generation), - ("label", &other_label), - ] { - let differing = (0..universe) - .filter(|&row| { - base.encode(ProbeRow::new(row), bound) != other.encode(ProbeRow::new(row), bound) - }) - .count(); - assert!(differing > 0, "a changed {name} changes the mapping"); - } -} - -#[test] -fn codec_derivation_is_deterministic() { - let universe = 4096; - let bound = codec::Universe::new(ProbeRow::new(universe)); - let first = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - codec::NODE_LABEL, - ); - let second = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - codec::NODE_LABEL, - ); - - for row in 0..universe { - assert_eq!( - first.encode(ProbeRow::new(row), bound), - second.encode(ProbeRow::new(row), bound) - ); - } -} - -/// The full-range codec written a second time. -/// -/// From the documented construction and pinned parameter picks rather than from `serve::codec`. -/// -/// Agreement between the two pins the wire mapping itself. A refactor that changes any derived -/// bit fails these tests. -mod codec_reference { - use core::hash::Hasher as _; - - use hkdf::Hkdf; - use sha2::Sha256; - use siphasher::sip::SipHasher24; - - use crate::file::generation::GenerationId; - - /// The pinned Feistel round count. - const ROUNDS: usize = 8; - - /// One universe's reference codec. - pub(super) struct Reference { - /// The universe size `N`. - universe: u32, - /// The per-round SipHash-2-4 keys. - keys: [[u8; 16]; ROUNDS], - } - - impl Reference { - /// Derives the reference codec of one universe. - pub(super) fn derive( - secret: &[u8], - generation: GenerationId, - label: &[u8], - universe: u32, - ) -> Self { - let salt = generation.digest().to_bytes(); - let mut material = [0_u8; 16 * ROUNDS]; - Hkdf::::new(Some(&salt), secret) - .expand(label, &mut material) - .expect("128 octets stay within HKDF-SHA256's expansion bound"); - - let mut keys = [[0_u8; 16]; ROUNDS]; - for (key, chunk) in keys.iter_mut().zip(material.as_chunks::<16>().0) { - *key = *chunk; - } - - Self { universe, keys } - } - - /// Encodes `row`. - pub(super) fn encode(&self, row: u32) -> u32 { - assert!( - row < self.universe, - "the reference shares the producer contract" - ); - self.permute(row) - } - - /// Decodes `wire`: the inverse pass, then the bounds check against the universe. - pub(super) fn decode(&self, wire: u32) -> Option { - let row = self.unpermute(wire); - (row < self.universe).then_some(row) - } - - /// Applies the network once. - /// - /// Round `i` maps `(L, R)` to `(R, L xor F_i(R))` over two 16-bit halves. - fn permute(&self, mut state: u32) -> u32 { - for key in &self.keys { - let left = state >> 16; - let right = state & 0xFFFF; - state = (right << 16) | (left ^ (round(key, right) & 0xFFFF)); - } - - state - } - - /// Applies the inverse network once. - /// - /// Derived from the round's own algebra: the output `(L', R')` of round `i` determines its - /// input as `R = L'` and `L = R' xor F_i(L')`, so the inverse walks the keys in reverse, - /// recovering each round's input from its output. - fn unpermute(&self, mut state: u32) -> u32 { - for key in self.keys.iter().rev() { - let out_left = state >> 16; - let out_right = state & 0xFFFF; - let left = out_right ^ (round(key, out_left) & 0xFFFF); - state = (left << 16) | out_left; - } - - state - } - } - - /// Evaluates one round function. - #[expect( - clippy::cast_possible_truncation, - reason = "the caller masks to the half width; the narrowing keeps the used bits" - )] - fn round(key: &[u8; 16], half: u32) -> u32 { - let mut hasher = SipHasher24::new_with_key(key); - hasher.write(&half.to_le_bytes()); - hasher.finish() as u32 - } -} - -#[test] -fn codec_agrees_with_the_spec_reference() { - for universe in [2_u32, 3, 5, 48, 100, 257, 1025, 4096] { - let bound = codec::Universe::new(ProbeRow::new(universe)); - let shared = secret(b"secret"); - let codec = - codec::RowCodec::::derive(&shared, codec_generation(), codec::NODE_LABEL); - let model = codec_reference::Reference::derive( - shared.as_bytes(), - codec_generation(), - codec::NODE_LABEL, - universe, - ); - - for row in 0..universe { - let wire = codec.encode(ProbeRow::new(row), bound).get(); - assert_eq!( - wire, - model.encode(row), - "both expressions of the codec agree at N={universe}, row {row}", - ); - assert_eq!( - model.decode(wire), - Some(row), - "the reference inverts the production encoding at N={universe}, row {row}", - ); - } - - // Both expressions also agree on the misses, because each rejects the same wire ids, so - // exactness of decoding is a property of the two together rather than of one alone. - for wire in (0..2_048).chain([1 << 31, u32::MAX]) { - assert_eq!( - codec - .decode(codec::WireRow::pinned(wire), bound) - .map(hashql_core::id::Id::as_u32), - model.decode(wire), - "both expressions agree on decode at N={universe}, wire {wire}", - ); - } - } -} - -#[test] -fn codec_mappings_survive_universe_growth() { - // The permutation is universe-independent: growing the accepted bound leaves every existing - // wire id fixed, so slots allocated within a generation never move ids already on the wire, - // and a slot's wire id decodes exactly under the bounds that admit it. - let codec = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - codec::NODE_LABEL, - ); - let small = codec::Universe::new(ProbeRow::new(1_000)); - let grown = codec::Universe::new(ProbeRow::new(500_000)); - - for row in 0..1_000 { - let wire = codec.encode(ProbeRow::new(row), small); - assert_eq!( - wire, - codec.encode(ProbeRow::new(row), grown), - "row {row} is stable under growth" - ); - assert_eq!(codec.decode(wire, grown), Some(ProbeRow::new(row))); - } - - // A slot past the base bound decodes under the grown bound alone: the extension changes - // which encodings exist, never how any encoding lays out. - for slot in [1_000, 1_001, 250_000, 499_999] { - let wire = codec.encode(ProbeRow::new(slot), grown); - assert_eq!(codec.decode(wire, grown), Some(ProbeRow::new(slot))); - assert_eq!( - codec.decode(wire, small), - None, - "slot {slot} decoded under the base bound" - ); - } -} - -#[test] -fn codec_stays_injective_at_scale() { - let universe = 300_000; - let bound = codec::Universe::new(ProbeRow::new(universe)); - let codec = codec::RowCodec::::derive( - &secret(b"secret"), - codec_generation(), - codec::NODE_LABEL, - ); - - let image: HashSet = (0..universe) - .map(|row| codec.encode(ProbeRow::new(row), bound).get()) - .collect(); - assert_eq!( - u32::try_from(image.len()).expect("the universe fits u32"), - universe, - "encoded ids stay distinct at scale", - ); -} - -#[test] -fn codec_wire_ids_estimate_the_full_range_never_the_universe() { - let universe = 10_000_u32; - let sample = 100_u32; - let trials = 128_u32; - - // A mapping bias would separate the first block of assignment order from a stride spanning the - // universe. - let block: Vec = (0..sample).collect(); - let spread: Vec = (0..sample).map(|index| index * 97).collect(); - - let mut block_total = 0_u64; - let mut spread_total = 0_u64; - let bound = codec::Universe::new(ProbeRow::new(universe)); - for trial in 0..trials { - let seed = trial.to_le_bytes(); - let codec = codec::RowCodec::::derive( - &secret(&seed), - codec_generation(), - codec::NODE_LABEL, - ); - let widest = |rows: &[u32]| { - rows.iter() - .map(|&row| u64::from(codec.encode(ProbeRow::new(row), bound).get())) - .max() - .expect("the selection is nonempty") - }; - block_total += widest(&block); - spread_total += widest(&spread); - } - - // Averaged over trials, the German-tank estimate m(1 + 1/k) - 1 applied to full-range ids - // recovers the u32 range rather than N, and comparisons stay in the scale of 2^32 · k · trials, - // multiplied out rather than divided. The comparison is regression evidence against gross - // mapping bias, because a codec issuing [0, N) or assignment-ordered ids fails both selections - // by orders of magnitude. A distribution smoke at 128 keys cannot establish - // indistinguishability from a random permutation or bound leakage at corpus observation volume, - // and that stronger property stays a design target. - // - // The tolerance derives from the estimator's own spread. The - // maximum of k uniform draws on [0, M) has variance - // M^2 k / ((k+1)^2 (k+2)); the scaled per-trial statistic - // (k+1) · max has standard deviation M · √(k / (k+2)), and - // the sum over t independent trials spreads by √(t) of that. - // A tolerance of twelve standard deviations never flakes and still binds the distribution - // two-sidedly, about four times tighter than the loose bound it replaces and five orders of - // magnitude away from what the retired [0, N) codec would have produced. - let scaled = |total: u64| total * u64::from(sample + 1); - let target = (1_u64 << 32) * u64::from(sample) * u64::from(trials); - let deviation = - 2.0_f64.powi(32) * (f64::from(trials) * f64::from(sample) / f64::from(sample + 2)).sqrt(); - #[expect( - clippy::cast_possible_truncation, - clippy::cast_sign_loss, - reason = "the tolerance is a positive count far below u64::MAX" - )] - let tolerance = (12.0 * deviation) as u64; - for (name, total) in [("block", block_total), ("spread", spread_total)] { - assert!( - scaled(total).abs_diff(target) < tolerance, - "the {name} selection estimates the full range: {total} total", - ); - } - // The difference of the two sums doubles the variance; its - // tolerance widens by √(2). - #[expect( - clippy::cast_possible_truncation, - clippy::cast_sign_loss, - reason = "the tolerance is a positive count far below u64::MAX" - )] - let difference_tolerance = (12.0 * core::f64::consts::SQRT_2 * deviation) as u64; - assert!( - scaled(block_total.abs_diff(spread_total)) < difference_tolerance, - "the selections' range estimates agree: {block_total} vs {spread_total}", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tests/schedule.rs b/libs/@local/graph/atlas/src/serve/tests/schedule.rs deleted file mode 100644 index ab9db07c4e8..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/schedule.rs +++ /dev/null @@ -1,1127 +0,0 @@ -//! Tests that replay the scope-cascade delivery law against an independent reference. -//! -//! The reference in [`reference`] is the restricted delivery law written a second time, from the -//! documented contract rather than from `serve::schedule`. It gathers the visible rows with their -//! pinned keys and ranks, then assigns first-occupant buckets to the complete key depth. It clamps -//! those buckets into the resolved catch-all and delivers contiguous bucket intervals in `(bucket, -//! key, rank)` order. Agreement across the battery pins the delivered rows, their order, the -//! per-bucket runs, the child bitmask, and the root's global metadata at once, and because the -//! reference reads nothing but the visible rows, every agreement is a witness that production -//! delivery reads nothing but the visible rows either. - -#![expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" -)] - -use alloc::sync::Arc; -use std::collections::{HashMap, HashSet}; - -use hashql_core::id::Id as _; - -use super::{ - Bound, CutOffset, EdgesLimits, FIXTURE_LOD, FULL, HEAD, ROW_IDS, TileCoordinate, TileLimits, - UntouchedStore, View, children_of, codec, decode_rows, edges_request, entity_string_of, - expected_edges_bytes, head_counts, head_global, mask_hiding, mask_hiding_rows, - open_edge_artifacts, publish, qualifying_columns, request, section, test_codec, wire_columns, -}; -use crate::{ - identity::{BasePosition, NodeRowId}, - morton::{Depth, MortonCell}, - salt::wire::Mode, - serve::{ - Atlas, ViewError, VisibilityProof, - delta::PlacementCohort, - schedule::{ArrivalOverlay, ViewSchedule}, - }, -}; - -/// The restricted delivery law, written a second time. -pub(super) mod reference { - use std::collections::HashSet; - - use hashql_core::id::Id as _; - - use crate::{ - identity::BasePosition, - morton::{Depth, MortonCell, MortonKey}, - salt::wire::Mode, - serve::{Atlas, VisibilityProof}, - }; - - /// One visible row with its base position, pinned key, and pinned rank. - #[derive(Debug, Copy, Clone)] - pub(crate) struct Row { - pub position: u32, - pub key: MortonKey, - pub rank: u32, - } - - /// Gathers the visible rows of `proof` with their generation-layout values. - pub(crate) fn rows(atlas: &Atlas, proof: &VisibilityProof) -> Vec { - let row_ids = atlas.rows.view(); - let ranks = atlas.ranks.view(); - let count = u32::try_from(atlas.morton.count()).expect("fixture counts fit u32"); - - (0..count) - .filter(|&position| proof.contains(row_ids[BasePosition::from_u32(position)])) - .map(|position| Row { - position, - key: atlas.morton.code(BasePosition::from_u32(position)), - rank: ranks[BasePosition::from_u32(position)].as_u32(), - }) - .collect() - } - - /// Assigns first-occupant buckets over exactly `rows`, to the complete key depth. - /// - /// Rank order, coarse to fine: a row takes the shallowest depth at which it is the first - /// representative of its cell, and rows never claiming a cell - co-located at the complete - /// key - take the deepest bucket. - pub(crate) fn buckets(rows: &[Row]) -> Vec { - let mut by_rank: Vec = (0..rows.len()).collect(); - by_rank.sort_unstable_by_key(|&local| rows[local].rank); - - let mut buckets = vec![Depth::MAX.get(); rows.len()]; - let mut assigned: Vec = Vec::new(); - let mut unassigned = by_rank; - for depth in 0..=Depth::MAX.get() { - let depth = Depth::new(depth).expect("depths at or below MAX are valid"); - let represented: HashSet = assigned - .iter() - .map(|&local| rows[local].key.prefix(depth)) - .collect(); - let mut claimed = HashSet::new(); - - unassigned.retain(|&local| { - let cell = rows[local].key.prefix(depth); - if represented.contains(&cell) || !claimed.insert(cell) { - return true; - } - - buckets[local] = depth.get(); - assigned.push(local); - false - }); - } - - buckets - } - - /// One reference delivery of positions in wire order, with the head's run vocabulary. - #[derive(Debug)] - pub(crate) struct Delivery { - pub positions: Vec, - pub runs: Vec, - } - - /// The scope schedule of one view at offset `k`, replayed from the law. - #[derive(Debug)] - pub(crate) struct Schedule { - rows: Vec, - clamped: Vec, - span: u8, - k: u8, - deepest: u8, - } - - impl Schedule { - /// Builds the reference schedule over `rows` at offset `k`. - pub(crate) fn new(rows: Vec, span: u8, max_tile_depth: u8, k: u8) -> Self { - let deepest = max_tile_depth + span + k; - assert!(deepest <= Depth::MAX.get(), "the battery stays on the grid"); - let clamped = buckets(&rows) - .into_iter() - .map(|bucket| bucket.min(deepest)) - .collect(); - - Self { - rows, - clamped, - span, - k, - deepest, - } - } - - /// The resolved cut of zoom `z`. - fn cut(&self, z: u8) -> u8 { - z + self.span + self.k - } - - /// One bucket's positions inside `cell`, ascending by `(key, rank)`. - fn run(&self, bucket: u8, cell: MortonCell) -> Vec { - let mut hits: Vec<&Row> = self - .rows - .iter() - .zip(&self.clamped) - .filter(|&(row, &clamped)| clamped == bucket && cell.contains(row.key)) - .map(|(row, _)| row) - .collect(); - hits.sort_unstable_by_key(|row| (row.key, row.rank)); - hits.into_iter().map(|row| row.position).collect() - } - - /// The delivery of `(z, cell, mode)`, made only of contiguous bucket intervals. - pub(crate) fn delivery(&self, z: u8, cell: MortonCell, mode: Mode) -> Delivery { - let cut = self.cut(z); - let first = match (mode, z) { - (Mode::Total, _) | (Mode::Delta, 0) => 0, - (Mode::Delta, _) => cut, - }; - - let mut positions = Vec::new(); - let mut runs = Vec::new(); - for bucket in first..=cut { - let run = self.run(bucket, cell); - runs.push(run.len() as u64); - positions.extend(run); - } - - Delivery { positions, runs } - } - - /// The expected child bitmask of `(z, cell)`. - pub(crate) fn children(&self, z: u8, cell: MortonCell) -> u8 { - let cut = self.cut(z); - if cut >= self.deepest { - return 0; - } - let Some(children) = cell.children() else { - return 0; - }; - - let mut bits = 0_u8; - for (index, child) in children.into_iter().enumerate() { - let occupied = self - .rows - .iter() - .zip(&self.clamped) - .any(|(row, &clamped)| clamped > cut && child.contains(row.key)); - bits |= u8::from(occupied) << index; - } - - bits - } - - /// The first zoom whose cumulative schedule delivers `position`, replayed from the law. - /// - /// [`Self::cut`] inverted: the smallest `z` with `clamped <= z + span + k`. Written as a - /// search rather than as the subtraction the implementation uses, so the two derivations - /// share no arithmetic. [`None`] when the view does not hold the position. - pub(crate) fn first_zoom(&self, position: u32) -> Option { - let local = self.rows.iter().position(|row| row.position == position)?; - let bucket = self.clamped[local]; - - (0..=u8::MAX).find(|&z| { - u16::from(bucket) <= u16::from(z) + u16::from(self.span) + u16::from(self.k) - }) - } - - /// The union of the listed tiles' delivered positions: the edges route's bounding set. - /// - /// A tile's delivered set is mode-independent, so the total delivery is the whole answer. - pub(crate) fn delivered_union(&self, tiles: &[(u8, MortonCell)]) -> HashSet { - tiles - .iter() - .flat_map(|&(z, cell)| self.delivery(z, cell, Mode::Total).positions) - .collect() - } - - /// The visible count and deepest occupied bucket expected in the root's global metadata. - pub(crate) fn global(&self) -> (u64, u64) { - let visible = self - .clamped - .iter() - .filter(|&&clamped| clamped <= self.cut(0)) - .count() as u64; - let min_resolution = self.clamped.iter().copied().max().map_or(0, u64::from); - - (visible, min_resolution) - } - } -} - -/// Builds the proof shapes that exercise the cascade. -/// -/// The operator proof aside, the shapes are independent hiding at two rates, the corpus root -/// schedule hidden whole, the densest `z = 1` subtree hidden whole, near-total hiding, and -/// everything hidden. -fn scope_battery(atlas: &Atlas) -> Vec<(&'static str, VisibilityProof)> { - use rand::{RngExt as _, SeedableRng as _}; - use rand_xoshiro::Xoshiro256PlusPlus; - - let row_ids = atlas.row_ids(); - let universe = u32::try_from(row_ids.len()).expect("the fixture universe fits u32"); - let mut rng = Xoshiro256PlusPlus::seed_from_u64(0x5C0F_E5CA); - // The masks below speak the fixture's raw-u32 row vocabulary; narrow checked, at this one - // boundary. - let row_at = |position: BasePosition| row_ids[position].as_u32(); - - let quarter: Vec = (0..universe).filter(|_| rng.random_ratio(1, 4)).collect(); - let most: Vec = (0..universe).filter(|_| rng.random_ratio(4, 5)).collect(); - - // The corpus root schedule hidden whole: the shape that once drove the fill hardest. - let root_cut = Depth::new(FIXTURE_LOD.span.get()).expect("the fixture span is a depth"); - let scheduled: Vec = (BasePosition::MIN..atlas.morton.fenceposts().segment(root_cut).end) - .map(row_at) - .collect(); - - // The densest z = 1 subtree hidden whole. - let densest = (0..2_u32) - .flat_map(|x| (0..2_u32).map(move |y| (x, y))) - .max_by_key(|&(x, y)| { - let cell = MortonCell::new(Depth::new(1).expect("1 is a depth"), x, y) - .expect("the z = 1 grid is on the key width"); - Depth::all() - .map(|bucket| { - let run = atlas.morton.run(bucket, cell); - run.end.as_usize() - run.start.as_usize() - }) - .sum::() - }) - .expect("the z = 1 grid is nonempty"); - let densest_cell = MortonCell::new(Depth::new(1).expect("1 is a depth"), densest.0, densest.1) - .expect("the densest cell is on the key width"); - let subtree: Vec = Depth::all() - .flat_map(|bucket| atlas.morton.run(bucket, densest_cell)) - .map(row_at) - .collect(); - - let sparse: Vec = (0..universe) - .filter(|&row| !row.is_multiple_of(16)) - .collect(); - let all: Vec = (0..universe).collect(); - - vec![ - ("quarter-hidden", mask_hiding(atlas, &quarter)), - ("most-hidden", mask_hiding(atlas, &most)), - ("schedule-hidden", mask_hiding(atlas, &scheduled)), - ("subtree-hidden", mask_hiding(atlas, &subtree)), - ("three-visible", mask_hiding(atlas, &sparse)), - ("all-hidden", mask_hiding(atlas, &all)), - ] -} - -/// Both expressions of the delivery law agree on every restricted delivery. -/// -/// The sweep compares the wire response with the reference at every proof shape, offset, zoom -/// coordinate, and mode. Each comparison covers the delivered wire ids in order, the per-bucket -/// runs, the delivered count, the child bitmask, and the root's global metadata. The same sweep -/// accumulates every delta chain and checks it against the total response, so -/// accumulation-equals-total covers every proof and offset rather than one fixture. -#[tokio::test] -async fn restricted_delivery_agrees_with_the_scope_cascade_reference() { - let (_generation, atlas) = publish("scope-reference").await; - let node_codec = test_codec(&atlas); - let position_of: HashMap = atlas - .row_ids() - .iter() - .enumerate() - .map(|(position, &row)| (row, u32::try_from(position).expect("positions fit u32"))) - .collect(); - - for (name, proof) in scope_battery(&atlas) { - let rows = reference::rows(&atlas, &proof); - for k in 0..=2_u8 { - let schedule = reference::Schedule::new( - rows.clone(), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - k, - ); - let offset = CutOffset::new(k); - let mut deltas: HashMap<(u8, u32, u32), Vec> = HashMap::new(); - - for z in 0..=FIXTURE_LOD.max_tile_depth { - let cells = 1_u32 << z; - for (x, y) in (0..cells).flat_map(|x| (0..cells).map(move |y| (x, y))) { - let cell = MortonCell::new(Depth::new(z).expect("zooms are depths"), x, y) - .expect("the sweep stays on each zoom's grid"); - - for mode in [Mode::Delta, Mode::Total] { - let at = format!("{name} k={k} {mode:?} {z}/{x}/{y}"); - let bytes = atlas - .tile( - &request(z, x, y, mode), - TileLimits::default(), - Bound::new(&atlas, &proof, offset).view(&atlas), - ) - .expect("the restricted tile serves"); - let head = section(&bytes, HEAD).expect("HEAD is present"); - let (delivered, runs) = head_counts(head); - let wire_rows = - decode_rows(section(&bytes, ROW_IDS).expect("ROW_IDS is present")); - let positions: Vec = wire_rows - .iter() - .map(|&wire| { - let row = node_codec - .decode(codec::WireRow::pinned(wire), atlas.node_universe()) - .expect("delivered wire ids decode"); - position_of[&row] - }) - .collect(); - - let expected = schedule.delivery(z, cell, mode); - assert_eq!( - positions, expected.positions, - "{at} delivers the law's rows" - ); - assert_eq!(runs, expected.runs, "{at} recounts the law's runs"); - assert_eq!( - delivered, - expected.positions.len() as u64, - "{at} counts its rows", - ); - assert_eq!( - u8::try_from(children_of(head)).expect("children fit u8"), - schedule.children(z, cell), - "{at} frontiers the law's children", - ); - - if z == 0 && mode == Mode::Delta { - let (visible, _bounds, min_resolution) = - head_global(head).expect("the root carries global metadata"); - let (expected_visible, expected_resolution) = schedule.global(); - assert_eq!(visible, expected_visible, "{at} counts the root view"); - assert_eq!( - min_resolution, expected_resolution, - "{at} names the deepest occupied scope bucket", - ); - } - - if mode == Mode::Delta { - deltas.insert((z, x, y), positions.clone()); - } else { - // Accumulating the delta chain reproduces the total as a set, - // without duplicates. An ancestor's delta spans its whole wider - // extent, so the chain restricts to this tile's cell before the - // comparison - exactly the accumulated state a client holds for it. - let accumulated: Vec = (0..=z) - .flat_map(|level| { - let shift = z - level; - deltas[&(level, x >> shift, y >> shift)].iter().copied() - }) - .collect(); - let chain_len = accumulated.len(); - let chain: HashSet = accumulated.into_iter().collect(); - assert_eq!(chain.len(), chain_len, "{at} repeats no row down chain"); - let code_of = - |position: u32| atlas.morton.code(BasePosition::from_u32(position)); - let in_extent: HashSet = chain - .into_iter() - .filter(|&position| cell.contains(code_of(position))) - .collect(); - let total: HashSet = positions.iter().copied().collect(); - assert_eq!(in_extent, total, "{at} accumulates to its total"); - } - } - } - } - } - } -} - -/// A resolved cut past the key width refuses at the binding, so no tile request can carry it. -/// -/// Every route takes a bound view, so a refused offset never reaches assembly and the whole-tile -/// refusal holds by construction. -#[tokio::test] -async fn resolved_cut_past_the_key_width_refuses_the_whole_tile() { - let (_generation, atlas) = publish("cut-refusal").await; - let proof = mask_hiding(&atlas, &[0]); - - let schedule = ViewSchedule::of(&atlas, &proof, PlacementCohort::EMPTY); - let result = View::bind( - atlas.grid, - &proof, - atlas.census(&proof), - &schedule, - CutOffset::new(32), - PlacementCohort::EMPTY, - None, - ); - assert!( - matches!(result, Err(ViewError::Schedule(_))), - "a cut past the key width must refuse, got {result:?}", - ); -} - -/// A proof paired with the other contract's schedule refuses at the binding. -/// -/// The refusal moved out of assembly when the delivery inputs became one bound value. No endpoint -/// can receive a mismatched pair. The pair is checked where it is assembled, and [`View::bind`] is -/// the only entry point still accepting the four inputs apart. -#[tokio::test] -async fn mismatched_proof_and_schedule_refuse_the_contract() { - let (_generation, atlas) = publish("contract-refusal").await; - let masked = mask_hiding(&atlas, &[0]); - let scope = ViewSchedule::of(&atlas, &masked, PlacementCohort::EMPTY); - - let corpus = ViewSchedule::Corpus(ArrivalOverlay::empty()); - let corpus_for_masked = View::bind( - atlas.grid, - &masked, - atlas.census(&masked), - &corpus, - CutOffset::ZERO, - PlacementCohort::EMPTY, - None, - ); - assert_eq!( - corpus_for_masked.expect_err("a masked proof must not serve the corpus schedule"), - ViewError::Contract, - ); - - let scope_for_full = View::bind( - atlas.grid, - &FULL, - atlas.census(&FULL), - &scope, - CutOffset::ZERO, - PlacementCohort::EMPTY, - None, - ); - assert_eq!( - scope_for_full.expect_err("an operator proof must not serve a scope cascade"), - ViewError::Contract, - ); -} - -/// An operator proof carrying a nonzero offset refuses at the binding. -/// -/// The corpus schedule has one cut per zoom, so an offset into it names bytes no route produces. -/// The case that reaches here is a token sealed before issuance fixed operator offsets at zero. -/// Refusing it keeps the manifest's declared cut and the served bytes one statement. The caller's -/// recovery is a renewal, whose fresh token seals zero. -/// -/// The offset zero case runs beside it, so the refusal is about the value rather than about the -/// pair. -#[tokio::test] -async fn operator_proof_refuses_a_nonzero_offset() { - let (_generation, atlas) = publish("operator-offset-refusal").await; - let corpus = ViewSchedule::Corpus(ArrivalOverlay::empty()); - - let refused = View::bind( - atlas.grid, - &FULL, - atlas.census(&FULL), - &corpus, - CutOffset::new(1), - PlacementCohort::EMPTY, - None, - ); - assert_eq!( - refused.expect_err("an operator proof must not carry a nonzero offset"), - ViewError::Offset(CutOffset::new(1)), - ); - - let bound = View::bind( - atlas.grid, - &FULL, - atlas.census(&FULL), - &corpus, - CutOffset::ZERO, - PlacementCohort::EMPTY, - None, - ); - assert!( - bound.is_ok(), - "the operator pair at offset zero must still bind, got {bound:?}", - ); -} - -/// The tile lists that discriminate the two delivery laws. -/// -/// The deepest zoom's cut is the catch-all under both laws, so a full-grid request delivers the -/// whole visible set either way and witnesses nothing. These lists stop short of it, where a row -/// the corpus cascade buried behind a hidden neighbour is a row the scope cascade lifts. -fn discriminating_tile_lists() -> Vec> { - vec![ - vec![TileCoordinate { z: 0, x: 0, y: 0 }], - (0..2_u32) - .flat_map(|x| (0..2_u32).map(move |y| TileCoordinate { z: 1, x, y })) - .collect(), - ] -} - -/// The edges route bounds its subgraph by the view's own cascade, not by the corpus walk. -/// -/// The listed tiles' delivered rows are what the tile route delivered under the same view, so the -/// expectation replays the reference cascade's cumulative prefixes and induces the subgraph over -/// exactly that union. Reading the corpus schedule under a scope answers about rows the client -/// never received: it drops an edge whose endpoint the scope lifted into a shallower bucket and -/// draws one between endpoints the scope's own tiles have yet to deliver. -/// -/// The sweep carries its own negative control. It counts the cases where the two laws disagree and -/// asserts the count is nonzero, so a fixture that stopped discriminating fails this assertion -/// rather than passing the test for the wrong reason. -#[tokio::test] -async fn scoped_edges_bound_the_view_cascade_delivery() { - let (generation, atlas) = publish("scope-edges").await; - let endpoints: Vec<[u64; 2]> = open_edge_artifacts(&generation) - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs") - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let row_ids = atlas.row_ids(); - let morton = &atlas.morton; - let row_at = |position: u32| row_ids[BasePosition::from_u32(position)].as_u32(); - - let mut discriminated = 0_usize; - for (name, proof) in scope_battery(&atlas) { - let rows = reference::rows(&atlas, &proof); - for k in 0..=2_u8 { - let schedule = reference::Schedule::new( - rows.clone(), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - k, - ); - - for tiles in discriminating_tile_lists() { - let at = format!("{name} k={k} {} tiles", tiles.len()); - let cells: Vec<(u8, MortonCell)> = tiles - .iter() - .map(|&coordinate| { - let depth = Depth::new(coordinate.z).expect("zooms are depths"); - ( - coordinate.z, - MortonCell::new(depth, coordinate.x, coordinate.y) - .expect("the lists stay on each zoom's grid"), - ) - }) - .collect(); - - let delivered: HashSet = schedule - .delivered_union(&cells) - .into_iter() - .map(row_at) - .collect(); - - // The law this cut replaced is the corpus schedule's own runs, masked. Counting - // where it parts from the cascade is what proves the sweep can fail. - let corpus: HashSet = cells - .iter() - .flat_map(|&(z, cell)| { - (0..=(z + FIXTURE_LOD.span.get())) - .filter_map(Depth::new) - .flat_map(move |bucket| morton.run(bucket, cell)) - }) - .map(|position| row_at(position.as_u32())) - .filter(|&row| proof.contains(NodeRowId::from_u32(row))) - .collect(); - discriminated += usize::from(corpus != delivered); - - let bytes = atlas - .edges( - &edges_request(tiles), - EdgesLimits::default(), - Bound::new(&atlas, &proof, CutOffset::new(k)).view(&atlas), - UntouchedStore, - ) - .expect("the scoped edges request serves"); - - let (sources, targets, edge_rows) = qualifying_columns(&endpoints, &delivered); - let columns = wire_columns(&atlas, &sources, &targets, &edge_rows); - assert_eq!( - bytes, - expected_edges_bytes(&generation, true, &columns), - "{at} draws the subgraph its own tiles delivered", - ); - } - } - } - - assert!( - discriminated > 0, - "no case in the sweep parts the corpus walk from the cascade, so it witnesses nothing", - ); -} - -/// A scoped locate names the zoom the view's own cascade first delivers the source at. -/// -/// The fly-to zoom and the partner tie-break both read the source's first visible zoom. Under a -/// scope that zoom must invert the view's own cut `z + span + k` over the view's own cascade. A -/// corpus bucket is a first-occupant result over hidden rows too. Answering from one flies the -/// client to a zoom its own tiles never deliver the source at. It also hands a hidden row the -/// choice of which authorized partners survive the cap. -/// -/// The reference finds the zoom by search where the implementation subtracts. No arithmetic is -/// shared between them. The sweep then counts where the scope answer parts from the corpus one. -#[tokio::test] -async fn scoped_locate_flies_to_the_view_cut_zoom() { - let (_generation, atlas) = publish("scope-locate").await; - let row_ids = atlas.row_ids(); - - let mut discriminated = 0_usize; - let mut resolved = 0_usize; - for (name, proof) in scope_battery(&atlas) { - let rows = reference::rows(&atlas, &proof); - for k in 0..=2_u8 { - let schedule = reference::Schedule::new( - rows.clone(), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - k, - ); - let bound = Bound::new(&atlas, &proof, CutOffset::new(k)); - let view = bound.view(&atlas); - - for &row in &rows { - let at = format!("{name} k={k} position {}", row.position); - let entity = row_ids[BasePosition::from_u32(row.position)].as_u32(); - let source = atlas - .resolve_source( - &view, - &entity_string_of(u8::try_from(entity).expect("fixture rows fit u8")), - ) - .expect("a visible row's own entity id resolves"); - resolved += 1; - - let expected = schedule - .first_zoom(row.position) - .expect("the reference holds every visible row"); - assert_eq!(source.zoom, expected, "{at} names the cut's first zoom"); - - // The fly-to tile is that zoom's cell holding the source, checked by containment - // rather than by replaying the addressing the implementation used. - assert_eq!(source.cell.z, source.zoom, "{at} flies to its own zoom"); - let cell = MortonCell::new( - Depth::new(source.cell.z).expect("zooms are depths"), - source.cell.x, - source.cell.y, - ) - .expect("the fly-to target is on its zoom's grid"); - assert!( - cell.contains(row.key), - "{at} flies to a tile holding the source" - ); - - let corpus = atlas - .morton - .bucket_of(BasePosition::from_u32(row.position)) - .get() - .saturating_sub(FIXTURE_LOD.span.get()); - discriminated += usize::from(corpus != source.zoom); - } - } - } - - assert!(resolved > 0, "the sweep resolved no source at all"); - assert!( - discriminated > 0, - "no case in the sweep parts the corpus first zoom from the cascade's, so it witnesses \ - nothing", - ); -} - -/// A scope admitting every corpus row still receives the restricted contract. -/// -/// The saturated mask is the shape a normalization would be tempted by. Its visible set is the -/// whole corpus. A delivery path that recognized saturation could therefore answer from the corpus -/// schedule without anyone seeing a wrong row. -/// -/// What such a path would lose is the declaration. This caller declared a scope and its manifest -/// sealed a scope's `k`. An offset is a value the corpus contract has no bytes for. -/// -/// The sweep therefore serves the whole grid at three offsets against the cascade reference. It -/// pins beside that the operator contract's refusal of the offset this caller is served at. -#[tokio::test] -async fn an_all_row_scope_serves_the_restricted_contract() { - let (_generation, atlas) = publish("scope-all-rows").await; - let all_rows = mask_hiding(&atlas, &[]); - - assert_eq!( - all_rows.kind(), - crate::serve::visibility::ProofKind::Scope, - "a mask admitting every row is still a scope declaration", - ); - let rows = reference::rows(&atlas, &all_rows); - assert_eq!( - rows.len(), - atlas.row_ids().len(), - "the case needs `V` equal to the corpus, or it is an ordinary scope", - ); - - for k in 0..=2_u8 { - let schedule = reference::Schedule::new( - rows.clone(), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - k, - ); - let offset = CutOffset::new(k); - assert_saturated_scope_grid(&atlas, &all_rows, &schedule, k); - - // The offset this caller is served at is one the corpus contract has no bytes for. - if k > 0 { - assert_eq!( - View::bind( - atlas.grid, - &FULL, - atlas.census(&FULL), - &ViewSchedule::Corpus(ArrivalOverlay::empty()), - offset, - PlacementCohort::EMPTY, - None, - ) - .expect_err("the operator contract admits no offset"), - ViewError::Offset(offset), - ); - } - } -} - -/// Saturated scopes share one cascade, and a scope hiding any row builds its own. -/// -/// A scope schedule is a function of the visible node rows alone, so every scope whose node mask -/// admits the whole corpus builds identical buckets, and one shared allocation answers them all. -/// The link mask never enters the cascade, so a scope masking link rows over a saturated node -/// axis shares it too. The overfire is the bug class on the other side. A scope hiding even one -/// node row must build its own cascade, because the shared one delivers the hidden row. -#[tokio::test] -async fn saturated_scopes_share_one_cascade() { - let (_generation, atlas) = publish("saturated-memo").await; - - let first = ViewSchedule::of(&atlas, &mask_hiding(&atlas, &[]), PlacementCohort::EMPTY); - let second = ViewSchedule::of(&atlas, &mask_hiding(&atlas, &[]), PlacementCohort::EMPTY); - let (ViewSchedule::Scope(first, _), ViewSchedule::Scope(second, _)) = (&first, &second) else { - panic!("a saturated mask is a declared scope and serves the scope contract"); - }; - assert!( - Arc::ptr_eq(first, second), - "two saturated scopes read one cascade" - ); - - let link_masked = ViewSchedule::of( - &atlas, - &mask_hiding_rows(&atlas, &[], &[0]), - PlacementCohort::EMPTY, - ); - let ViewSchedule::Scope(link_masked, _) = &link_masked else { - panic!("a link-masked proof is a declared scope"); - }; - assert!( - Arc::ptr_eq(first, link_masked), - "the link mask never enters the cascade, so a saturated node axis shares it" - ); - - let masked = ViewSchedule::of(&atlas, &mask_hiding(&atlas, &[0]), PlacementCohort::EMPTY); - let ViewSchedule::Scope(masked, _) = &masked else { - panic!("a masked proof is a declared scope"); - }; - assert!( - !Arc::ptr_eq(first, masked), - "a scope hiding a node row builds its own cascade" - ); -} - -/// Sweeps the whole fixture grid at one offset for -/// [`an_all_row_scope_serves_the_restricted_contract`]. -#[track_caller] -fn assert_saturated_scope_grid( - atlas: &Atlas, - all_rows: &VisibilityProof, - schedule: &reference::Schedule, - k: u8, -) { - let offset = CutOffset::new(k); - let node_codec = test_codec(atlas); - let position_of: HashMap = atlas - .row_ids() - .iter() - .enumerate() - .map(|(position, &row)| (row, u32::try_from(position).expect("positions fit u32"))) - .collect(); - - { - for z in 0..=FIXTURE_LOD.max_tile_depth { - let cells = 1_u32 << z; - for (x, y) in (0..cells).flat_map(|x| (0..cells).map(move |y| (x, y))) { - let cell = MortonCell::new(Depth::new(z).expect("zooms are depths"), x, y) - .expect("the sweep stays on each zoom's grid"); - - for mode in [Mode::Delta, Mode::Total] { - let at = format!("all-rows k={k} {mode:?} {z}/{x}/{y}"); - let bytes = atlas - .tile( - &request(z, x, y, mode), - TileLimits::default(), - Bound::new(atlas, all_rows, offset).view(atlas), - ) - .expect("the saturated scope serves"); - let head = section(&bytes, HEAD).expect("HEAD is present"); - let (delivered, runs) = head_counts(head); - let positions: Vec = - decode_rows(section(&bytes, ROW_IDS).expect("ROW_IDS is present")) - .iter() - .map(|&wire| { - position_of[&node_codec - .decode(codec::WireRow::pinned(wire), atlas.node_universe()) - .expect("delivered wire ids decode")] - }) - .collect(); - - let expected = schedule.delivery(z, cell, mode); - assert_eq!( - positions, expected.positions, - "{at} delivers the law's rows" - ); - assert_eq!(runs, expected.runs, "{at} recounts the law's runs"); - assert_eq!( - delivered, - expected.positions.len() as u64, - "{at} counts its rows", - ); - - // At the offset both contracts can serve, they serve the same bytes. That - // coincidence is the reason a byte comparison cannot police this case: a - // normalization into the operator variant would be invisible here. It is - // pinned rather than avoided, so that the day the two responses part, someone - // has to decide which of them moved. - if k == 0 { - let operator = atlas - .tile( - &request(z, x, y, mode), - TileLimits::default(), - Bound::new(atlas, &FULL, CutOffset::ZERO).view(atlas), - ) - .expect("the operator contract serves"); - assert_eq!( - operator, bytes, - "{at} parts from the corpus contract's bytes" - ); - } - } - } - } - - // The offset this caller is served at is one the corpus contract has no bytes for. - if k > 0 { - assert_eq!( - View::bind( - atlas.grid, - &FULL, - atlas.census(&FULL), - &ViewSchedule::Corpus(ArrivalOverlay::empty()), - offset, - PlacementCohort::EMPTY, - None, - ) - .expect_err("the operator contract admits no offset"), - ViewError::Offset(offset), - ); - } - } -} - -/// Under a scope, the locate cap selects among authorized partners by the view's own zoom. -/// -/// The cap's tie-break reads each partner's first visible zoom. Under a scope it must read the -/// zoom the caller's own cascade delivers that partner at. -/// -/// Reading the corpus bucket instead would hand rows the caller cannot see the choice of which -/// authorized partner survives a binding cap. The source would be the same and its authorized -/// neighbours would be the same. The survivors would not. -/// -/// The expectation derives the survivor set from the reference cascade's zoom. The reference finds -/// that zoom by search where the implementation subtracts. -/// -/// What this corpus witnesses is bounded. The cap binds in thirty cases. The survivors are exactly -/// the independent derivation's in all of them. In eight of those cases the two laws deliver some -/// authorized partner at different zooms. The scope-derived input is therefore read rather than -/// assumed. No binding cap turns on it. The fixture's squared distances separate every pair before -/// the zoom is consulted. A case that turns on it needs two authorized partners equidistant from -/// one source and delivered at different cascade zooms. This corpus holds no such pair. The -/// counters assert that state rather than describe it. A corpus that gains such a pair therefore -/// fails here and earns the stronger assertion it then deserves. -#[tokio::test] -async fn a_scoped_locate_cap_selects_among_authorised_partners() { - let (_generation, atlas) = publish("scope-locate-cap").await; - let row_ids = atlas.row_ids(); - - let mut counts = CapCounts::default(); - for (name, proof) in scope_battery(&atlas) { - let rows = reference::rows(&atlas, &proof); - let visible: HashSet = rows - .iter() - .map(|row| row_ids[BasePosition::from_u32(row.position)].as_u32()) - .collect(); - - for k in 0..=2_u8 { - let schedule = reference::Schedule::new( - rows.clone(), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - k, - ); - let bound = Bound::new(&atlas, &proof, CutOffset::new(k)); - let view = bound.view(&atlas); - - for source_row in [0_u32, 1, 2, 3, 5, 7, 40] { - if visible.contains(&source_row) { - counts.add(assert_scoped_caps( - &atlas, - &view, - &schedule, - source_row, - &format!("{name} k={k}"), - )); - } - } - } - } - - let CapCounts { - bound_caps, - parted_inputs, - discriminated, - } = counts; - assert!(bound_caps > 0, "no cap in the sweep binds at all"); - assert!( - parted_inputs > 0, - "no binding cap in the sweep reaches a partner the two laws deliver at different zooms, \ - so the tie-break's scope-derived input is never exercised ({bound_caps} binding caps)", - ); - assert_eq!( - discriminated, 0, - "a binding cap now turns on the zoom the tie-break reads, which is stronger evidence than \ - this case claims: assert the surviving partner directly instead of counting agreement", - ); -} - -/// What one sweep of [`a_scoped_locate_cap_selects_among_authorised_partners`] observed. -#[derive(Debug, Default, Copy, Clone)] -struct CapCounts { - /// Caps that truncated the ego graph. - bound_caps: usize, - /// Caps whose tie-break read a partner zoom the two laws disagree on. - parted_inputs: usize, - /// Caps where that disagreement changed which partners survived. - discriminated: usize, -} - -impl CapCounts { - fn add(&mut self, other: Self) { - self.bound_caps += other.bound_caps; - self.parted_inputs += other.parted_inputs; - self.discriminated += other.discriminated; - } -} - -/// Asserts every binding cap of one scoped source, and counts what the sweep saw. -#[track_caller] -fn assert_scoped_caps( - atlas: &Atlas, - view: &crate::serve::View<'_>, - schedule: &reference::Schedule, - source_row: u32, - at: &str, -) -> CapCounts { - use crate::serve::locate::LocateLimits; - - let position_of = |row: u32| atlas.positions_of_row()[NodeRowId::from_u32(row)]; - let distance_of = |from: u32, to: u32| { - let positions = atlas.positions(); - let origin = positions[position_of(from)]; - let point = positions[position_of(to)]; - let (dx, dy) = (point.x() - origin.x(), point.y() - origin.y()); - // The derivation must mirror the selection key bit for bit: a fused mul_add rounds - // differently and reorders near-ties. - #[expect( - clippy::suboptimal_flops, - reason = "unfused arithmetic mirrors the selection key exactly" - )] - (dx * dx + dy * dy).to_bits() - }; - let corpus_zoom = |row: u32| { - atlas - .morton - .bucket_of(position_of(row)) - .get() - .saturating_sub(FIXTURE_LOD.span.get()) - }; - let scope_zoom = |row: u32| { - schedule - .first_zoom(position_of(row).as_u32()) - .expect("an authorized partner is in the view") - }; - let partner_of = |edge: crate::serve::neighbourhood::ServedEdge| { - let edge = super::fitted(edge); - if edge.source.as_u32() == source_row { - edge.target.as_u32() - } else { - edge.source.as_u32() - } - }; - - let source = atlas - .resolve_source( - view, - &entity_string_of(u8::try_from(source_row).expect("fixture rows fit u8")), - ) - .expect("a visible source resolves"); - let full = atlas.locate_subgraph(source, LocateLimits::default(), view); - - let mut counts = CapCounts::default(); - for cap in 0..full.edges.len() { - let at = format!("{at} ego({source_row}) cap {cap}"); - counts.bound_caps += 1; - - let survivors = |zoom: &dyn Fn(u32) -> u8| { - let mut ordered = full.edges.clone(); - ordered.sort_unstable_by_key(|&(edge, id)| { - let partner = partner_of(edge); - (distance_of(source_row, partner), zoom(partner), id) - }); - ordered.truncate(cap); - ordered.sort_unstable_by_key(|&(_, id)| id); - ordered - }; - - let subgraph = atlas.locate_subgraph( - source, - LocateLimits { - edges: u32::try_from(cap).expect("fixture edge counts are small"), - ..LocateLimits::default() - }, - view, - ); - assert!( - !subgraph.complete, - "{at} does not bind, so it selects nothing", - ); - - let expected = survivors(&scope_zoom); - assert_eq!(subgraph.edges, expected, "{at} keeps other partners"); - - // The delivered nodes are exactly the survivors' partners beside the source. - let mut expected_rows: Vec = vec![source_row]; - let mut partners: Vec = expected - .iter() - .map(|&(edge, _)| partner_of(edge)) - .filter(|&row| row != source_row) - .collect(); - partners.sort_unstable(); - partners.dedup(); - expected_rows.extend(partners); - let mut delivered: Vec = super::delivered_row_ids(atlas, &subgraph) - .iter() - .map(|row| row.as_u32()) - .collect(); - delivered.sort_unstable(); - expected_rows.sort_unstable(); - assert_eq!(delivered, expected_rows, "{at} delivers other partners"); - - counts.discriminated += usize::from(survivors(&corpus_zoom) != expected); - counts.parted_inputs += usize::from(full.edges.iter().any(|&(edge, _)| { - let partner = partner_of(edge); - scope_zoom(partner) != corpus_zoom(partner) - })); - } - - counts -} diff --git a/libs/@local/graph/atlas/src/serve/tests/withdrawal.rs b/libs/@local/graph/atlas/src/serve/tests/withdrawal.rs deleted file mode 100644 index 623ef3cf25a..00000000000 --- a/libs/@local/graph/atlas/src/serve/tests/withdrawal.rs +++ /dev/null @@ -1,903 +0,0 @@ -//! Admission-subtraction witnesses: the ingress snapshot's withdrawn rows leave served tiles. -//! -//! Every case runs the served assembly path with a real published snapshot, folded from feed -//! events exactly as the consumer folds them, so the witnesses cover the crossing from feed -//! events to served bytes rather than the subtraction walk alone. The corpus-proof case is the -//! exposure the register exists to close, because `full_visibility` builds no masks and nothing -//! else subtracts on that path. Each case carries a same-path negative control, which repeats the -//! request and the walk under a snapshot whose withdrawals touch nothing the tile delivers. -//! -//! The fold cases pin the law's resolution-time form. A folded scoped proof and the admission -//! walk hide the same withdrawn rows, a corpus proof declines the fold whole, and the root's -//! global aggregates follow the folded view. The split's other half - the issuance's occupancy -//! input ignoring the fold - is the cache entry's own contract, witnessed beside its type. - -use hashql_core::{ - collections::fast_hash_set, - id::{Id as _, IdSlice}, -}; - -use super::{ - Artifacts, Atlas, Bound, EdgesLimits, FIXTURE_LOD, FULL, HEAD, HashSet, Mode, ROW_IDS, - TileHead, TileLimits, TileResponse, UntouchedStore, coordinate_of, edges_request, - expected_edges_bytes, extremes_vacating_a_root_cell, fixture_row_ids, head_global, mask_hiding, - open_artifacts, open_edge_artifacts, publish, qualifying_columns, request, section, test_codec, - walk, wire_columns, withdrawing, -}; -use crate::{ - bitset::CompressedBitSet, - identity::{BasePosition, EdgeRowId, NodeRowId}, - math::{Bounds2, Vec2}, - morton::{Depth, MortonCell}, - salt::wire::tile::{DeliveredSet, GlobalHead, TileCoordinate}, - serve::{ - VisibilityProof, - delta::{DeltaSnapshot, PlacementCohort}, - walk::full::occupied_children, - }, -}; - -/// Serves one tile under `proof` with `delta` as the request's ingress capture. -fn tile_with( - atlas: &Atlas, - proof: &crate::serve::VisibilityProof, - delta: Option<&DeltaSnapshot>, - tile: &crate::serve::TileRequest, -) -> Vec { - let mut bound = Bound::of(atlas, proof); - if let Some(delta) = delta { - bound = bound.withdrawing(delta); - } - - atlas - .tile(tile, TileLimits::default(), bound.view(atlas)) - .expect("the fixture tile serves") -} - -/// The corpus-proof root subtracts a withdrawn row, splitting its delivered range. -/// -/// Byte-exact against the encoder. The withdrawn position leaves the range and the owning run -/// decrements, while the root's global aggregates stay generation-computed. A control snapshot -/// withdrawing a fitted row the root does not deliver walks the same subtraction to identical -/// bytes, and one withdrawing an unfitted identity skips the walk whole. -#[tokio::test] -async fn corpus_subtract_splits_range() { - let (generation, atlas) = publish("withdrawal-root").await; - let Artifacts { - quad, - morton, - coordinates, - rows, - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - // The root delta delivers buckets `0..=span` as one contiguous range. - let lengths = &morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())]; - let delivered: u64 = lengths.iter().sum(); - let end = u32::try_from(delivered).expect("fixture counts fit u32"); - assert!(delivered > 2, "the witness needs an interior position"); - - // Withdraw the row at base position 1, an interior position of the root's range, so the - // subtraction must split rather than trim. - let withdrawn_position = 1_usize; - let seed = u8::try_from(row_ids[withdrawn_position]).expect("fixture rows fit u8"); - let snapshot = withdrawing(&atlas, &[seed]); - assert!(snapshot.withdraws_any_node(), "the fold resolved the row"); - - let tile = request(0, 0, 0, Mode::Delta); - let bytes = tile_with(&atlas, &FULL, Some(&snapshot), &tile); - - // Position 1's owning run is bucket 0 when that bucket holds more than one point, and the - // next occupied bucket otherwise. - let mut runs: Vec = lengths - .iter() - .map(|&length| u32::try_from(length).expect("fixture counts fit u32")) - .collect(); - let mut consumed = 0_u32; - let owner = runs - .iter() - .position(|&run| { - consumed += run; - consumed > 1 - }) - .expect("the root delivers past position 1"); - runs[owner] -= 1; - - let node_codec = test_codec(&atlas); - let wire_rows: Vec<_> = row_ids - .iter() - .map(|&row| node_codec.encode(NodeRowId::from_u32(row), atlas.node_universe())) - .collect(); - let expected = TileResponse { - head: TileHead { - generation: atlas.generation().digest(), - variant: 0, - coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, - mode: Mode::Delta, - first_bucket: 0, - runs: &runs, - // The corpus aggregates stay generation-computed. A withdrawn extreme point keeps - // stretching the reported extent until refit, and the corpus census never subtracts. - global: Some(GlobalHead { - visible: delivered, - bounds: Some( - Bounds2::new(Vec2::new(-1.0, -1.0), Vec2::new(1.0, 1.0)) - .expect("the wire square is a valid extent"), - ), - min_resolution: morton - .fenceposts() - .lengths() - .iter() - .rposition(|&length| length > 0) - .map_or(0, |bucket| bucket as u64), - }), - children: (0..4).fold(0_u8, |bits, quadrant| { - bits | (u8::from(quad.nodes()[0].child(quadrant).is_some()) << quadrant) - }), - }, - delivered: DeliveredSet::Ranges(&[ - BasePosition::from_u32(0)..BasePosition::from_u32(1), - BasePosition::from_u32(2)..BasePosition::from_u32(end), - ]), - positions: IdSlice::from_raw(points), - rows: IdSlice::from_raw(&wire_rows), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - } - .encode(); - assert_eq!(bytes, expected, "the split subtraction is byte-exact"); - - // Same-path control: a fitted row past the root's cut walks the same subtraction and - // changes nothing this tile serves. - let baseline = tile_with(&atlas, &FULL, None, &tile); - let undelivered = u8::try_from(row_ids[usize::try_from(delivered).expect("counts fit usize")]) - .expect("fixture rows fit u8"); - let deeper = withdrawing(&atlas, &[undelivered]); - assert!(deeper.withdraws_any_node(), "the control resolves a row"); - assert_eq!( - tile_with(&atlas, &FULL, Some(&deeper), &tile), - baseline, - "withdrawing an undelivered row moves no byte", - ); - - // Skip-path control: an unfitted identity resolves into no row bitset, so the walk skips. - let unfitted = withdrawing(&atlas, &[0xC8]); - assert!(!unfitted.withdraws_any_node(), "nothing resolved to a row"); - assert_eq!( - tile_with(&atlas, &FULL, Some(&unfitted), &tile), - baseline, - "an empty withdrawn projection serves the baseline bytes", - ); -} - -/// A scoped view's gathered delivery drops the withdrawn point and only that point. -/// -/// The scoped root gathers positions from the scope cascade, the list-shaped delivered set. The -/// witness reads the served columns. The withdrawn wire id leaves `ROW_IDS` while the count -/// drops by exactly one, and the encoder's `sum(runs) == delivered` assertion has already -/// vouched for the head. The negative control withdraws a row the mask already hides, through -/// the same path. -#[tokio::test] -async fn scoped_drops_withdrawn_only() { - let (generation, atlas) = publish("withdrawal-scoped").await; - let Artifacts { rows, .. } = open_artifacts(&generation); - let row_ids = fixture_row_ids(&rows); - - // A masked proof, so the view serves its own cascade in the gathered shape. - let hidden = 0_u32; - let proof = mask_hiding(&atlas, &[hidden]); - let withdrawn = row_ids[1]; - assert_ne!(withdrawn, hidden, "the witness row must stay visible"); - let seed = u8::try_from(withdrawn).expect("fixture rows fit u8"); - - let tile = request(0, 0, 0, Mode::Delta); - let baseline = tile_with(&atlas, &proof, None, &tile); - let subtracted = tile_with(&atlas, &proof, Some(&withdrawing(&atlas, &[seed])), &tile); - - let decode = |bytes: &[u8]| -> Vec { - let (chunks, remainder) = section(bytes, ROW_IDS) - .expect("ROW_IDS is present") - .as_chunks::<4>(); - assert!(remainder.is_empty(), "row sections are whole u32 columns"); - chunks.iter().copied().map(u32::from_le_bytes).collect() - }; - - let node_codec = test_codec(&atlas); - let wire = node_codec - .encode(NodeRowId::from_u32(withdrawn), atlas.node_universe()) - .get(); - let before = decode(&baseline); - let after = decode(&subtracted); - assert!(before.contains(&wire), "the baseline delivers the row"); - assert!(!after.contains(&wire), "the withdrawn row leaves the wire"); - assert_eq!(after.len(), before.len() - 1, "only that point leaves"); - - // Same-path control: withdrawing the row the mask already hides subtracts nothing, because - // the cascade never delivered it. Fixture node row `r` owns seed `r`, so the hidden row's - // identity is its own row id. - let masked_seed = u8::try_from(hidden).expect("fixture rows fit u8"); - assert_eq!( - tile_with( - &atlas, - &proof, - Some(&withdrawing(&atlas, &[masked_seed])), - &tile - ), - baseline, - "a hidden row's withdrawal moves no byte", - ); -} - -/// A tile whose whole delivered set withdraws serves the existing empty shape. -/// -/// The run keeps its positional slot at zero, the range list empties, and the head's -/// generation-computed structure (`children` included) stays what the artifacts say, exactly as -/// the wire's zero-length-entry law reads. The negative control is the baseline: the same cell -/// with no snapshot serves its full run. -#[tokio::test] -async fn all_withdrawn_empty_shape() { - let (generation, atlas) = publish("withdrawal-empty").await; - let Artifacts { - quad, - morton: _, - coordinates, - rows, - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - // A populated non-root cell, whose delta delivery is the node's own run. - let root = MortonCell::new(Depth::MIN, 0, 0).expect("the root cell exists"); - let mut nodes = Vec::new(); - walk(&quad, 0, root, &mut nodes); - let (node, cell) = nodes[1..] - .iter() - .copied() - .find(|&(node, _)| { - let run = quad.nodes()[node as usize].run(); - run.end > run.start - }) - .expect("the fixture quadtree has a populated non-root node"); - let run = quad.nodes()[node as usize].run(); - let coordinate = coordinate_of(cell); - - let seeds: Vec = run - .map(|position| { - u8::try_from(row_ids[usize::try_from(position).expect("fixture positions fit usize")]) - .expect("fixture rows fit u8") - }) - .collect(); - let snapshot = withdrawing(&atlas, &seeds); - - let tile = request(coordinate.z, coordinate.x, coordinate.y, Mode::Delta); - let bytes = tile_with(&atlas, &FULL, Some(&snapshot), &tile); - - let expected = TileResponse { - head: TileHead { - generation: atlas.generation().digest(), - variant: 0, - coordinate, - mode: Mode::Delta, - first_bucket: coordinate.z + FIXTURE_LOD.span.get(), - runs: &[0], - global: None, - children: occupied_children(&quad.nodes()[node as usize]), - }, - delivered: DeliveredSet::Ranges(&[]), - positions: IdSlice::from_raw(points), - rows: IdSlice::from_raw(&[]), - arrivals: IdSlice::from_raw(&[]), - masks: None, - trailer: None, - } - .encode(); - assert_eq!(bytes, expected, "the empty shape is the existing one"); - - let baseline = tile_with(&atlas, &FULL, None, &tile); - assert_ne!(baseline, bytes, "the baseline still serves the run"); -} - -/// Edges subtract withdrawn endpoints from the bounding set and withdrawn links at the rule site. -/// -/// Byte-exact against the independent derivation the existing edges witnesses use. A withdrawn -/// endpoint kills every edge at it, a withdrawn link dies while both endpoints keep serving, and -/// the control snapshot withdrawing an unfitted identity leaves the baseline bytes. -#[tokio::test] -async fn edges_subtract_withdrawn_endpoints_and_links() { - let (generation, atlas) = publish("withdrawal-edges").await; - let artifacts = open_artifacts(&generation); - let row_ids = fixture_row_ids(&artifacts.rows); - let edge_artifacts = open_edge_artifacts(&generation); - let endpoints = edge_artifacts - .endpoints - .u64_le_pairs() - .expect("the endpoint column is little-endian u64 pairs"); - let endpoints: Vec<[u64; 2]> = endpoints - .iter() - .map(|pair| pair.map(zerocopy::U64::get)) - .collect(); - let endpoints = endpoints.as_slice(); - - // The root delivers buckets 0..=m, the head of the base order. - let head: u64 = artifacts.morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let head = usize::try_from(head).expect("fixture counts fit usize"); - let delivered: HashSet = row_ids[..head].iter().copied().collect(); - - let root = TileCoordinate { z: 0, x: 0, y: 0 }; - let serve = |delta: Option<&DeltaSnapshot>| -> Vec { - let mut bound = Bound::of(&atlas, &FULL); - if let Some(delta) = delta { - bound = bound.withdrawing(delta); - } - - atlas - .edges( - &edges_request(vec![root]), - EdgesLimits::default(), - bound.view(&atlas), - UntouchedStore, - ) - .expect("the fixture edges serve") - }; - - let baseline = serve(None); - let (sources, targets, edge_rows) = qualifying_columns(endpoints, &delivered); - assert!(!edge_rows.is_empty(), "the witness needs a delivered edge"); - assert_eq!( - baseline, - expected_edges_bytes( - &generation, - true, - &wire_columns(&atlas, &sources, &targets, &edge_rows), - ), - "the baseline anchors the derivation", - ); - - // A withdrawn endpoint leaves the bounding set, killing every edge at it. - let endpoint = sources[0]; - let survivors: HashSet = delivered - .iter() - .copied() - .filter(|&row| row != endpoint) - .collect(); - let (sources_after, targets_after, rows_after) = qualifying_columns(endpoints, &survivors); - assert!( - rows_after.len() < edge_rows.len(), - "the withdrawn endpoint carried at least one edge", - ); - let node_seed = u8::try_from(endpoint).expect("fixture rows fit u8"); - assert_eq!( - serve(Some(&withdrawing(&atlas, &[node_seed]))), - expected_edges_bytes( - &generation, - true, - &wire_columns(&atlas, &sources_after, &targets_after, &rows_after), - ), - "an endpoint withdrawal kills its edges", - ); - - // A withdrawn link dies as a tombstone while both endpoints keep serving. - let tombstone = edge_rows[0]; - let mut sources_kept = Vec::new(); - let mut targets_kept = Vec::new(); - let mut rows_kept = Vec::new(); - for ((&source, &target), &row) in sources.iter().zip(&targets).zip(&edge_rows) { - if row != tombstone { - sources_kept.push(source); - targets_kept.push(target); - rows_kept.push(row); - } - } - let link_seed = super::EDGE_SEED + u8::try_from(tombstone).expect("fixture edge rows fit u8"); - assert_eq!( - serve(Some(&withdrawing(&atlas, &[link_seed]))), - expected_edges_bytes( - &generation, - true, - &wire_columns(&atlas, &sources_kept, &targets_kept, &rows_kept), - ), - "a link tombstone dies while its endpoints survive", - ); - - // Same-path control: an unfitted identity resolves to no row and moves no byte. - assert_eq!( - serve(Some(&withdrawing(&atlas, &[0xC8]))), - baseline, - "an unresolved withdrawal serves the baseline bytes", - ); -} - -/// Locate answers a withdrawn source as nonexistent and drops a withdrawn partner's edges. -/// -/// Both ingress paths converge on the source check, and the incident walk reaches candidates -/// through the one edge-rule site, so a partner's withdrawal kills its edge with no second -/// mechanism. The control withdraws an identity the subgraph never touches. -#[tokio::test] -async fn locate_withdrawn_refusal() { - let (_generation, atlas) = publish("withdrawal-locate").await; - - // Fixture node row 3 carries exactly one edge, to node row 7. - let source_seed = 3_u8; - let partner_seed = 7_u8; - let source_id = super::entity_string_of(source_seed); - let limits = crate::serve::LocateLimits::default(); - - let baseline_bound = Bound::of(&atlas, &FULL); - let baseline_view = baseline_bound.view(&atlas); - let source = atlas - .resolve_source(&baseline_view, &source_id) - .expect("the fixture source resolves"); - let baseline = atlas.locate_subgraph(source, limits, &baseline_view); - assert_eq!(baseline.edges.len(), 1, "row 3 carries exactly one edge"); - - // A withdrawn source answers as nonexistent on both ingress paths. - let source_gone = withdrawing(&atlas, &[source_seed]); - let bound = Bound::of(&atlas, &FULL).withdrawing(&source_gone); - let view = bound.view(&atlas); - assert!( - atlas.resolve_source(&view, &source_id).is_none(), - "a withdrawn source is unknown", - ); - let wire = test_codec(&atlas).encode( - NodeRowId::from_u32(u32::from(source_seed)), - atlas.node_universe(), - ); - assert!( - atlas.resolve_wire_source(&view, wire).is_none(), - "the wire-keyed path refuses the same way", - ); - - // A withdrawn partner's edge leaves the incident set through the edge-rule site. - let partner_gone = withdrawing(&atlas, &[partner_seed]); - let bound = Bound::of(&atlas, &FULL).withdrawing(&partner_gone); - let view = bound.view(&atlas); - let source = atlas - .resolve_source(&view, &source_id) - .expect("the source itself stays resolvable"); - let subgraph = atlas.locate_subgraph(source, limits, &view); - assert!( - subgraph.edges.is_empty(), - "the withdrawn partner's edge leaves the ego graph", - ); - - // Same-path control: withdrawing an identity outside the subgraph moves nothing. - let unrelated = withdrawing(&atlas, &[9]); - let bound = Bound::of(&atlas, &FULL).withdrawing(&unrelated); - let view = bound.view(&atlas); - let source = atlas - .resolve_source(&view, &source_id) - .expect("the control leaves the source resolvable"); - let control = atlas.locate_subgraph(source, limits, &view); - assert_eq!(control.edges, baseline.edges, "the ego graph is untouched"); - assert_eq!( - control.delivered, baseline.delivered, - "the partners are untouched" - ); -} - -/// Translate answers a withdrawn identity as an absent key in either domain, and a fitted -/// edge dies with its withdrawn endpoints. -/// -/// The identity-domain check runs right after parse, before any row resolution, and the witness -/// runs under full visibility, the path with no mask width to refuse a row, so cohort and -/// snapshot discipline are its only boundary. The fixture edge is edge row 0, endpoints node -/// rows 0 and 1, so withdrawing seed 0 kills it at its source and seed 1 at its target - the -/// next-request death law the neighbourhood read applies. The control withdraws an identity -/// the request never names, no endpoint among them, and must leave the response equal. -#[tokio::test] -async fn translate_answers_withdrawn_identities_as_absent_keys() { - use crate::serve::translate::{TranslateLimits, TranslateRequest}; - - let (_generation, atlas) = publish("withdrawal-translate").await; - let node_id = super::entity_string_of(0); - let edge_id = super::entity_string_of(super::EDGE_SEED); - let ask = || TranslateRequest { - entity_ids: vec![node_id.clone(), edge_id.clone()], - }; - let translate = |delta: Option<&DeltaSnapshot>| { - atlas - .translate( - ask(), - TranslateLimits::default(), - &FULL, - delta, - PlacementCohort::EMPTY, - ) - .expect("the request is under the cap") - }; - - let baseline = translate(None); - assert!(baseline.nodes.contains_key(&node_id), "the node resolves"); - assert!(baseline.edges.contains_key(&edge_id), "the edge resolves"); - - // A withdrawn node identity leaves the nodes map and kills the fitted edge at its - // source endpoint in the same request. - let node_gone = translate(Some(&withdrawing(&atlas, &[0]))); - assert!(!node_gone.nodes.contains_key(&node_id), "an absent key"); - assert!( - !node_gone.edges.contains_key(&edge_id), - "the edge dies with its withdrawn source endpoint" - ); - - // The target endpoint kills it the same way, while the un-withdrawn node keeps resolving. - let target_gone = translate(Some(&withdrawing(&atlas, &[1]))); - assert!( - !target_gone.edges.contains_key(&edge_id), - "the edge dies with its withdrawn target endpoint" - ); - assert_eq!(target_gone.nodes, baseline.nodes, "the node still resolves"); - - // A withdrawn link identity leaves the edges map, the node untouched. - let edge_gone = translate(Some(&withdrawing(&atlas, &[super::EDGE_SEED]))); - assert!(!edge_gone.edges.contains_key(&edge_id), "an absent key"); - assert_eq!(edge_gone.nodes, baseline.nodes, "the node still resolves"); - - // Same-path control: a withdrawal the request never names moves nothing. - assert_eq!( - translate(Some(&withdrawing(&atlas, &[9]))), - baseline, - "an unrelated withdrawal leaves the response equal", - ); -} - -/// A folded scoped proof delivers the subtracted rows, and subtracting over it moves no byte. -/// -/// The claims split by width on purpose. Subtracting the snapshot a proof already folded is -/// byte-exact vacuous, because no delivered set holds a folded row - the identity the skip -/// rests on. Across the fold boundary itself the withdrawn row is absent either way, and the -/// folded cascade may deliver more: the schedule re-levels over the visible rows, promoting a -/// row into the slot the withdrawal freed where the subtracted document keeps the gap - the -/// same refresh-boundary semantics arrivals already have. The control folds a snapshot whose -/// one withdrawal the mask already hides, which must move no byte. -#[tokio::test] -async fn a_folded_proof_delivers_the_subtracted_rows_with_nothing_to_subtract() { - let (_generation, atlas) = publish("withdrawal-fold").await; - - // The withdrawn row sits at base position 1, inside the head the root delivers. Fixture - // node row `r` owns seed `r`, so its identity withdraws as its own row id. - let hidden = 0_u8; - let withdrawn = atlas.rows.view()[BasePosition::from_u32(1)].as_u32(); - assert_ne!( - withdrawn, - u32::from(hidden), - "the witness row must stay visible" - ); - let seed = u8::try_from(withdrawn).expect("fixture rows fit u8"); - - let proof = mask_hiding(&atlas, &[u32::from(hidden)]); - let snapshot = withdrawing(&atlas, &[seed]); - let tile = request(0, 0, 0, Mode::Delta); - - let mut folded = proof.clone(); - folded.fold_withdrawn(&snapshot); - - let baseline = tile_with(&atlas, &proof, None, &tile); - let served = tile_with(&atlas, &folded, None, &tile); - assert_eq!( - served, - tile_with(&atlas, &folded, Some(&snapshot), &tile), - "subtracting the folded snapshot moves no byte", - ); - assert_ne!(served, baseline, "the folded withdrawal bites"); - - let rows_of = |bytes: &[u8]| -> Vec { - let (chunks, remainder) = section(bytes, ROW_IDS) - .expect("ROW_IDS is present") - .as_chunks::<4>(); - assert!(remainder.is_empty(), "row sections are whole u32 columns"); - chunks.iter().copied().map(u32::from_le_bytes).collect() - }; - let wire = test_codec(&atlas) - .encode(NodeRowId::from_u32(withdrawn), atlas.node_universe()) - .get(); - let folded_rows = rows_of(&served); - let subtracted_rows = rows_of(&tile_with(&atlas, &proof, Some(&snapshot), &tile)); - assert!( - !folded_rows.contains(&wire) && !subtracted_rows.contains(&wire), - "the withdrawn row leaves both routes' wires", - ); - assert!( - subtracted_rows.iter().all(|row| folded_rows.contains(row)), - "the folded cascade delivers every subtracted survivor, backfill aside", - ); - - // Same-path control: folding a withdrawal of the row the mask already hides is the vacuous - // fold, because the mask never admitted it. - let mut vacuous = proof; - vacuous.fold_withdrawn(&withdrawing(&atlas, &[hidden])); - assert_eq!( - tile_with(&atlas, &vacuous, None, &tile), - baseline, - "a fold of hidden rows moves no byte", - ); -} - -/// The withdrawal seeds that part every root aggregate from the unfolded view's. -/// -/// The set withdraws the rows attaining the extent's four extremes, the rest of one root cell, -/// and every row of the deepest occupied bucket, so the extent, the count, and the depth all -/// part from the unfolded view's, and an aggregate that fails to follow the fold is a detectable -/// answer on each axis. The root cell is what parts the count on any layout, because a cell -/// keeping one surviving row keeps its representative and delivers the same number of rows as -/// before. Returns the corpus extent and the withdrawn rows beside their seeds. -fn aggregate_parting_seeds( - atlas: &Atlas, - points: &[Vec2], - row_ids: &[u32], - morton: &super::MortonFile, -) -> (Bounds2, Vec, Vec) { - let (corpus, mut withdrawn) = extremes_vacating_a_root_cell(atlas, points, row_ids); - let lengths = morton.fenceposts().lengths(); - let (deepest, _) = lengths - .iter() - .enumerate() - .rfind(|&(_, &length)| length > 0) - .expect("the fixture occupies a bucket"); - let start: u64 = lengths[..deepest].iter().sum(); - let start = usize::try_from(start).expect("fixture counts fit usize"); - let end = start + usize::try_from(lengths[deepest]).expect("fixture counts fit usize"); - withdrawn.extend((start..end).map(|position| row_ids[position])); - withdrawn.sort_unstable(); - withdrawn.dedup(); - assert!( - !withdrawn.is_empty() && withdrawn.len() < points.len(), - "the snapshot withdraws the extremes and the deepest bucket and leaves a non-empty view" - ); - let seeds: Vec = withdrawn - .iter() - .map(|&row| u8::try_from(row).expect("fixture rows fit u8")) - .collect(); - (corpus, withdrawn, seeds) -} - -/// The scoped root's aggregates follow the fold. -/// -/// The fold rewrites the proof's masks, so the scoped census and cascade - the root's global -/// map, extent included - describe the surviving rows rather than the resolution's. A withdrawn -/// row aggregates exactly as a hidden row does: the scoped cascade is visible-only, and the -/// aggregates follow the view. The same view unfolded publishes the resolution's own numbers on -/// every axis, so each equality distinguishes the folded view rather than restating it. The -/// owner ruled the aggregate split at the fold's landing, so the witness pins settled law. -#[tokio::test] -async fn the_folded_scoped_root_publishes_the_folded_views_aggregates() { - let (generation, atlas) = publish("withdrawal-fold-aggregates").await; - let Artifacts { - coordinates, - rows, - morton, - .. - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - let (corpus, withdrawn, seeds) = aggregate_parting_seeds(&atlas, points, &row_ids, &morton); - let snapshot = withdrawing(&atlas, &seeds); - assert!(snapshot.withdraws_any_node(), "the fold resolved the rows"); - - let proof = mask_hiding(&atlas, &[]); - let mut folded = proof.clone(); - folded.fold_withdrawn(&snapshot); - - // The expectations come from the columns and the schedule reference over the folded view: - // the rows of its cascade at or below the root cut, the tight extent of the surviving set, - // and the deepest occupied scope bucket. - let survives = |position: usize| !withdrawn.contains(&row_ids[position]); - let expected_extent = Bounds2::from_points( - (0..points.len()) - .filter(|&position| survives(position)) - .map(|position| points[position]), - ) - .expect("the folded view holds points"); - assert!( - expected_extent.min().x() > corpus.min().x() - && expected_extent.min().y() > corpus.min().y() - && expected_extent.max().x() < corpus.max().x() - && expected_extent.max().y() < corpus.max().y(), - "the withdrawal vacates all four extremes, so the folded extent is strictly inside" - ); - let (expected_visible, expected_deepest) = super::schedule::reference::Schedule::new( - super::schedule::reference::rows(&atlas, &folded), - FIXTURE_LOD.span.get(), - FIXTURE_LOD.max_tile_depth, - 0, - ) - .global(); - - let tile = request(0, 0, 0, Mode::Delta); - let (visible, extent, min_resolution) = head_global( - section(&tile_with(&atlas, &folded, None, &tile), HEAD).expect("HEAD is present"), - ) - .expect("the root publishes its global map"); - - assert_eq!( - visible, expected_visible, - "the published count is the root schedule of the folded view's own cascade" - ); - assert_eq!( - extent, - Some([ - expected_extent.min().x(), - expected_extent.min().y(), - expected_extent.max().x(), - expected_extent.max().y(), - ]), - "the published extent is the folded view's own" - ); - assert_eq!( - min_resolution, expected_deepest, - "the published depth is the folded view's deepest occupied scope bucket" - ); - - // And the same view unfolded publishes the resolution's own numbers, so the assertions - // above distinguish the folded view from the unfolded one rather than restating it. A - // failure here is fixture drift - the withdrawal set no longer moves the aggregate - not a - // defect in the fold. - let (bare_visible, bare_extent, bare_depth) = head_global( - section(&tile_with(&atlas, &proof, None, &tile), HEAD).expect("HEAD is present"), - ) - .expect("the root publishes its global map"); - assert_ne!( - extent, bare_extent, - "fixture drift: the withdrawal set no longer moves the root extent, so the folded-extent \ - equality above has lost its teeth" - ); - assert!( - visible < bare_visible, - "fixture drift: the withdrawal set no longer removes delivered points, so the \ - folded-count equality above has lost its teeth" - ); - assert!( - min_resolution < bare_depth, - "fixture drift: the withdrawal set no longer vacates the deepest occupied bucket, so the \ - folded-depth equality above has lost its teeth" - ); -} - -/// The corpus root's aggregates stay generation-computed under the same withdrawal. -/// -/// The corpus arm serves the withdrawal as an ingress subtraction: full domains decline the -/// fold and the unmasked census reads the artifacts, so a withdrawn extreme point keeps -/// stretching the corpus extent until refit. One snapshot therefore pins both regimes against -/// the scoped witness above, recording the split the owner ruled at the fold's landing. -#[tokio::test] -async fn corpus_root_aggregates_ignore_fold() { - let (generation, atlas) = publish("withdrawal-fold-corpus-aggregates").await; - let Artifacts { - coordinates, - rows, - morton, - .. - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - let (_, _, seeds) = aggregate_parting_seeds(&atlas, points, &row_ids, &morton); - let snapshot = withdrawing(&atlas, &seeds); - assert!(snapshot.withdraws_any_node(), "the fold resolved the rows"); - - let tile = request(0, 0, 0, Mode::Delta); - let full_baseline = head_global( - section(&tile_with(&atlas, &FULL, None, &tile), HEAD).expect("HEAD is present"), - ); - assert_eq!( - head_global( - section(&tile_with(&atlas, &FULL, Some(&snapshot), &tile), HEAD) - .expect("HEAD is present"), - ), - full_baseline, - "the corpus aggregates stay generation-computed under the same withdrawal" - ); -} - -/// The fold leaves a corpus proof admitting everything. -/// -/// Full domains carry no mask to fold, and narrowing one would turn the operator's declared -/// authority into a scope. The admission walk stays that proof's whole withdrawal authority, -/// which the corpus-proof subtraction witnesses above already pin. -#[tokio::test] -async fn the_fold_leaves_a_corpus_proof_admitting_everything() { - let (_generation, atlas) = publish("withdrawal-fold-corpus").await; - - let mut folded = FULL.clone(); - folded.fold_withdrawn(&withdrawing(&atlas, &[1])); - - assert_eq!(folded, FULL, "full domains stay full through the fold"); -} - -/// The fold removes a link tombstone's row and a withdrawn admitted delta identity. -/// -/// The tombstone's endpoints keep serving, because a link carries authorization its endpoints do -/// not imply and a withdrawal runs the same domains in reverse. The admitted delta-link set drops -/// exactly the withdrawn identity, and its sibling stays. -#[tokio::test] -async fn the_fold_removes_link_tombstones_and_withdrawn_delta_identities() { - let (_generation, atlas) = publish("withdrawal-fold-links").await; - - // In the edge domain, the tombstone's row leaves the folded mask and its endpoints stay. - let edge = EdgeRowId::from_u32(0); - let [source, target] = atlas.endpoint_pairs()[edge]; - let mut folded = mask_hiding(&atlas, &[]); - assert!( - folded.verify_edge(edge, source, target).is_some(), - "the mask admits the link before the fold", - ); - - folded.fold_withdrawn(&withdrawing(&atlas, &[super::EDGE_SEED])); - assert!( - folded.verify_edge(edge, source, target).is_none(), - "the folded mask refuses the tombstone's row", - ); - assert!( - folded.verify(source).is_some() && folded.verify(target).is_some(), - "a link tombstone leaves its endpoints serving", - ); - - // An unfitted withdrawal leaves the admitted delta-link identity set. - let withdrawn = super::entity_id_of(0xC8); - let retained = super::entity_id_of(0xC9); - let mut links = fast_hash_set(); - links.insert(withdrawn); - links.insert(retained); - - let mut proof = - VisibilityProof::from_masks(CompressedBitSet::new(), CompressedBitSet::new(), links); - proof.fold_withdrawn(&withdrawing(&atlas, &[0xC8])); - assert!( - !proof.admits_delta_link(withdrawn), - "a withdrawn identity leaves the admitted set", - ); - assert!(proof.admits_delta_link(retained), "its sibling stays"); -} - -/// A withdrawal published after the entry's fold subtracts as the residue. -/// -/// The folded proof carries the earlier publication's row and the request's capture carries the -/// later one's, so the served tile hides both. The residue is idempotent over the fold: the -/// capture re-names the folded row, and subtracting it edits nothing because no delivered set -/// holds it. -#[tokio::test] -async fn a_withdrawal_after_the_fold_subtracts_as_the_residue() { - let (_generation, atlas) = publish("withdrawal-fold-residue").await; - - // Rows the root delivers, taken at base positions 1 and 2. Fixture node row `r` owns - // seed `r`. - let row_ids = atlas.rows.view(); - let earlier = row_ids[BasePosition::from_u32(1)].as_u32(); - let later = row_ids[BasePosition::from_u32(2)].as_u32(); - let seed_of = |row: u32| u8::try_from(row).expect("fixture rows fit u8"); - - let mut folded = mask_hiding(&atlas, &[]); - folded.fold_withdrawn(&withdrawing(&atlas, &[seed_of(earlier)])); - - let capture = withdrawing(&atlas, &[seed_of(earlier), seed_of(later)]); - let tile = request(0, 0, 0, Mode::Delta); - let served = tile_with(&atlas, &folded, Some(&capture), &tile); - - let rows_of = |bytes: &[u8]| -> Vec { - let (chunks, remainder) = section(bytes, ROW_IDS) - .expect("ROW_IDS is present") - .as_chunks::<4>(); - assert!(remainder.is_empty(), "row sections are whole u32 columns"); - chunks.iter().copied().map(u32::from_le_bytes).collect() - }; - let codec = test_codec(&atlas); - let wire_of = |row: u32| { - codec - .encode(NodeRowId::from_u32(row), atlas.node_universe()) - .get() - }; - - let delivered = rows_of(&served); - assert!( - !delivered.contains(&wire_of(earlier)) && !delivered.contains(&wire_of(later)), - "the folded row and the residue row both leave the wire", - ); - assert!( - rows_of(&tile_with(&atlas, &folded, None, &tile)).contains(&wire_of(later)), - "without the capture the later withdrawal still serves, so the residue has teeth", - ); -} diff --git a/libs/@local/graph/atlas/src/serve/tile.rs b/libs/@local/graph/atlas/src/serve/tile.rs deleted file mode 100644 index adb03675632..00000000000 --- a/libs/@local/graph/atlas/src/serve/tile.rs +++ /dev/null @@ -1,804 +0,0 @@ -//! Tile delivery. -//! -//! One `z/x/y` request answered with `SALTILET` envelope bytes, in delta or total mode, with the -//! `TYPE_MASK` column riding requests that colour types. - -use core::{error::Error, fmt}; - -use hashql_core::id::{Id as _, IdSlice}; -use type_system::ontology::id::VersionedUrl; - -use super::{ - Atlas, - colour::{MaskSet, Palette}, - density::ViewOccupancy, - grid, - hydrate::NodeDetails, - schedule::{ArrivalIndex, ArrivalRow, ViewRow}, - view::{View, ViewError}, - visibility::VisibilityProof, - walk::{ - DeliveredPoints, ViewCensus, Walk, full::occupied_children, subtract::subtract_withdrawn, - }, -}; -use crate::{ - dataset::auxiliary::{Icon, Label, Legend}, - file::quad::Node, - morton::MortonCell, - salt::{ - postings::closure::IconSource, - wire::{ - Mode, - tile::{GlobalHead, TileCoordinate, TileHead, TileResponse, TileTrailer}, - }, - }, -}; - -/// The tile endpoint's request limits. -/// -/// Transport configuration with documented defaults, never wire constants: the transport constructs -/// one value and the manifest publishes the same value, so enforcement and advertisement cannot -/// disagree. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub struct TileLimits { - /// Most `coloredTypeIds` entries one request may carry. - /// - /// The manifest publishes this value as `limits.coloredTypeIds`. The cap bounds the - /// `TYPE_MASK` stride, which carries one bit per requested type: at 32 that is four bytes per - /// point. - pub colored_type_ids: u32 = 32, -} - -const impl Default for TileLimits { - fn default() -> Self { - Self { .. } - } -} - -/// A tile request was rejected. -/// -/// Every variant is a named, data-carrying rejection for the transport layer to map onto its error -/// vocabulary; none of them can result from a well-formed request against the serving contract's -/// limits, which the manifest publishes as data. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum TileError { - /// The request carries more `coloredTypeIds` than the cap admits. - Types { - /// The carried id count. - count: usize, - /// The cap the manifest publishes as `limits.coloredTypeIds`. - maximum: u32, - }, - /// The zoom exceeds the generation's deepest served tile. - Depth { - /// The requested zoom. - z: u8, - /// The generation's deepest served zoom. - maximum: u8, - }, - /// The coordinate lies outside the zoom's `2^z` grid. - Grid { - /// The requested zoom. - z: u8, - /// The requested x index. - x: u32, - /// The requested y index. - y: u32, - }, - /// The delivery view did not bind. - /// - /// A binding refusal converts into this variant through [`From`], so one error union carries a - /// route's binding and assembly rejections together. [`Atlas::tile`] takes the view already - /// bound, so its own rejections are all request-shaped. - View(ViewError), -} - -impl fmt::Display for TileError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Types { count, maximum } => { - write!( - fmt, - "the request carries {count} coloredTypeIds where the cap admits {maximum}" - ) - } - Self::Depth { z, maximum } => { - write!(fmt, "zoom {z} exceeds the deepest served tile {maximum}") - } - Self::Grid { z, x, y } => { - write!(fmt, "({x}, {y}) lies outside the 2^{z} tile grid") - } - Self::View(error) => error.fmt(fmt), - } - } -} - -impl Error for TileError {} - -impl From for TileError { - fn from(value: ViewError) -> Self { - Self::View(value) - } -} - -#[derive(Debug, Copy, Clone, PartialEq, Eq, Default, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub(crate) enum TileDetail { - #[default] - Minimal, - Auxiliary, -} - -/// The query context of one tile request: the ratified POST body, every field optional. -#[derive(Debug, Clone, Default, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub(crate) struct TileQuery { - /// The delivery mode, defaulting to delta when the request names none. - #[serde(default)] - pub mode: Mode, - /// Versioned type URLs conditioning the `TYPE_MASK` column, in request order. - /// - /// Entries parse at the transport boundary: a malformed URL rejects the body, while a - /// well-formed URL this generation never ingested is legal and reads zero bits. - #[serde(default)] - #[schemars(with = "Vec")] - pub colored_type_ids: Vec, - /// Whether the response carries the detail trailer. - #[serde(default)] - pub detail: TileDetail, -} - -/// One tile read. -/// -/// The route's coordinate plus the body's query context, joined by the transport layer. -#[derive(Debug, Clone)] -pub(crate) struct TileRequest { - /// The tile address from the route. - pub coordinate: TileCoordinate, - /// The query context from the request body. - pub query: TileQuery, -} - -/// One assembled tile. -/// -/// Everything [`Atlas::encode_tile`] needs except the columns it gathers at encode time. -/// -/// The document owns its derived data, so it crosses thread boundaries between assembly, hydration, -/// and encoding. The envelope orders hydration last, and the split mirrors that order: assembly and -/// encoding are CPU-bound, hydration awaits the store between them. -#[derive(Debug)] -struct TileDocument { - coordinate: TileCoordinate, - mode: Mode, - first_bucket: u8, - runs: Vec, - delivered: DeliveredPoints, - children: u8, - global: Option, - mask_set: Option, -} - -impl Atlas { - /// Answers one tile request over its bound delivery view. - /// - /// `SALTILET` envelope bytes, ready to send under `application/vnd.hash.saltile-v1`. A - /// request asking for the detail trailer resolves per-point labels and icons in process, - /// captured display first and the generation's own payloads second, so every section of - /// the envelope assembles from the opened artifacts and the view's own cohort alone. - /// - /// # Errors - /// - /// As [`Atlas::assemble_tile`]. - pub(crate) fn tile( - &self, - request: &TileRequest, - limits: TileLimits, - view: View<'_>, - ) -> Result, TileError> { - let document = self.assemble_tile(request, limits, &view)?; - - let arrivals = view.arrivals(); - - let details = match request.query.detail { - TileDetail::Minimal => None, - TileDetail::Auxiliary => { - let mut labels = Vec::with_capacity(document.delivered.count()); - let mut icons = Vec::with_capacity(document.delivered.count()); - - // Hoisted once per response: a captureless cohort answers no overlay read, - // so the per-row identity lookup below runs only when a capture could answer - // it, and a base tile under no cohort keeps its pre-overlay path exactly. - let cohort = view.cohort(); - let overlaid = cohort.captures_any(); - - for row in document.delivered.iter() { - match row { - ViewRow::Base(position) => { - let row = self.rows.view()[position]; - // Captured display first, generation payload second (the - // register's own precedence), so a revised fitted identity serves - // its freshest label. - let label = if overlaid { - self.node_ids.id(row).and_then(|id| cohort.legend_of(id)) - } else { - None - } - .map_or_else( - || { - self.node_ids - .payload_of(row) - .map_or(Label::EMPTY, |legend| legend.label()) - }, - Legend::label, - ); - - labels.push(label); - - if let Some(types) = self.postings.direct_types(position) - && let Some((_, icon, _)) = types - .iter() - .enumerate() - .filter_map(|(index, &r#type)| { - let IconSource { source, depth } = - self.closure.icon_source(r#type)?; - - self.ontology_ids - .payload_of(source) - .map(|icon| (index, icon, depth)) - }) - .min_by_key(|&(index, _, depth)| (depth, index)) - { - icons.push(icon); - } else { - icons.push(Icon::empty()); - } - } - ViewRow::Arrival(index) => { - let arrival = &arrivals[index]; - let legend: &Legend = arrival.legend.as_ref(); - labels.push(legend.label()); - - // The legend names its representative as an ontology row. The - // baked closure artifact resolves the generation's own types, and - // a row past its domain is an allocated one, whose icon the - // cohort's snapshot recorded at allocation. - let representative = legend.representative_ontology(); - let icon = if representative.as_usize() < self.closure.types() { - self.closure - .icon_source(representative) - .and_then(|IconSource { source, .. }| { - self.ontology_ids.payload_of(source) - }) - .unwrap_or(Icon::empty()) - } else { - cohort.allocated_icon_of(representative).expect( - "the arrival table and the cohort derive from one snapshot, \ - which recorded an icon at every row it allocated", - ) - }; - icons.push(icon); - } - } - } - - Some(NodeDetails::new(labels, icons)) - } - }; - - Ok(self.encode_tile(&document, arrivals, details.as_ref())) - } - - /// Censuses the visible view `proof` admits over this generation. - /// - /// A root tile publishes corpus-wide aggregates, resolved once per scope - an unmasked proof - /// answers from the artifacts, and a masked one costs one pass over the base column. Every - /// root-tile request under a scope then reads the census rather than recomputing it, which - /// keeps the walk off the request path. - /// - /// The caller must pass the census taken from this same proof. Assembly reads it as the view's - /// own aggregates without re-deriving them, so a census paired with a different proof publishes - /// that other scope's extent. Pinning a proof to its own generation states the same contract - /// for the same reason. - #[must_use] - pub(crate) fn census(&self, proof: &VisibilityProof) -> ViewCensus { - Walk::of(self, proof).visible_census(self.grid.cut(0), self.positions(), self.bounds) - } - - /// Aggregates the Morton occupancy the delivery-cut policy reads for `proof`. - /// - /// Producing it costs one pass over the code column and an allocation for the visible keys, - /// so a scoped resolution takes it once rather than paying the pass per request. - /// [`VisibilityProof::kind`] is the cheap question a caller asks first. A deployment - /// without a density policy still pays this pass per scoped resolution and then reads - /// nothing from it, a cost accepted for one resolution path rather than two. - #[must_use] - pub(crate) fn visible_occupancy(&self, proof: &VisibilityProof) -> ViewOccupancy { - Walk::of(self, proof).visible_occupancy() - } - - /// Assembles one tile request into its owned document. - /// - /// Every rejection happens here, so encoding cannot fail. - /// - /// Exactly the requests that supply `coloredTypeIds` carry the `TYPE_MASK` column. Bit `i` of a - /// point's mask reads 1 when the point carries the request's type `i` or one of its - /// descendants. An id that resolves to no type in this generation is legal and reads 0 in every - /// mask. - /// - /// An operator view delivers the generation's corpus schedule. Under a scoped view, delivery - /// instead follows the bound cut `z + span + k`, which gives one contiguous interval - /// of scope buckets per response, ascending bucket then Morton order, with every - /// schedule-derived `HEAD` field (`visible` counts, the `children` bitmask, the root's global - /// metadata) reduced over the visible view alone. A hidden point contributes to none of them, - /// so a scope's tile carries no evidence of what the mask removed: a fully masked tile is a - /// tile that never had rows. - /// - /// The view's admitted arrivals join both delivery laws at their own first-occupant - /// buckets. A bound cut merges them from its overlay or its own slots, and the operator - /// delivery splices them into its ranges. Every `HEAD` field folds them in the same way. - /// - /// Version 0 serves the full unfiltered visible set in both modes. The body vocabulary admits - /// no visibility filter, so a request naming one rejects as `invalid-body` rather than - /// receiving bytes that ignore it without saying so. - /// - /// # Errors - /// - /// Returns [`TileError::Types`] when the request carries more `coloredTypeIds` than - /// `limits.colored_type_ids`, [`TileError::Depth`] when the zoom exceeds the generation's - /// deepest served tile, and [`TileError::Grid`] when the coordinate lies outside the zoom's - /// grid. The delivery contract is `view`'s, checked when it bound, so no rejection here is - /// about it. - fn assemble_tile( - &self, - request: &TileRequest, - limits: TileLimits, - view: &View<'_>, - ) -> Result { - if request.query.colored_type_ids.len() > limits.colored_type_ids as usize { - return Err(TileError::Types { - count: request.query.colored_type_ids.len(), - maximum: limits.colored_type_ids, - }); - } - - let coordinate = request.coordinate; - let maximum = self.grid.max_tile_depth(); - if coordinate.z > maximum { - return Err(TileError::Depth { - z: coordinate.z, - maximum, - }); - } - - let cell = grid::cell_of(coordinate).ok_or(TileError::Grid { - z: coordinate.z, - x: coordinate.x, - y: coordinate.y, - })?; - - // The bound cut is the whole contract discriminant: present exactly under a scoped view, - // which is what binding proved when it paired the proof with its schedule. - let scope_cut = view.cut(); - let proof = view.proof(); - - let walk = Walk::of(self, proof); - // Every consumer of `node` sits on the operator path (`is_full` is `cut.is_none()`), so a - // scoped tile skips the quadtree lookup. - let node = if scope_cut.is_none() { - walk.node_of(cell) - } else { - None - }; - - #[expect( - clippy::option_if_let_else, - reason = "both arms borrow `walk` and `node`, which `map_or_else` closures cannot \ - share" - )] - let (mut delivered, first_bucket, mut runs, children) = if let Some(cut) = scope_cut { - let delivery = match request.query.mode { - Mode::Delta => cut.delta(coordinate.z, cell), - Mode::Total => cut.total(coordinate.z, cell), - }; - let children = cut.children(coordinate.z, cell); - - ( - DeliveredPoints::Positions(delivery.rows), - delivery.first_bucket, - delivery.runs, - children, - ) - } else { - self.corpus_delivery(&walk, view, request.query.mode, coordinate.z, cell, node) - }; - - // Admission subtraction: the ingress snapshot's withdrawn rows leave the document here, - // before the trailer gathers any detail, so labels stay aligned to the surviving points - // by construction. An empty projection skips the walk whole. The admission widens for a - // view holding arrivals, because a retained cohort can serve an identity the - // ingress set withdraws without any fitted row entering the bitsets. - if let Some(delta) = view.delta() - && (delta.withdraws_any_node() - || (delta.withdraws_any() && !view.arrivals().is_empty())) - { - let row_ids = self.row_ids(); - let arrivals = view.arrivals(); - subtract_withdrawn(&mut delivered, &mut runs, |row| match row { - ViewRow::Base(position) => delta.withdraws_node(row_ids[position]), - ViewRow::Arrival(index) => delta.withdraws(arrivals[index].identity), - }); - } - - let global = (coordinate.z == 0).then(|| view.root_head()); - - let palette = Palette::of(&request.query.colored_type_ids); - let mask_set = (!palette.is_empty()).then(|| self.resolve_masks(&palette)); - - Ok(TileDocument { - coordinate, - mode: request.query.mode, - first_bucket, - runs, - delivered, - children, - global, - mask_set, - }) - } - - /// Assembles the operator delivery: the corpus fast paths with the view's arrivals merged. - /// - /// The view's arrivals join the delivery as splices, and the child bitmask as occupancy the - /// cumulative schedule has yet to deliver. The deepest zoom's cut is the catch-all, below - /// which nothing exists. A view holding no arrival keeps the borrowed range shape whole. - fn corpus_delivery( - &self, - walk: &Walk<'_>, - view: &View<'_>, - mode: Mode, - z: u8, - cell: MortonCell, - node: Option<&Node>, - ) -> (DeliveredPoints, u8, Vec, u8) { - let mut full = match (mode, z) { - (Mode::Delta, 0) => walk.root_delta(), - (Mode::Delta, _) => walk.delta(z, node), - (Mode::Total, _) => walk.total(z, cell), - }; - let mut children = node.map_or(0, occupied_children); - - let overlay = view.overlay(); - let splices = if overlay.is_empty() { - Vec::new() - } else { - let cut = self.grid.cut(z); - if cut < self.grid.deepest() - && let Some(cells) = cell.children() - { - for (index, child) in cells.into_iter().enumerate() { - if overlay.occupied_past(cut, child) { - children |= 1_u8 << index; - } - } - } - - walk.splice_arrivals(&mut full, overlay, cell) - }; - - let delivered = if splices.is_empty() { - DeliveredPoints::Ranges(full.ranges) - } else { - DeliveredPoints::Spliced { - ranges: full.ranges, - splices, - } - }; - - (delivered, full.first_bucket, full.runs, children) - } - - /// Encodes an assembled document. - /// - /// `SALTILET` envelope bytes, ready to send under `application/vnd.hash.saltile-v1`, with the - /// detail trailer included iff the caller supplies `details`. - /// - /// # Panics - /// - /// This panics when supplied details do not cover the document's delivered points, a transport - /// bug rather than request data. - #[must_use] - fn encode_tile( - &self, - document: &TileDocument, - arrivals: &IdSlice, - details: Option<&NodeDetails>, - ) -> Vec { - let masks = document - .mask_set - .as_ref() - .map(|set| set.memberships(&self.postings)); - - let trailer = details.map(|details| TileTrailer { - labels: details.labels(), - icons: details.icons(), - }); - - let response = TileResponse { - head: TileHead { - generation: self.generation.id().digest(), - variant: 0, - coordinate: document.coordinate, - mode: document.mode, - first_bucket: document.first_bucket, - runs: &document.runs, - global: document.global, - children: document.children, - }, - delivered: document.delivered.as_wire(), - positions: self.positions(), - rows: self.wire_rows(), - arrivals, - masks: masks.as_deref(), - trailer, - }; - - response.encode() - } -} - -#[cfg(test)] -mod tests { - use hash_graph_postgres_store::store::{EntityEvent, EntityUpdate}; - use hash_graph_temporal_versioning::Timestamp; - use hashql_core::id::{Id as _, IdSlice}; - use type_system::{ - knowledge::entity::{ - EntityId, - id::{EntityEditionId, EntityUuid}, - }, - principal::actor_group::WebId, - }; - use uuid::Uuid; - - use super::{Mode, TileCoordinate, TileDetail, TileLimits}; - use crate::{ - dataset::auxiliary::{Icon, Label, OwnedLabel}, - identity::{BasePosition, NodeRowId}, - math::{Bounds2, Vec2}, - postgres::id::{ArchivedEntityId, ArchivedEntityUuid, ArchivedOntologyTypeUuid}, - salt::wire::tile::{DeliveredSet, GlobalHead, TileHead, TileResponse, TileTrailer}, - serve::{ - CutOffset, - delta::{DeltaEvent, DeltaRegister, DeltaRevision, DeltaSnapshot, PlacementCohort}, - hydrate::NodeDetails, - tests::{ - Artifacts, Bound, FIXTURE_LOD, FULL, fixture_row_ids, open_artifacts, publish, - request, test_codec, viewing, - }, - }, - }; - - /// The detailed-tile path. - /// - /// Assembly, entity gathering, and encoding with a hydrated trailer, spliced where the - /// transport awaits the store. - /// - /// The gathered entities carry the fixture's rewritten store-width ids, whose payloads are - /// empty, so all-empty details stand in for the hydrated columns. The encoded bytes must equal - /// the wire document built directly with the trailer. - #[tokio::test] - #[expect( - clippy::single_range_in_vec_init, - reason = "an array of one range is what a root delta delivery IS" - )] - async fn detailed_tiles_encode_the_hydrated_trailer() { - let (generation, atlas) = publish("detailed-trailer").await; - let Artifacts { - quad, - morton, - coordinates, - rows, - } = open_artifacts(&generation); - let points = coordinates.points().expect("wire coordinates are points"); - let row_ids = fixture_row_ids(&rows); - - // The transport path assembles, gathers, hydrates, encodes. - let mut detailed = request(0, 0, 0, Mode::Delta); - detailed.query.detail = TileDetail::Auxiliary; - let document = viewing(&atlas, &FULL, |view| { - atlas - .assemble_tile(&detailed, TileLimits::default(), view) - .expect("assembly serves every detail mode") - }); - - let delivered: u64 = morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let delivered = usize::try_from(delivered).expect("fixture counts fit usize"); - assert_eq!(document.delivered.count(), delivered); - - // Hydration is the transport's store round trip; the encode - // path under test takes its details directly, all-empty here, - // and the encoded envelope equals the directly built wire - // document. - let details = NodeDetails::empty(document.delivered.count()); - let bytes = atlas.encode_tile(&document, IdSlice::from_raw(&[]), Some(&details)); - - let no_labels: Vec<&Label> = vec![Label::EMPTY; delivered]; - let no_icons: Vec<&Icon> = vec![Icon::empty(); delivered]; - let end = u32::try_from(delivered).expect("fixture counts fit u32"); - let expected = TileResponse { - head: TileHead { - generation: atlas.generation().digest(), - variant: 0, - coordinate: TileCoordinate { z: 0, x: 0, y: 0 }, - mode: Mode::Delta, - first_bucket: 0, - runs: &morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .map(|&length| u32::try_from(length).expect("fixture counts fit u32")) - .collect::>(), - global: Some(GlobalHead { - visible: delivered as u64, - // The fixture's random points span both axes, so - // the frame extent anchors at the full wire - // square. - bounds: Some( - Bounds2::new(Vec2::new(-1.0, -1.0), Vec2::new(1.0, 1.0)) - .expect("the wire square is a valid extent"), - ), - min_resolution: morton - .fenceposts() - .lengths() - .iter() - .rposition(|&length| length > 0) - .map_or(0, |bucket| bucket as u64), - }), - children: (0..4).fold(0_u8, |bits, quadrant| { - bits | (u8::from(quad.nodes()[0].child(quadrant).is_some()) << quadrant) - }), - }, - delivered: DeliveredSet::Ranges(&[ - BasePosition::from_u32(0)..BasePosition::from_u32(end) - ]), - arrivals: IdSlice::from_raw(&[]), - positions: IdSlice::from_raw(points), - rows: IdSlice::from_raw(&{ - let node_codec = test_codec(&atlas); - row_ids - .iter() - .map(|&row| node_codec.encode(NodeRowId::from_u32(row), atlas.node_universe())) - .collect::>() - }), - masks: None, - trailer: Some(TileTrailer { - labels: &no_labels, - icons: &no_icons, - }), - } - .encode(); - assert_eq!(bytes, expected, "the trailer path is byte-exact"); - } - - /// A captured display reaches the tile trailer's node labels. - /// - /// The register captures a revised display for the fitted identity at the root delta - /// tile's first delivered slot, and the route's bytes must equal the same assembly encoded - /// with the captured label at that slot alone. The control capture names an undelivered - /// identity and must answer the baseline bytes. - #[tokio::test] - async fn captured_display_reaches_tile_labels() { - let (generation, atlas) = publish("revised-tile-labels").await; - let Artifacts { morton, rows, .. } = open_artifacts(&generation); - let row_ids = fixture_row_ids(&rows); - let delivered: u64 = morton.fenceposts().lengths()[..=usize::from(FIXTURE_LOD.span.get())] - .iter() - .sum(); - let delivered = usize::try_from(delivered).expect("fixture counts fit usize"); - assert!( - delivered < row_ids.len(), - "the charter needs a row the root delta tile does not deliver" - ); - - // The fixture rewrite keys each row's identity by the row id itself. - let identity = |seed: u8| ArchivedEntityId { - web_id: Uuid::from_bytes([seed; 16]).into(), - entity_uuid: ArchivedEntityUuid::from_bytes( - Uuid::from_bytes([seed ^ 0xFF; 16]).into_bytes(), - ), - }; - let revised_seed = u8::try_from(row_ids[0]).expect("fixture rows fit u8"); - let undelivered_seed = u8::try_from(row_ids[delivered]).expect("fixture rows fit u8"); - - let renamed = OwnedLabel::from("renamed"); - let capturing = |seed: u8| -> DeltaSnapshot { - let mut register = DeltaRegister::new( - atlas.node_universe(), - atlas.edge_universe(), - atlas.ontology_universe(), - ); - let event = EntityEvent::Updated(EntityUpdate { - entity: EntityId { - web_id: WebId::new(Uuid::from_bytes([seed; 16])), - entity_uuid: EntityUuid::new(Uuid::from_bytes([seed ^ 0xFF; 16])), - draft_id: None, - }, - edition: EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - archived: false, - changed_at: Timestamp::from_unix_timestamp(1), - }); - register.apply(DeltaEvent::from(&event)); - register - .capture_display( - identity(seed), - EntityEditionId::new(Uuid::from_u128(u128::from(seed))), - &renamed, - Icon::new("revised-icon"), - ArchivedOntologyTypeUuid::from(Uuid::from_u128(0x117C)), - &atlas, - ) - .expect("the fixture ontology domain has room"); - - register.snapshot( - &atlas, - DeltaRevision::FIRST, - Timestamp::from_unix_timestamp(2), - ) - }; - - let mut detailed = request(0, 0, 0, Mode::Delta); - detailed.query.detail = TileDetail::Auxiliary; - - let serve = |snapshot: Option<&DeltaSnapshot>| { - let bound = Bound::resolved( - &atlas, - &FULL, - PlacementCohort::of(snapshot), - CutOffset::ZERO, - ); - atlas - .tile(&detailed, TileLimits::default(), bound.view(&atlas)) - .expect("the tile request is on the served grid") - }; - - // The expected envelope is the same assembly, encoded with the captured label at the - // first delivered slot alone. - let expected = |label: &Label| { - let document = viewing(&atlas, &FULL, |view| { - atlas - .assemble_tile(&detailed, TileLimits::default(), view) - .expect("assembly serves every detail mode") - }); - let mut labels: Vec<&Label> = vec![Label::EMPTY; delivered]; - labels[0] = label; - let icons: Vec<&Icon> = vec![Icon::empty(); delivered]; - - atlas.encode_tile( - &document, - IdSlice::from_raw(&[]), - Some(&NodeDetails::new(labels, icons)), - ) - }; - - let captured = capturing(revised_seed); - assert_eq!( - serve(Some(&captured)), - expected(&renamed), - "the captured label serves at its own slot alone" - ); - - let baseline = serve(None); - assert_eq!( - baseline, - expected(Label::EMPTY), - "the baseline serves the payload labels" - ); - - let unrelated = capturing(undelivered_seed); - assert_eq!( - serve(Some(&unrelated)), - baseline, - "a capture the tile never delivers moves nothing" - ); - } -} diff --git a/libs/@local/graph/atlas/src/serve/translate.rs b/libs/@local/graph/atlas/src/serve/translate.rs deleted file mode 100644 index a27d4f10059..00000000000 --- a/libs/@local/graph/atlas/src/serve/translate.rs +++ /dev/null @@ -1,385 +0,0 @@ -//! Translate. -//! -//! Upstream entity ids to atlas identity, correlating entities the client fetched through the -//! graph API with dots already on screen. -//! -//! The response is two maps keyed by the requested id string echoed verbatim, byte-for-byte and -//! without normalization, so client-side map lookups are literal and which map answers gives the -//! kind. A node answers its row id plus the wire-frame position in the same `f32` domain as the -//! `POSITIONS` column, so a translated entity occupies the same pixel as its tile-delivered dot. -//! Edges answer their endpoints' node row ids. An edge carries no wire id of its own, since the -//! requested entity id is its identity in every binary response, and an edge has no position by -//! nature. -//! -//! An id that resolves to nothing yields an absent key rather than an error or a null entry. -//! Nonexistent ids, draft-suffixed ids (the corpus indexes live entities), and entities the -//! visibility proof hides are indistinguishable by doctrine (missing = denied). A node answers only -//! when the proof admits its row, and an edge only when the proof admits its link row together with -//! both endpoints, so the link domain carries an authorization its endpoints do not imply. Every -//! scope resolves both domains, and a link id the proof does not admit is an absent key, -//! indistinguishable from an id belonging to neither domain. -//! -//! A live post-fit arrival resolves through the entry's placement cohort. A placed identity -//! answers the slot and wire coordinate its first placement took, under the same -//! absent-key law. Identity resolution is cohort-bound, so an identity placed after the scope -//! resolved stays an absent key until the scope refreshes, and a withdrawn identity in the -//! ingress capture answers an absent key in every domain. A fitted edge dies with its -//! endpoints, answering an absent key while the capture withdraws either endpoint's node row. -//! The neighbourhood read applies the same next-request death. A cohort-published post-fit -//! link answers its endpoint rows when the proof's identity set admits it and both endpoints -//! serve. An endpoint serves in its own domain, a fitted row through the proof and a placed -//! arrival through the cohort's retention, so a link attaching an unservable entity answers -//! an absent key. Translation reads the published identity artifacts, the fitted coordinate -//! column, and the entry's retained cohort, never the store. - -use alloc::collections::BTreeMap; -use core::{error::Error, fmt}; - -use hashql_core::id::IdSlice; -use type_system::knowledge::entity::id::ENTITY_ID_DELIMITER; - -use super::{ - Atlas, WireRow, - codec::{RowCodec, Universe}, - delta::{DeltaSnapshot, PlacementCohort}, - visibility::VisibilityProof, -}; -use crate::{ - identity::{BasePosition, EdgeRowId, NodeRowId}, - math::Vec2, - postgres::id::ArchivedEntityId, - salt::fit::prepare::identity::IdentityTableArchive, -}; - -/// The translate endpoint's request cap. -/// -/// Transport configuration with a documented default, never a wire constant: the transport -/// constructs one value and the manifest publishes the same value, so enforcement and advertisement -/// cannot disagree. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub struct TranslateLimits { - /// Most entity ids one request may carry. - /// - /// The manifest publishes this value as `limits.translateEntityIds`. - pub entity_ids: u32 = 1024, -} - -const impl Default for TranslateLimits { - fn default() -> Self { - Self { .. } - } -} - -/// A translate request was rejected. -/// -/// Every variant is a named, data-carrying rejection for the transport layer to map onto its error -/// vocabulary. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum TranslateError { - /// The request lists more entity ids than the cap admits. - Ids { - /// The listed id count. - count: usize, - /// The cap the manifest publishes as `limits.translateEntityIds`. - maximum: u32, - }, -} - -impl fmt::Display for TranslateError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Ids { count, maximum } => { - write!( - fmt, - "the request lists {count} entity ids where the cap admits {maximum}" - ) - } - } - } -} - -impl Error for TranslateError {} - -/// The POST body of one translate read. -#[derive(Debug, Clone, serde::Deserialize, schemars::JsonSchema)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub(crate) struct TranslateRequest { - /// The upstream entity ids to translate, in the `webId~entityUuid` form. - /// - /// Duplicates are legal and collapse. - pub entity_ids: Vec, -} - -/// A node's atlas identity. -/// -/// The row id every binary response uses, plus the node's position in the map's coordinate frame. -#[derive(Debug, Copy, Clone, PartialEq, serde::Serialize, schemars::JsonSchema)] -pub(crate) struct TranslatedNode { - /// The node row id, the `ROW_IDS` domain. - pub id: WireRow, - /// The wire-frame x coordinate, the `POSITIONS` domain. - pub x: f32, - /// The wire-frame y coordinate, the `POSITIONS` domain. - pub y: f32, -} - -/// An edge's atlas identity, its endpoints' node row ids. -/// -/// An edge has no row id of its own. Binary responses identify it by its link entity id, which the -/// requester already holds, so translation answers the two points it joins. -#[derive(Debug, Copy, Clone, PartialEq, Eq, serde::Serialize, schemars::JsonSchema)] -pub(crate) struct TranslatedEdge { - /// The source node row id, the `ROW_IDS` domain. - pub source: WireRow, - /// The target node row id, the `ROW_IDS` domain. - pub target: WireRow, -} - -/// The translate response. -/// -/// Both maps take the requested id string as their key, echoed verbatim, so the answering map gives -/// the kind and lookups survive partial results. -/// -/// The maps serialize in key order, so identical requests yield identical response bytes. -#[derive(Debug, Clone, PartialEq, serde::Serialize, schemars::JsonSchema)] -pub(crate) struct TranslateResponse { - /// Resolved nodes by requested id. - pub nodes: BTreeMap, - /// Resolved edges by requested id. - pub edges: BTreeMap, -} - -impl Atlas { - /// Answers one translate request. - /// - /// Upstream entity ids to atlas row ids, plus wire-frame positions for nodes. Translation - /// consumes the request and reuses each resolved id string as its response key, so it allocates - /// nothing per id. - /// - /// A node id resolves when the proof holds its row. Link ids resolve when the proof holds the - /// link row and both of its endpoints, and answer an absent key otherwise, the same answer an - /// id of neither domain receives. `delta` is the request's ingress withdrawal capture. A - /// withdrawn identity answers an absent key in either domain, and a fitted edge answers the - /// same while the capture withdraws either endpoint's node row - both indistinguishable - /// from every other absence. `cohort` is the entry's own. A placed arrival it holds - /// answers as a node on its slot when the proof admits it, and a published post-fit link - /// answers as an edge on its endpoint rows when the proof's identity set admits the link - /// and both endpoints serve. - /// - /// # Errors - /// - /// Returns [`TranslateError::Ids`] when the request lists more entity ids than - /// `limits.entity_ids`. - pub(crate) fn translate( - &self, - request: TranslateRequest, - limits: TranslateLimits, - proof: &VisibilityProof, - delta: Option<&DeltaSnapshot>, - cohort: PlacementCohort<'_>, - ) -> Result { - translate( - request, - limits, - proof, - delta, - cohort, - &TranslateColumns { - node_ids: &self.node_ids, - edge_ids: &self.edge_ids, - positions: self.positions(), - position_of_row: self.positions_of_row(), - endpoints: self.endpoint_pairs(), - node_codec: &self.node_codec, - universe: cohort.universe(self.node_universe), - fitted: self.node_universe, - }, - ) - } -} - -/// One generation's translate inputs. -/// -/// The columns are the identity tables, the fitted coordinates, and the wire codec. -pub(super) struct TranslateColumns<'generation> { - /// The node identity table. - pub node_ids: &'generation IdentityTableArchive, - /// The edge identity table. - pub edge_ids: &'generation IdentityTableArchive, - /// The wire-coordinate column, base order. - pub positions: &'generation IdSlice, - /// The position permutation, row order. - pub position_of_row: &'generation IdSlice, - /// The endpoint column, mapping each edge row to `[source, target]`. - pub endpoints: &'generation IdSlice, - /// The node universe's wire row-id codec. - pub node_codec: &'generation RowCodec, - /// The accepted row bound, the cohort's universe, read at every wire encode in one answer. - pub universe: Universe, - /// The generation's own fitted row bound, which splits a published link's endpoint into - /// the domain that serves it. - pub fitted: Universe, -} - -impl TranslateColumns<'_> { - /// Returns an edge's endpoint rows. - /// - /// # Panics - /// - /// This panics beyond the edge-row domain, which resolution rules out. - const fn endpoint_rows(&self, edge: EdgeRowId) -> [NodeRowId; 2] { - self.endpoints[edge] - } -} - -/// Resolves a request's ids against one generation's translate columns. -/// -/// Under the authority's visibility proof. -pub(super) fn translate( - request: TranslateRequest, - limits: TranslateLimits, - proof: &VisibilityProof, - delta: Option<&DeltaSnapshot>, - cohort: PlacementCohort<'_>, - columns: &TranslateColumns<'_>, -) -> Result { - debug_assert!( - columns.fitted.size() <= columns.universe.size(), - "the fitted bound lies inside the accepted universe", - ); - - if request.entity_ids.len() > limits.entity_ids as usize { - return Err(TranslateError::Ids { - count: request.entity_ids.len(), - maximum: limits.entity_ids, - }); - } - - // Requests speak the upstream string form and response keys echo it verbatim - the parse - // boundary is the API contract, so resolution starts from the string, never a typed id. - let mut nodes = BTreeMap::new(); - let mut edges = BTreeMap::new(); - for id_string in request.entity_ids { - let Some(key) = parse(&id_string) else { - continue; - }; - - // The identity-domain check covers both maps before any row resolution, and an operator - // proof has no mask width to refuse a withdrawn row, so this is that path's one boundary. - if delta.is_some_and(|delta| delta.withdraws(key)) { - continue; - } - - if let Some(row) = columns.node_ids.row_of(key) { - if proof.verify(row).is_none() { - // Hidden: an absent key, indistinguishable from nonexistent. - continue; - } - - let position = columns.position_of_row[row]; - let point = columns.positions[position]; - nodes.insert( - id_string, - TranslatedNode { - id: columns.node_codec.encode(row, columns.universe), - x: point.x(), - y: point.y(), - }, - ); - } else if let Some(edge) = columns.edge_ids.row_of(key) { - let [source, target] = columns.endpoint_rows(edge); - if proof.verify_edge(edge, source, target).is_none() { - // A hidden link row or endpoint hides the edge: an - // absent key. - continue; - } - - // A withdrawn endpoint kills every edge at it on the next request, the - // neighbourhood read's own rule. The edge's own withdrawal is the identity check - // above; the endpoints are other identities, and this row projection is the one - // boundary that sees them. - if delta - .is_some_and(|delta| delta.withdraws_node(source) || delta.withdraws_node(target)) - { - continue; - } - - edges.insert( - id_string, - TranslatedEdge { - source: columns.node_codec.encode(source, columns.universe), - target: columns.node_codec.encode(target, columns.universe), - }, - ); - } else if let Some(arrival) = cohort.node(key) { - if proof.verify(arrival.id).is_none() { - // Hidden: the scope's own resolution never admitted the arrival. - continue; - } - - nodes.insert( - id_string, - TranslatedNode { - id: columns.node_codec.encode(arrival.id, columns.universe), - x: arrival.position.x(), - y: arrival.position.y(), - }, - ); - } else if let Some(link) = cohort.edge(key) { - if !proof.admits_delta_link(key) { - // Hidden: the scope's own resolution never admitted the link. - continue; - } - - // Publication hands a link's endpoints as rows in the accepted universe. The - // fitted bound splits each into the domain that serves it, and either refused - // endpoint refuses the whole link. - let endpoint = |row: NodeRowId| { - if columns.fitted.contains(row) { - (proof.contains(row) && !delta.is_some_and(|delta| delta.withdraws_node(row))) - .then_some(row) - } else { - let (identity, _) = cohort.node_at(row)?; - - (proof.contains(row) && !delta.is_some_and(|delta| delta.withdraws(identity))) - .then_some(row) - } - }; - let (Some(source), Some(target)) = (endpoint(link.source), endpoint(link.target)) - else { - continue; - }; - - edges.insert( - id_string, - TranslatedEdge { - source: columns.node_codec.encode(source, columns.universe), - target: columns.node_codec.encode(target, columns.universe), - }, - ); - } else { - // Known to neither identity domain nor the cohort: an absent key by contract. - } - } - - Ok(TranslateResponse { nodes, edges }) -} - -/// Parses one upstream entity id into the identity tables' key form. -/// -/// A draft-suffixed id (`webId~entityUuid~draftId`) reads unresolved by contract - the corpus -/// indexes live entities - as does anything that is not two `~`-delimited uuids. -pub(super) fn parse(id: &str) -> Option { - let (web_id, entity_uuid) = id.split_once(ENTITY_ID_DELIMITER)?; - if entity_uuid.contains(ENTITY_ID_DELIMITER) { - return None; - } - - let web_id: uuid::Uuid = web_id.parse().ok()?; - let entity_uuid: uuid::Uuid = entity_uuid.parse().ok()?; - - Some(ArchivedEntityId { - web_id: web_id.into(), - entity_uuid: entity_uuid.into(), - }) -} diff --git a/libs/@local/graph/atlas/src/serve/view.rs b/libs/@local/graph/atlas/src/serve/view.rs deleted file mode 100644 index e2f744a2461..00000000000 --- a/libs/@local/graph/atlas/src/serve/view.rs +++ /dev/null @@ -1,331 +0,0 @@ -//! The delivery vocabulary of one request. -//! -//! A corpus-bearing response reads a proof, a census, a schedule and a cut offset. The proof states -//! which rows the scope may see. The census states what that view aggregates to over the whole -//! corpus. The schedule states which cascade orders its delivery. The cut offset states where that -//! cascade is read. A scope resolution produces the proof, the census and the schedule once, the -//! presented token seals the cut offset per request, and [`View`] binds them together. -//! -//! Binding holds the pairing laws. A census and a schedule each travel with the proof that produced -//! them, and each [`ViewSchedule`] variant pairs with exactly one proof constructor. One -//! constructor checks every pairing at the request boundary, so assembly receives a view already -//! holding its resolved cut and every assembly rejection is about the request. - -use core::{error::Error, fmt}; - -use hashql_core::id::IdSlice; - -use super::{ - Atlas, ViewCensus, VisibilityProof, - cache::CacheEntry, - delta::{DeltaSnapshot, PlacementCohort}, - density::CutOffset, - grid::Grid, - schedule::{ - ArrivalIndex, ArrivalOverlay, ArrivalRow, ScheduleWidthError, ViewSchedule, - cut::ScheduleCut, - }, - visibility::ProofKind, -}; -use crate::{math::Bounds2, salt::wire::tile::GlobalHead}; - -/// A delivery view that does not bind. -/// -/// Every variant names an input this process produced rather than a request the caller shaped. A -/// transport answers [`ViewError::Contract`] and [`ViewError::Schedule`] with its internal problem. -/// [`ViewError::Offset`] is the one a caller can act on. Its token sealed an offset under a -/// contract this process no longer serves, and a fresh issuance reseals it. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum ViewError { - /// The proof and the schedule it travelled with pair the wrong variants. - /// - /// Each proof constructor pairs with exactly one [`ViewSchedule`] variant, and binding refuses - /// a mismatched pair rather than serving either contract. - Contract, - /// The resolved cut offset lies past the key width. - /// - /// A sealed offset resolves against this same generation's schedule, so an out-of-domain value - /// is a defect to surface. Binding refuses it whole rather than clamping it or substituting - /// another schedule. - Schedule(ScheduleWidthError), - /// An operator proof travelled with a nonzero delivery-cut offset. - /// - /// The corpus schedule has one cut per zoom and takes no offset, so an operator view serves at - /// offset zero. A nonzero value means an issuance sealed and declared an offset no route can - /// serve, - /// and binding refuses it rather than answering corpus bytes under a declared cut nothing - /// produced. - Offset(CutOffset), -} - -impl fmt::Display for ViewError { - fn fmt(&self, fmt: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Contract => fmt.write_str( - "the visibility proof and its delivery schedule disagree about the serving \ - contract", - ), - Self::Schedule(error) => error.fmt(fmt), - Self::Offset(offset) => write!( - fmt, - "the operator proof carries the nonzero delivery-cut offset {}, which the corpus \ - schedule cannot serve", - offset.get(), - ), - } - } -} - -impl Error for ViewError {} - -/// One request's bound delivery view. -/// -/// The proof, the census resolved beside it, and the view's schedule read at the request's cut -/// offset. Every assembly path takes this value as its whole statement of visibility, so the -/// endpoints share one delivery vocabulary and one contract check. -/// -/// The value borrows the resolution it binds, which a scope holds for its reuse window, so -/// construction costs only per-request arithmetic and repeats no per-scope work. -#[derive(Debug, Copy, Clone)] -pub(crate) struct View<'scope> { - /// The rows the scope may see. - proof: &'scope VisibilityProof, - /// The tight wire-frame extent of the visible set, [`None`] when it is empty. - bounds: Option, - /// The root delivery, by the proof kind binding established. - root: Root<'scope>, - /// The served grid, the catch-all depth the corpus arm clamps an arrival bucket into. - grid: Grid, - /// The view's arrival overlay, taken from the schedule it bound. - /// - /// A bound cut merges it into every delivery query, and the corpus assembly reads it - /// directly, because the corpus fast paths take no cut. - overlay: &'scope ArrivalOverlay, - /// The entry's placement cohort, the arrivals snapshot its resolution read. - /// - /// The cohort names the accepted row universe every wire decode in the request runs under, - /// and the reverse lookup for the slots past the generation's fitted rows. The view's - /// arrival tables derive from this same snapshot, so the vessels and the decodes agree on - /// one publication. - cohort: PlacementCohort<'scope>, - /// The withdrawal snapshot captured at the request's ingress, absent before the first - /// publication and for a serve that starts no consumer. - /// - /// One capture answers every admission in the request, so the delta-sensitive assembly stays - /// a pure function of the generation, the request, the proof, and this one snapshot. The - /// capture contributes current withdrawals alone. - delta: Option<&'scope DeltaSnapshot>, -} - -/// The root delivery of one view, the one discriminant the assembly paths read. -#[derive(Debug, Copy, Clone)] -enum Root<'scope> { - /// The corpus schedule's root aggregates of an operator proof, read from the artifacts. - Corpus { - /// The points the corpus schedule's root cut delivers. - visible: u64, - /// The deepest occupied bucket. - min_resolution: u64, - }, - /// A scoped proof's own cascade, read at the request's offset. - Scope(ScheduleCut<'scope>), -} - -impl<'scope> View<'scope> { - /// Binds one resolved scope's delivery inputs at `k`. - /// - /// An operator proof binds the corpus schedule with no cut, and a scoped proof its own cascade - /// read at `k`. `delta` is the request's ingress withdrawal capture, carried whole to every - /// admission: binding checks nothing about it, because absence is lawful before the first - /// publication and the snapshot pairs with the request rather than with the scope. - /// - /// Caller requirement: `census` is [`Atlas::census`] of `proof`, `schedule` is - /// [`ViewSchedule::of`] over that same proof, and `cohort` is the resolution's own - the - /// snapshot the schedule folded its arrivals from. Binding checks only that the proof and - /// the schedule name the same serving contract. A census resolved from another proof of the - /// same shape passes that check, so the pairing is the caller's to hold. - /// - /// # Errors - /// - /// Returns [`ViewError::Contract`] when `proof`, `schedule` and `census` do not all name one - /// proof kind, [`ViewError::Offset`] when an operator proof carries a nonzero `k`, and - /// [`ViewError::Schedule`] when `k` resolves past the key width. - pub(super) fn bind( - grid: Grid, - proof: &'scope VisibilityProof, - census: ViewCensus, - schedule: &'scope ViewSchedule, - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - k: CutOffset, - cohort: PlacementCohort<'scope>, - delta: Option<&'scope DeltaSnapshot>, - ) -> Result { - let (root, overlay) = match (proof.kind(), schedule, census) { - (ProofKind::Corpus, ViewSchedule::Corpus(_), ViewCensus::Corpus { .. }) - if k != CutOffset::ZERO => - { - return Err(ViewError::Offset(k)); - } - ( - ProofKind::Corpus, - ViewSchedule::Corpus(overlay), - ViewCensus::Corpus { - visible, - min_resolution, - bounds: _, - }, - ) => ( - Root::Corpus { - visible, - min_resolution, - }, - overlay, - ), - (ProofKind::Scope, ViewSchedule::Scope(scope, overlay), ViewCensus::Scope { .. }) => ( - Root::Scope(scope.cut(overlay, grid, k).map_err(ViewError::Schedule)?), - overlay, - ), - (ProofKind::Corpus, ViewSchedule::Corpus(_), ViewCensus::Scope { .. }) - | ( - ProofKind::Corpus, - ViewSchedule::Scope(..), - ViewCensus::Corpus { .. } | ViewCensus::Scope { .. }, - ) - | ( - ProofKind::Scope, - ViewSchedule::Corpus(_), - ViewCensus::Corpus { .. } | ViewCensus::Scope { .. }, - ) - | (ProofKind::Scope, ViewSchedule::Scope(..), ViewCensus::Corpus { .. }) => { - return Err(ViewError::Contract); - } - }; - - Ok(Self { - proof, - bounds: census.bounds(), - root, - grid, - overlay, - cohort, - delta, - }) - } - - /// Binds one held resolution at `k`. - /// - /// An entry's census and schedule both derive from that entry's own proof, so the pairing holds - /// by construction. - /// - /// # Errors - /// - /// Returns [`ViewError::Offset`] when the entry holds an operator proof and `k` is nonzero, and - /// [`ViewError::Schedule`] when `k` resolves past the key width. [`ViewError::Contract`] is - /// unreachable through this constructor, because the entry pairs its own proof with the - /// schedule built over that proof. - pub(crate) fn of( - atlas: &Atlas, - entry: &'scope CacheEntry, - #[expect( - clippy::min_ident_chars, - reason = "`k` is the delivery-cut offset's name throughout the density contract" - )] - k: CutOffset, - delta: Option<&'scope DeltaSnapshot>, - ) -> Result { - Self::bind( - atlas.grid, - entry.proof(), - *entry.census(), - entry.view_schedule(), - k, - entry.cohort(), - delta, - ) - } - - /// Returns the rows the view may see. - #[must_use] - pub(crate) const fn proof(&self) -> &'scope VisibilityProof { - self.proof - } - - /// Returns the root delivery's deepest occupied bucket, zero for an empty view. - /// - /// The root tile's `HEAD` and the manifest's `scopeSchedule.maxZoom` both read this. - pub(crate) fn min_resolution(&self) -> u64 { - match self.root { - Root::Corpus { min_resolution, .. } => { - let arrivals = self - .overlay - .min_resolution(self.grid.deepest()) - .map_or(0, |bucket| u64::from(bucket.get())); - - min_resolution.max(arrivals) - } - Root::Scope(cut) => cut.min_resolution(), - } - } - - /// Returns the root tile's aggregates over the whole visible set. - /// - /// The corpus arm adds the overlay's arrivals the root delivers to the census's count, exactly - /// as a bound cut folds its own. - pub(crate) fn root_head(&self) -> GlobalHead { - GlobalHead { - visible: match self.root { - Root::Corpus { visible, .. } => { - visible - + self - .overlay - .delivered_through(self.grid.cut(0), self.grid.deepest()) - } - Root::Scope(cut) => cut.root_delivered(), - }, - bounds: self.bounds, - min_resolution: self.min_resolution(), - } - } - - /// Returns the bound scope cut, [`None`] under an operator proof. - pub(super) const fn cut(&self) -> Option> { - match self.root { - Root::Corpus { .. } => None, - Root::Scope(cut) => Some(cut), - } - } - - /// Returns the view's arrival overlay. - /// - /// Empty for a scope that folded its arrivals into its own cascade, and for a view whose - /// resolution read no cohort. - pub(super) const fn overlay(&self) -> &'scope ArrivalOverlay { - self.overlay - } - - /// Views the arrival table the delivered [`ViewRow::Arrival`] vessels address. - /// - /// The bound cut answers under a scoped proof and the overlay under an operator proof, so - /// one table serves each view. - /// - /// [`ViewRow::Arrival`]: super::schedule::ViewRow::Arrival - pub(crate) const fn arrivals(&self) -> &'scope IdSlice { - match self.root { - Root::Scope(cut) => cut.arrivals(), - Root::Corpus { .. } => self.overlay.arrivals(), - } - } - - /// Returns the ingress withdrawal snapshot, [`None`] before the first publication. - pub(super) const fn delta(&self) -> Option<&'scope DeltaSnapshot> { - self.delta - } - - /// Returns the entry's placement cohort, the arrivals snapshot its resolution read. - pub(super) const fn cohort(&self) -> PlacementCohort<'scope> { - self.cohort - } -} diff --git a/libs/@local/graph/atlas/src/serve/visibility.rs b/libs/@local/graph/atlas/src/serve/visibility.rs deleted file mode 100644 index e57375d2efc..00000000000 --- a/libs/@local/graph/atlas/src/serve/visibility.rs +++ /dev/null @@ -1,412 +0,0 @@ -//! The visibility join point. -//! -//! The server-held proof of which rows are visible, and the single path every row ingress -//! factors through. -//! -//! [`VisibilityProof`] is the visible row set `V_u` as a value, the caller-supplied row masks -//! interpreted against the opened generation's row universes. The value carries the row sets alone. -//! Generation, scope, and permission-epoch binding are its caller's contract rather than fields of -//! the type. The proof covers two row domains, node rows and link rows, because a link entity -//! carries authorization of its own that its endpoints do not imply. A third, identity-keyed -//! domain covers delta links, the post-fit links a placement cohort publishes: their edge rows -//! are register-allocated past the baked bound, where no mask built over the generation's -//! universe can name them. Admission therefore keys by entity identity. Reading two entities never -//! implies reading the relation between them, so an edge delivers only when the proof admits its -//! link row and both of its endpoints. A node mask and a link mask are separate types, so nothing -//! reads one where the other belongs. Every assembly path takes a proof by construction. No -//! `Option` exists whose `None` means "everything", and the full-visibility value is a distinct, -//! named constructor rather than a default. -//! -//! [`Atlas::resolve`] is the single resolution path. It decodes the wire id and then tests mask -//! membership, with decode failure, out-of-universe values, and mask misses all collapsing to the -//! same [`None`] before any rendering observes the cause. [`VisibleRow`] has no other constructor, -//! so a row that reaches point-lookup assembly carries its visibility in the type; [`VisibleEdge`] -//! is the same discipline over the link domain, created only by [`VisibilityProof::verify_edge`], -//! which is the delivery rule itself. Set-shaped node paths (tile gathers) mask through -//! [`VisibilityProof::intersect`] and [`VisibilityProof::contains`] wholesale instead of creating -//! a value per row. - -use hashql_core::{ - collections::FastHashSet, - id::{Id, bit_vec::DenseBitSet}, -}; - -use super::{ - Atlas, WireRow, - delta::{DeltaSnapshot, PlacementCohort}, -}; -use crate::{ - bitset::CompressedBitSet, - identity::{EdgeRowId, NodeRowId}, - postgres::id::ArchivedEntityId, -}; - -/// One domain's visible row set, either everything or exactly the mask. -/// -/// The domain is the type parameter, so a node mask and a link mask are values of different types -/// and reading one where the other belongs does not compile. -#[derive(Debug, Clone, PartialEq, Eq)] -enum Rows { - /// Every row of the domain is visible. - Full, - /// Exactly the set rows are visible. - Mask(CompressedBitSet), -} - -impl Rows { - /// Returns whether `row` is visible. - /// - /// The answer is fail-closed at every edge. A row the mask does not admit stays hidden, - /// including one above the mask's representable domain, so a mask evaluated against a narrower - /// universe hides the excess. - fn contains(&self, row: T) -> bool { - match self { - Self::Full => true, - Self::Mask(mask) => mask.contains(row), - } - } - - /// Returns whether this set is the unmasked one. - const fn is_full(&self) -> bool { - matches!(self, Self::Full) - } - - /// Returns the axis's retained mask bytes, zero for the unmasked one. - fn heap_bytes(&self) -> u64 { - match self { - Self::Full => 0, - Self::Mask(mask) => mask.heap_bytes(), - } - } -} - -/// Which delivery contract a proof serves. -/// -/// [`VisibilityProof::full_visibility`] holds the corpus outright and answers [`Self::Corpus`]. -/// [`VisibilityProof::from_masks`] declares a scope and answers [`Self::Scope`], whatever its masks -/// admit. A proof's kind and the `ViewSchedule` variant built over it are one bit spelled twice, -/// and binding a view refuses a pair that disagrees. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum ProofKind { - /// The whole corpus is visible, so the generation's corpus schedule serves this proof. - /// - /// Every zoom keeps its recorded cut, and no delivery-cut offset applies. - Corpus, - /// The proof declares a scope, so that scope's own cascade serves it. - /// - /// The cascade takes a delivery-cut offset, which an issuance over this proof resolves. - Scope, -} - -/// The delta-link identities a proof admits, either every one or exactly the captured set. -/// -/// A delta link's edge row is register-allocated past the baked bound, outside any mask built -/// over the generation's universe, so admission keys by entity identity where node and -/// link rows mask by row id. The full arm admits every live delta link of the caller's snapshot, -/// the same all-admitted semantics the row domains carry. The admitted arm holds exactly the -/// identities the scope's own resolution returned. -#[derive(Debug, Clone, PartialEq, Eq)] -enum DeltaLinks { - /// The proof admits every live delta link of the caller's snapshot. - Full, - /// The proof admits exactly the captured identities. - Admitted(FastHashSet), -} - -impl DeltaLinks { - /// Returns whether the set admits the delta link `id` names. - /// - /// Fail-closed exactly as [`Rows::contains`]: an identity the captured set does not hold is - /// not admitted, whatever the caller's snapshot publishes for it. - fn admits(&self, id: ArchivedEntityId) -> bool { - match self { - Self::Full => true, - Self::Admitted(links) => links.contains(&id), - } - } - - /// Returns the axis's retained set bytes, zero for the unmasked one. - fn heap_bytes(&self) -> u64 { - match self { - Self::Full => 0, - Self::Admitted(links) => links.allocation_size() as u64, - } - } -} - -/// The server-held visibility proof, the visible node-row and link-row sets every assembly path -/// masks by. -/// -/// A proof enters a handler by construction, because the assembly signatures take one, so a missing -/// proof is unrepresentable rather than defaulted. [`Self::from_masks`] builds the proof of an -/// evaluated scope, one mask per identity domain, and [`Self::full_visibility`] the proof that -/// admits every row of every domain, for a context that holds authority over the whole corpus. -/// -/// Membership is fail-closed. A row a held mask does not admit is not visible, so a proof built -/// against a smaller universe hides the excess rather than revealing it. The value authenticates -/// nothing, so masks from the wrong universe mask the wrong rows undetected, and pinning the proof -/// to its own generation is the caller's contract. -#[derive(Debug, Clone, PartialEq, Eq)] -pub(crate) struct VisibilityProof { - nodes: Rows, - edges: Rows, - delta_links: DeltaLinks, -} - -impl VisibilityProof { - /// Constructs the proof under which every row of every domain is visible. - /// - /// The proof carries authority over the whole corpus, and a caller states that authority by - /// choosing this constructor. The type performs no session resolution, so this value carries no - /// evidence of its own. - /// - /// Caller requirement: a served request answers under the proof of the scope resolved for it. - /// This value is the authority of a context that holds the corpus outright, such as operator - /// tooling over a trusted port and the fixtures that assert unmasked delivery. - #[must_use] - pub(crate) const fn full_visibility() -> Self { - Self { - nodes: Rows::Full, - edges: Rows::Full, - delta_links: DeltaLinks::Full, - } - } - - /// Constructs a proof from server-held visibility masks, one per domain. - /// - /// An admitted row is visible: `nodes` masks the node rows, `edges` the link rows, and - /// `delta_links` admits the delta-link identities beside them. The proof needs all of them, - /// because none implies another - the link rows a caller may read are not a function of the - /// node rows it may read, and a delta link's register-allocated row is one neither mask - /// could name. - /// - /// Caller requirement: the masks are server-held state (a fresh visibility evaluation or a - /// verified sealed blob), never client-supplied values - the constructor accepts any masks and - /// verifies no origin. A row either mask does not admit stays hidden, and a delta link - /// outside the captured set never serves. - #[must_use] - pub(crate) const fn from_masks( - nodes: CompressedBitSet, - edges: CompressedBitSet, - delta_links: FastHashSet, - ) -> Self { - Self { - nodes: Rows::Mask(nodes), - edges: Rows::Mask(edges), - delta_links: DeltaLinks::Admitted(delta_links), - } - } - - /// Folds `snapshot`'s withdrawals out of the proof's masked domains. - /// - /// A withdrawn fitted node or edge row leaves its mask, and a withdrawn identity leaves the - /// admitted delta-link set, so a folded scoped proof admits nothing `snapshot` withdraws. - /// Full domains stay full: admission subtraction remains the corpus proof's whole withdrawal - /// authority, and the fold never narrows an authority the resolution declared corpus-wide. - /// - /// The arrival surfaces need no fold. A snapshot publishes placed arrivals, slots, and links - /// only for identities standing live at its publication, so no admitted slot or captured - /// link can name an identity the same snapshot withdraws. - /// - /// Caller requirement: `snapshot` is the one the proof's own resolution read. Folding a - /// different publication leaves rows that publication withdraws admitted, and the masks - /// carry no record of which snapshot they folded. - pub(crate) fn fold_withdrawn(&mut self, snapshot: &DeltaSnapshot) { - if let Rows::Mask(nodes) = &mut self.nodes { - for row in snapshot.withdrawn_node_rows() { - nodes.remove(row); - } - } - - if let Rows::Mask(edges) = &mut self.edges { - for row in snapshot.withdrawn_edge_rows() { - edges.remove(row); - } - } - - if let DeltaLinks::Admitted(links) = &mut self.delta_links { - links.retain(|&id| !snapshot.withdraws(id)); - } - } - - /// Returns which delivery contract this proof serves. - /// - /// The answer reads which constructor built the value, never how many rows its masks admit. - /// Masks admitting every row of a generation are still a declared scope, and reading them as - /// the corpus proof would serve that scope the operator's own schedule and cut. Both callers - /// are cheap paths that ask before doing work: the masked gathers skip their masking when the - /// whole corpus is visible, and an issuance resolves an occupancy aggregate only for a scope. - #[must_use] - pub(crate) const fn kind(&self) -> ProofKind { - if self.nodes.is_full() && self.edges.is_full() { - ProofKind::Corpus - } else { - ProofKind::Scope - } - } - - /// Returns whether node row `row` is visible. - pub(super) fn contains(&self, row: NodeRowId) -> bool { - self.nodes.contains(row) - } - - /// Removes every hidden row from `set`, leaving the visible subset. - pub(super) fn intersect(&self, set: &mut DenseBitSet) { - if self.nodes.is_full() { - return; - } - - let hidden: Vec = set.iter().filter(|&row| !self.contains(row)).collect(); - for row in hidden { - set.remove(row); - } - } - - /// Counts the visible node rows of the universe `[0, n)`. - pub(super) fn visible_below(&self, n: u64) -> u64 { - match &self.nodes { - Rows::Full => n, - Rows::Mask(mask) => mask.iter().take_while(|row| row.as_u64() < n).count() as u64, - } - } - - /// Returns whether the node axis admits every row of the universe `[0, n)`. - /// - /// The answer reads visibility rather than the constructor, so a mask admitting the whole - /// universe answers `true` exactly as the full constructor does, and stays a declared scope - /// under [`Self::kind`]. Rows a mask admits at or above `n` never count against the answer. - pub(super) fn nodes_saturated_below(&self, n: u64) -> bool { - match &self.nodes { - Rows::Full => true, - Rows::Mask(mask) => mask.contains_below(n), - } - } - - /// Returns the proof's retained mask bytes. - /// - /// A full axis holds no mask and counts zero, so the figure prices exactly what holding - /// this proof keeps allocated. The delta-link set prices its identities beside the two row - /// masks. - pub(super) fn heap_bytes(&self) -> u64 { - self.nodes.heap_bytes() + self.edges.heap_bytes() + self.delta_links.heap_bytes() - } - - /// Returns whether the proof admits the delta link `id` names. - /// - /// The full-visibility proof admits every one, and a scoped proof exactly the identities its - /// own resolution captured. The caller's snapshot bounds which links exist. This answers - /// which of them the scope may see. - pub(super) fn admits_delta_link(&self, id: ArchivedEntityId) -> bool { - self.delta_links.admits(id) - } - - /// Proves one node row visible: the sole [`VisibleRow`] constructor. - /// - /// Wire-domain ingress reaches this through [`Atlas::resolve`]; identity-domain ingress - /// (translate, locate by entity id) lands on internal rows without a decode and calls this - /// directly. Either way every failure is the same [`None`]. - pub(super) fn verify(&self, row: NodeRowId) -> Option { - self.contains(row).then_some(VisibleRow(row)) - } - - /// Proves one edge deliverable, as the sole [`VisibleEdge`] constructor. - /// - /// An edge delivers only when the proof admits its link row and both of its endpoints. The link - /// row carries the link entity's authorization, which the endpoints do not imply. The endpoints - /// carry the two entities every edge response names. A hidden link row, a hidden endpoint, and - /// an edge the generation never held all answer the same [`None`]. - /// - /// Caller requirement: `source` and `target` are `edge`'s own endpoints as the generation - /// records them, read off the endpoint column rather than supplied by a request. - pub(super) fn verify_edge( - &self, - edge: EdgeRowId, - source: NodeRowId, - target: NodeRowId, - ) -> Option { - (self.edges.contains(edge) && self.contains(source) && self.contains(target)) - .then_some(VisibleEdge(edge)) - } -} - -/// A node row that carries its visibility proof in the type. -/// -/// The only constructor is [`Atlas::resolve`], so an unproven row cannot reach an assembly -/// function that takes a [`VisibleRow`]. The value is the internal row id for in-process gathers, -/// and it never crosses the wire. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct VisibleRow(NodeRowId); - -impl VisibleRow { - /// Returns the proven row id. - pub(super) const fn get(self) -> NodeRowId { - self.0 - } -} - -/// A link row that carries its delivery proof in the type. -/// -/// The only constructor is [`VisibilityProof::verify_edge`], which is the delivery rule itself, so -/// an assembly value holding one cannot exist for an edge the proof withholds. Constructing the -/// value consumes the check rather than consulting it. The value is the internal row id for -/// in-process gathers, and it never crosses the wire. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) struct VisibleEdge(EdgeRowId); - -impl VisibleEdge { - /// Returns the proven row id. - pub(super) const fn get(self) -> EdgeRowId { - self.0 - } -} - -/// One wire node ingress resolved into the domain that publishes it. -/// -/// The decode answers from two domains under one universe. A row below the generation's -/// fitted bound is a fitted row and carries its visibility proof in the vessel. A row at or past -/// the bound is a cohort slot serving a placed arrival, answered by the identity it publishes, -/// and the caller resolves the arrival's delivery values against the view's own arrival table, -/// which holds exactly the admitted arrivals. -#[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub(crate) enum ResolvedRow { - /// A fitted row, proven visible. - Fitted(VisibleRow), - /// A placed arrival's identity, admitted on its cohort slot. - Arrival(ArchivedEntityId), -} - -impl Atlas { - /// Resolves a wire node row id into the domain that publishes it. - /// - /// The single join point of the response discipline. The decode runs under `cohort`'s - /// accepted universe, so a wire id allocated for a cohort slot resolves exactly while an entry - /// bound to that publication serves the request. A decoded row below the generation's fitted - /// bound resolves against the proof as a fitted row, and a row at or past it resolves through - /// the cohort's slot index as a placed arrival, admitted exactly when the proof's widened - /// mask holds the slot. - /// - /// Decode failure (out-of-universe wire values), mask misses (rows the proof hides), and - /// slots the cohort does not serve all answer the same [`None`]. Forbidden and nonexistent - /// are indistinguishable to everything downstream. The transport renders one problem body - /// for both, and nothing upstream of this point logs or branches on the cause. - #[must_use] - pub(crate) fn resolve( - &self, - proof: &VisibilityProof, - cohort: PlacementCohort<'_>, - wire: WireRow, - ) -> Option { - let row = self - .node_codec - .decode(wire, cohort.universe(self.node_universe))?; - if self.node_universe.contains(row) { - proof.verify(row).map(ResolvedRow::Fitted) - } else { - let (identity, _) = cohort.node_at(row)?; - - proof - .contains(row) - .then_some(ResolvedRow::Arrival(identity)) - } - } -} diff --git a/libs/@local/graph/atlas/src/serve/walk/census.rs b/libs/@local/graph/atlas/src/serve/walk/census.rs deleted file mode 100644 index c7d943851da..00000000000 --- a/libs/@local/graph/atlas/src/serve/walk/census.rs +++ /dev/null @@ -1,152 +0,0 @@ -//! Point counts and occupancy over the walkable columns. -//! -//! Every masked count covers the visible view alone. A hidden point adds nothing to population, -//! extent, or resolution, so a scope's numbers carry no evidence of what the mask removed. - -use hashql_core::id::{Id as _, IdSlice, bit_vec::DenseBitSet}; - -use super::Walk; -use crate::{ - identity::{BasePosition, NodeRowId}, - math::{Bounds2, Vec2}, - morton::{Depth, MortonCell, MortonKey}, - serve::{density::ViewOccupancy, visibility::ProofKind}, -}; - -/// The corpus-wide census of one visible view. -/// -/// The aggregates a root tile publishes about the whole view rather than about its own cell. Every -/// one of them is a function of the generation's artifacts and the proof alone. -/// -/// An operator proof's root delivers under the corpus schedule, whose count and depth the artifacts -/// answer. A scoped proof's root delivers under its own cascade, and the cascade answers its own -/// count and depth. -/// -/// A hidden point adds nothing to the extent, and a scope's census carries no evidence of what its -/// mask removed. -/// -/// The census type differs from [`ViewOccupancy`] on purpose. An occupancy counts occupied *cells* -/// per depth, the form the delivery-cut policy reads, because the ratified policy may not read row -/// counts. Row counts and coordinates make a census instead. Two views a policy must not -/// distinguish can therefore carry different censuses. -#[derive(Debug, Copy, Clone, PartialEq)] -pub(crate) enum ViewCensus { - /// An operator proof's census, read from the artifacts. - Corpus { - /// The points the corpus schedule's root cut delivers. - visible: u64, - /// The generation's tight wire-frame extent, [`None`] when it holds no point. - bounds: Option, - /// The deepest occupied bucket. - min_resolution: u64, - }, - /// A scoped proof's census, one masked pass over the base column. - Scope { - /// The tight wire-frame extent of the visible set, [`None`] when it is empty. - bounds: Option, - }, -} - -impl ViewCensus { - /// The census of a scoped view holding no visible point. - #[cfg(test)] // The cache tests pair proofs with the empty view. - pub(crate) const EMPTY: Self = Self::Scope { bounds: None }; - - /// Returns the tight wire-frame extent of the visible set, [`None`] when it is empty. - pub(crate) const fn bounds(self) -> Option { - match self { - Self::Corpus { bounds, .. } | Self::Scope { bounds } => bounds, - } - } -} - -impl Walk<'_> { - /// Inserts the rows of one tile's cumulative delivered set. - /// - /// A tile's delivered set is mode-independent, because its cumulative delta set equals its - /// total set, so the gather is one run scan per bucket of the cumulative schedule, deduplicated - /// by the set itself. - pub(crate) fn delivered_rows_into( - &self, - z: u8, - cell: MortonCell, - set: &mut DenseBitSet, - ) { - for depth in self.grid.cut_buckets(z) { - for position in self.run(depth, cell) { - set.insert(self.row_ids[position]); - } - } - } - - /// Returns the deepest occupied bucket, zero when no point exists. - pub(crate) fn deepest_occupied(&self) -> u64 { - self.morton - .fenceposts() - .lengths() - .iter() - .rposition(|&length| length > 0) - .map_or(0, |bucket| bucket as u64) - } - - /// Censuses the masked view in one pass over the base column, folding every admitted - /// coordinate into the extent. - fn masked_census(&self, positions: &IdSlice) -> ViewCensus { - ViewCensus::Scope { - bounds: Bounds2::from_points( - positions - .iter_enumerated() - .filter(|&(position, _)| self.admits(position)) - .map(|(_, &point)| point), - ), - } - } - - /// Censuses the visible view over the whole corpus. - /// - /// `cut` is the root's cumulative schedule bucket, `positions` the base coordinate column, and - /// `bounds` the generation's own extent, absent exactly when the generation holds no point. - /// - /// An operator proof admits every row, and the artifacts hold its census: the fencepost prefix - /// is the visible count and the generation's extent is the visible extent. - pub(crate) fn visible_census( - &self, - cut: Depth, - positions: &IdSlice, - bounds: Option, - ) -> ViewCensus { - if self.proof.kind() == ProofKind::Corpus { - return ViewCensus::Corpus { - visible: u64::from(self.morton.fenceposts().segment(cut).end.as_u32()), - bounds, - min_resolution: self.deepest_occupied(), - }; - } - - self.masked_census(positions) - } - - /// Aggregates the visible view's Morton occupancy. - /// - /// One pass over the code column gathers the visible rows' keys, which [`ViewOccupancy::of`] - /// folds into the occupied-cell counts a delivery-cut policy reads. A hidden row contributes no - /// key, so it occupies no cell at any depth: the aggregate is a statement about the view and - /// nothing else. - /// - /// The gather owns its keys, since the fold sorts them. The proof's own visible count sizes the - /// buffer exactly. - pub(crate) fn visible_occupancy(&self) -> ViewOccupancy { - let count = self.morton.count(); - let visible = usize::try_from(self.proof.visible_below(count)) - .expect("a visible row count fits usize"); - - let mut keys: Vec = Vec::with_capacity(visible); - for position in BasePosition::MIN..self.morton.fenceposts().bound() { - if self.admits(position) { - keys.push(self.morton.code(position)); - } - } - - ViewOccupancy::of(&mut keys) - } -} diff --git a/libs/@local/graph/atlas/src/serve/walk/full.rs b/libs/@local/graph/atlas/src/serve/walk/full.rs deleted file mode 100644 index ab48b1f1b57..00000000000 --- a/libs/@local/graph/atlas/src/serve/walk/full.rs +++ /dev/null @@ -1,201 +0,0 @@ -//! The full-visibility deliveries. -//! -//! With every row visible, a tile's delivered set is contiguous in base order: the cascade sort -//! placed each bucket's points in one code-column run per cell, so delivery is range assembly over -//! the fenceposts and the quad node's recorded run - no per-point work at all. A view holding -//! admitted arrivals splices them among the ranges afterward, paying per arrival rather than per -//! delivered row. - -use core::ops::Range; - -use hashql_core::id::Id as _; - -use super::Walk; -use crate::{ - file::quad::Node, - identity::BasePosition, - morton::{Depth, MortonCell}, - serve::schedule::{ArrivalOverlay, Splice}, -}; - -/// One full-visibility delivery. -/// -/// The wire head's run vocabulary, with the first delivered bucket, the per-bucket point counts, -/// and the delivered base-position ranges in delivery order. -#[derive(Debug)] -pub(crate) struct RangeDelivery { - /// The first bucket the response's runs describe. - pub first_bucket: u8, - /// Per-bucket point counts, bucket-major from `first_bucket`. - pub runs: Vec, - /// The delivered base-position ranges, in delivery order. - pub ranges: Vec>, -} - -impl Walk<'_> { - /// Assembles the zoom-0 delta delivery. - /// - /// Buckets `0..=cut` as fencepost differences, one contiguous base-order range. - #[expect( - clippy::single_range_in_vec_init, - reason = "an array of one range is what a delta delivery IS" - )] - pub(crate) fn root_delta(&self) -> RangeDelivery { - let cut = self.grid.cut(0); - let runs = self - .grid - .cut_buckets(0) - .map(|depth| self.bucket_length(depth)) - .collect(); - let end = self.segment_end(cut); - - RangeDelivery { - first_bucket: 0, - runs, - ranges: vec![BasePosition::MIN..end], - } - } - - /// Assembles a non-root delta delivery. - /// - /// The node's own-bucket run verbatim, one zero-length run when the cell has no node. - pub(crate) fn delta(&self, z: u8, node: Option<&Node>) -> RangeDelivery { - let cut = self.grid.cut(z); - node.map_or_else( - || RangeDelivery { - first_bucket: cut.get(), - runs: vec![0], - ranges: Vec::new(), - }, - |node| { - // The node run is file vocabulary; open validated the point count against the - // u32 domain, so the positions are in range. - let run = node.run(); - let run = BasePosition::from_u64(run.start)..BasePosition::from_u64(run.end); - RangeDelivery { - first_bucket: cut.get(), - runs: vec![run.end.as_u32() - run.start.as_u32()], - ranges: vec![run], - } - }, - ) - } - - /// Assembles a total delivery. - /// - /// One code-column run per bucket of the cumulative schedule, bucket-major. - pub(crate) fn total(&self, z: u8, cell: MortonCell) -> RangeDelivery { - let capacity = self.grid.cut(z).get() as usize + 1; - let mut runs = Vec::with_capacity(capacity); - let mut ranges = Vec::with_capacity(capacity); - - for depth in self.grid.cut_buckets(z) { - let run = self.run(depth, cell); - runs.push(run.end.as_u32() - run.start.as_u32()); - ranges.push(run); - } - - RangeDelivery { - first_bucket: 0, - runs, - ranges, - } - } - - /// Splices the view's arrival overlay into a full-visibility delivery. - /// - /// Each delivered bucket takes the overlay's arrivals of that bucket, clamped into the - /// grid's catch-all. An arrival splices after the fitted rows whose keys are at most its - /// own, because every fitted row outranks every arrival. The bucket's run gains its arrival - /// count, and each splice records the arrival's index in the final merged order. The - /// delivered ranges stay untouched, so a delivery whose cells hold no arrival returns no - /// splice and moves no byte. - #[expect( - clippy::cast_possible_truncation, - reason = "a delivery walks at most the 33 buckets of the key width" - )] - pub(crate) fn splice_arrivals( - &self, - delivery: &mut RangeDelivery, - overlay: &ArrivalOverlay, - cell: MortonCell, - ) -> Vec { - let deepest = self.grid.deepest(); - let codes = self.morton.codes(); - let mut splices = Vec::new(); - let mut segments = SegmentCursor { - ranges: delivery.ranges.iter(), - current: BasePosition::MIN..BasePosition::MIN, - }; - - // Fitted rows delivered in the buckets already walked. - let mut delivered = 0_u32; - - for (index, run) in delivery.runs.iter_mut().enumerate() { - let bucket = Depth::new(delivery.first_bucket + index as u8) - .expect("the delivered buckets lie within the grid's schedule"); - let fitted = *run; - let segment = segments.take(fitted); - - let arrivals = overlay.run(bucket, cell, deepest); - *run += u32::try_from(arrivals.len()).expect("run lengths lie within the u32 universe"); - for (key, arrival) in arrivals { - let before = - codes[segment.clone()].partition_point(|code| code.get() <= key.to_bits()); - let at = delivered - + u32::try_from(before + splices.len()) - .expect("delivery indices lie within the u32 universe"); - - splices.push(Splice { at, arrival }); - } - - delivered += fitted; - } - - splices - } -} - -/// A cursor splitting a delivery's ranges into its runs' position segments. -/// -/// Each run of a full-visibility delivery covers one contiguous position segment by the cascade -/// sort's construction, so the cursor hands runs their segments in delivery order. A zero-length -/// run takes an empty segment. -struct SegmentCursor<'delivery> { - /// The ranges not yet entered. - ranges: core::slice::Iter<'delivery, Range>, - /// The positions remaining in the entered range. - current: Range, -} - -impl SegmentCursor<'_> { - /// Takes the next `count` positions as one segment. - fn take(&mut self, count: u32) -> Range { - while self.current.is_empty() { - match self.ranges.next() { - Some(range) => self.current = range.clone(), - None => break, - } - } - - let start = self.current.start; - let end = BasePosition::from_u32(start.as_u32() + count); - debug_assert!( - count == 0 || end <= self.current.end, - "a run never straddles the delivery's ranges", - ); - self.current = end..self.current.end; - - start..end - } -} - -/// Reads the occupied-child bitmask off a node record. -/// -/// Bit `i` set when Morton child `i` holds a point below the node's cut, which by the -/// node-existence rule is exactly when the child node exists. -pub(crate) fn occupied_children(node: &Node) -> u8 { - (0..4).fold(0_u8, |bits, quadrant| { - bits | (u8::from(node.child(quadrant).is_some()) << quadrant) - }) -} diff --git a/libs/@local/graph/atlas/src/serve/walk/mod.rs b/libs/@local/graph/atlas/src/serve/walk/mod.rs deleted file mode 100644 index 60d787eb601..00000000000 --- a/libs/@local/graph/atlas/src/serve/walk/mod.rs +++ /dev/null @@ -1,153 +0,0 @@ -//! The LOD walk: schedule-driven point delivery over one generation's spatial columns. -//! -//! A tile's delivery reads the Morton column through the bucket schedule. Zoom `z` delivers the -//! buckets at or below its cut, each bucket contributing the code-column run of the tile's cell. -//! [`Walk`] carries the columns that walk needs - the quadtree, the Morton column, the row column, -//! the validated schedule - bound to one visibility proof, and the submodules deliver over it: -//! -//! - [`full`]: the full-visibility deliveries, borrowed-shape base-order ranges, with the view's -//! arrivals spliced among them. -//! - [`census`]: point counts and occupancy, unmasked and masked. -//! -//! A scoped view's delivery reads its own cascade instead - [`ScopeSchedule`] - built over exactly -//! the rows its proof admits. The walk supplies that build's gather and the per-tile masked counts. -//! -//! Positions, counts, and run lengths live in the `u32` domain: the position type owns that width -//! at every fencepost accessor, so every segment and run the walk reads is already typed. -//! -//! [`ScopeSchedule`]: super::schedule::ScopeSchedule - -mod census; -pub(super) mod full; -pub(super) mod subtract; - -use core::ops::Range; - -use hashql_core::id::{Id as _, IdSlice}; - -pub(crate) use self::census::ViewCensus; -use super::{ - Atlas, - grid::Grid, - schedule::{Splice, ViewRow}, - visibility::VisibilityProof, -}; -use crate::{ - file::{ - morton::read::MortonFile, - quad::{Node, read::QuadFile}, - }, - identity::{BasePosition, NodeRowId}, - morton::{Depth, MortonCell}, - salt::wire::tile::{DeliveredRows, DeliveredSet}, -}; - -/// One generation's walkable columns under one visibility proof. -/// -/// The value borrows the opened generation, so construction is free per request and the walk -/// methods take no column parameters. -#[derive(Debug, Copy, Clone)] -pub(super) struct Walk<'atlas> { - grid: Grid, - morton: &'atlas MortonFile, - quad: &'atlas QuadFile, - row_ids: &'atlas IdSlice, - proof: &'atlas VisibilityProof, -} - -impl<'atlas> Walk<'atlas> { - /// Binds the generation's spatial columns to `proof`. - pub(super) fn of(atlas: &'atlas Atlas, proof: &'atlas VisibilityProof) -> Self { - Self { - grid: atlas.grid, - morton: &atlas.morton, - quad: &atlas.quad, - row_ids: atlas.rows.view(), - proof, - } - } - - /// Returns the quad node owning `cell`. - /// - /// [`None`] when the schedule delivers nothing new at or below it. - pub(super) fn node_of(&self, cell: MortonCell) -> Option<&'atlas Node> { - let index = self.quad.locate(cell)?; - Some(&self.quad.nodes()[index as usize]) - } - - /// Returns whether the proof admits the row at base position `position`. - fn admits(&self, position: BasePosition) -> bool { - self.proof.contains(self.row_ids[position]) - } - - /// Returns bucket `depth`'s positions inside `cell`. - fn run(&self, depth: Depth, cell: MortonCell) -> Range { - self.morton.run(depth, cell) - } - - /// Returns the first position past the cumulative schedule at `cut`. - fn segment_end(&self, cut: Depth) -> BasePosition { - self.morton.fenceposts().segment(cut).end - } - - /// Returns bucket `depth`'s point count. - fn bucket_length(&self, depth: Depth) -> u32 { - let segment = self.morton.fenceposts().segment(depth); - segment.end.as_u32() - segment.start.as_u32() - } -} - -/// The delivered point set of one tile. -/// -/// Borrowed-shape ranges when every row is visible, a gathered row list under a mask, and ranges -/// with spliced arrivals when an operator delivery interleaves a cohort, where a row is either a -/// base position or a cohort arrival. -#[derive(Debug)] -pub(super) enum DeliveredPoints { - /// Contiguous base-position ranges in delivery order. - Ranges(Vec>), - /// Gathered view rows in delivery order, visibility already applied. - Positions(Vec), - /// Contiguous base-position ranges with arrivals spliced among them. - Spliced { - /// The delivered base-position ranges, in delivery order. - ranges: Vec>, - /// The spliced arrivals, ascending by delivery index. - splices: Vec, - }, -} - -impl DeliveredPoints { - /// Views the set in the wire encoder's borrowed shape. - pub(super) const fn as_wire(&self) -> DeliveredSet<'_> { - match self { - Self::Ranges(ranges) => DeliveredSet::Ranges(ranges), - Self::Positions(list) => DeliveredSet::Positions(list), - Self::Spliced { ranges, splices } => DeliveredSet::Spliced { ranges, splices }, - } - } - - /// Counts the delivered points. - pub(super) fn count(&self) -> usize { - match self { - Self::Ranges(ranges) => ranges - .iter() - .map(|range| range.end.as_usize() - range.start.as_usize()) - .sum(), - Self::Positions(list) => list.len(), - Self::Spliced { ranges, splices } => { - let bases: usize = ranges - .iter() - .map(|range| range.end.as_usize() - range.start.as_usize()) - .sum(); - - bases + splices.len() - } - } - } - - /// Iterates the delivered rows in delivery order, a range's positions as base rows. - pub(super) fn iter(&self) -> DeliveredRows<'_> { - self.as_wire().into_iter() - } -} diff --git a/libs/@local/graph/atlas/src/serve/walk/subtract.rs b/libs/@local/graph/atlas/src/serve/walk/subtract.rs deleted file mode 100644 index 016e45cf4a0..00000000000 --- a/libs/@local/graph/atlas/src/serve/walk/subtract.rs +++ /dev/null @@ -1,440 +0,0 @@ -//! Withdrawal subtraction over one tile's delivered set. -//! -//! Admission subtraction edits the assembled document, because the corpus fast paths never -//! consult the proof and a full-visibility view builds no masks. A scoped entry also folds its -//! own snapshot's withdrawals into its masks at resolution, so this walk is the corpus path's -//! whole withdrawal authority and every path's answer for withdrawals newer than the entry. The -//! delivered positions partition into the head's `runs` sequentially in delivery order, so one -//! walk with a run cursor drops each withdrawn position and decrements the run that owns it. A -//! range that straddles a withdrawal splits around it. That construction preserves -//! `sum(runs) == delivered`, and the encoder's assertion keeps its force. -//! -//! A run subtracted to zero keeps its positional slot, which is the wire's own law for -//! zero-length entries, so a tile whose delivered set subtracts to nothing serves the existing -//! empty-tile shape. An input range subtracted to nothing leaves the range list instead, which -//! moves no wire byte: the columns gather positions and the runs carry the counts, so an empty -//! range was already invisible to both. -//! -//! A spliced delivery subtracts in its merged order. Withdrawn base rows split their ranges and -//! withdrawn arrivals leave the splice list, and every surviving splice takes its index in the -//! subtracted order, so the shape's own law - the delivery index counts rows of both kinds - -//! survives the edit. - -use hashql_core::id::Id as _; - -use super::DeliveredPoints; -use crate::{ - identity::BasePosition, - serve::schedule::{Splice, ViewRow}, -}; - -/// Drops every row `withdraws` names from `delivered`, decrementing the owning run. -/// -/// Caller requirement: `runs` partitions `delivered` sequentially in delivery order, the producer -/// contract the encoder asserts. A shorter partition panics here on the run cursor, which is a -/// producer defect rather than request data. -pub(crate) fn subtract_withdrawn( - delivered: &mut DeliveredPoints, - runs: &mut [u32], - withdraws: impl Fn(ViewRow) -> bool, -) { - // The cursor walks the partition as produced, while the decrements edit `runs` in place. - let partition = runs.to_vec(); - let mut cursor = RunCursor { - runs: &partition, - index: 0, - consumed: 0, - }; - - match delivered { - DeliveredPoints::Positions(list) => list.retain(|&row| { - let run = cursor.advance(); - if withdraws(row) { - runs[run] -= 1; - false - } else { - true - } - }), - DeliveredPoints::Ranges(ranges) => { - let mut split = Vec::with_capacity(ranges.len()); - for range in &*ranges { - let mut keep = range.start; - - for position in range.clone() { - let run = cursor.advance(); - if withdraws(ViewRow::Base(position)) { - if keep < position { - split.push(keep..position); - } - - keep = BasePosition::from_u32(position.as_u32() + 1); - runs[run] -= 1; - } - } - - if keep < range.end { - split.push(keep..range.end); - } - } - *ranges = split; - } - DeliveredPoints::Spliced { ranges, splices } => { - let mut split = Vec::with_capacity(ranges.len()); - let mut kept_splices = Vec::with_capacity(splices.len()); - let mut pending = splices.iter().peekable(); - - // The pre-subtraction and post-subtraction delivery indices, counting rows of both - // kinds. - let mut output = 0_u32; - let mut kept = 0_u32; - - for range in &*ranges { - let mut keep = range.start; - for position in range.clone() { - while let Some(splice) = pending.next_if(|splice| splice.at == output) { - let run = cursor.advance(); - if withdraws(ViewRow::Arrival(splice.arrival)) { - runs[run] -= 1; - } else { - kept_splices.push(Splice { - at: kept, - arrival: splice.arrival, - }); - kept += 1; - } - - output += 1; - } - - let run = cursor.advance(); - if withdraws(ViewRow::Base(position)) { - if keep < position { - split.push(keep..position); - } - keep = BasePosition::from_u32(position.as_u32() + 1); - runs[run] -= 1; - } else { - kept += 1; - } - output += 1; - } - if keep < range.end { - split.push(keep..range.end); - } - } - - // The splices past the last base row are the delivery's tail. - for splice in pending { - debug_assert_eq!( - splice.at, output, - "a trailing splice sits at the delivery's own tail index", - ); - let run = cursor.advance(); - if withdraws(ViewRow::Arrival(splice.arrival)) { - runs[run] -= 1; - } else { - kept_splices.push(Splice { - at: kept, - arrival: splice.arrival, - }); - kept += 1; - } - output += 1; - } - - *ranges = split; - *splices = kept_splices; - } - } -} - -/// A cursor over the delivered set's run partition, in delivery order. -/// -/// Each call to [`RunCursor::advance`] consumes one position and returns the index of the run -/// that owns it, skipping zero-length runs, which keep their positional slots without owning any -/// position. -struct RunCursor<'partition> { - /// The run partition as produced, before any decrement. - runs: &'partition [u32], - /// The run owning the next position. - index: usize, - /// Positions already consumed from the current run. - consumed: u32, -} - -impl RunCursor<'_> { - /// Consumes one position, returning its owning run's index. - const fn advance(&mut self) -> usize { - while self.consumed == self.runs[self.index] { - self.index += 1; - self.consumed = 0; - } - - self.consumed += 1; - self.index - } -} - -#[cfg(test)] -mod tests { - #![expect( - clippy::single_range_in_vec_init, - reason = "an array of one range is what a delta delivery IS" - )] - - use hashql_core::id::Id as _; - - use super::{DeliveredPoints, subtract_withdrawn}; - use crate::{ - identity::BasePosition, - serve::schedule::{ArrivalIndex, Splice, ViewRow}, - }; - - fn positions(list: &[u32]) -> DeliveredPoints { - DeliveredPoints::Positions( - list.iter() - .map(|&raw| ViewRow::Base(BasePosition::from_u32(raw))) - .collect(), - ) - } - - fn ranges(list: &[core::ops::Range]) -> DeliveredPoints { - DeliveredPoints::Ranges( - list.iter() - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)) - .collect(), - ) - } - - /// The row's raw index in its own domain, for the all-base fixtures here. - fn base_of(row: ViewRow) -> u32 { - match row { - ViewRow::Base(position) => position.as_u32(), - ViewRow::Arrival(index) => index.as_u32(), - } - } - - fn collected(delivered: &DeliveredPoints) -> Vec { - delivered.iter().map(base_of).collect() - } - - /// The invariant every case below re-checks: the runs re-sum to the delivered count. - #[track_caller] - fn assert_partitioned(delivered: &DeliveredPoints, runs: &[u32]) { - assert_eq!( - delivered.count() as u64, - runs.iter().map(|&count| u64::from(count)).sum::(), - "sum(runs) == delivered survives subtraction", - ); - } - - #[test] - fn gathered_positions_drop_and_their_runs_decrement() { - // The scoped delivery [4, 5] | [6, 20] | [21], three runs. - let mut delivered = positions(&[4, 5, 6, 20, 21]); - let mut runs = vec![2, 2, 1]; - - subtract_withdrawn(&mut delivered, &mut runs, |position| { - base_of(position) == 5 || base_of(position) == 21 - }); - - assert_eq!(collected(&delivered), [4, 6, 20]); - assert_eq!(runs, [1, 2, 0], "each drop debits the owning run"); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn a_range_straddling_a_withdrawal_splits() { - // A single-range, single-run delivery, the delta shape. - let mut delivered = ranges(&[10..15]); - let mut runs = vec![5]; - - subtract_withdrawn(&mut delivered, &mut runs, |position| { - base_of(position) == 12 - }); - - assert_eq!(collected(&delivered), [10, 11, 13, 14]); - assert_eq!(runs, [4]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn adjacent_withdrawals_collapse_without_empty_slivers() { - let mut delivered = ranges(&[0..6]); - let mut runs = vec![6]; - - subtract_withdrawn(&mut delivered, &mut runs, |position| { - (2..=3).contains(&base_of(position)) - }); - - assert_eq!(collected(&delivered), [0, 1, 4, 5]); - assert_eq!(runs, [4]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn range_edges_withdraw_without_slivers() { - let mut delivered = ranges(&[3..8]); - let mut runs = vec![5]; - - subtract_withdrawn(&mut delivered, &mut runs, |position| { - base_of(position) == 3 || base_of(position) == 7 - }); - - assert_eq!(collected(&delivered), [4, 5, 6]); - assert_eq!(runs, [3]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn bucket_major_ranges_debit_their_own_buckets() { - // One range per bucket, as a total delivers, with zero-length entries keeping slots. - let mut delivered = ranges(&[0..1, 4..4, 4..7, 7..9]); - let mut runs = vec![1, 0, 3, 2]; - - subtract_withdrawn(&mut delivered, &mut runs, |position| { - base_of(position) == 0 || base_of(position) == 8 - }); - - assert_eq!(collected(&delivered), [4, 5, 6, 7]); - assert_eq!(runs, [0, 0, 3, 1], "zero-length runs keep their slots"); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn an_all_withdrawn_tile_subtracts_to_the_empty_shape() { - let mut delivered = ranges(&[2..5]); - let mut runs = vec![3]; - - subtract_withdrawn(&mut delivered, &mut runs, |_| true); - - assert_eq!(collected(&delivered), [] as [u32; 0]); - assert_eq!(runs, [0]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn nothing_withdrawn_leaves_the_set_intact() { - let mut delivered = positions(&[1, 2, 3]); - let mut runs = vec![1, 2]; - - subtract_withdrawn(&mut delivered, &mut runs, |_| false); - - assert_eq!(collected(&delivered), [1, 2, 3]); - assert_eq!(runs, [1, 2]); - assert_partitioned(&delivered, &runs); - } - - fn spliced(list: &[core::ops::Range], at: &[(u32, u32)]) -> DeliveredPoints { - DeliveredPoints::Spliced { - ranges: list - .iter() - .map(|range| BasePosition::from_u32(range.start)..BasePosition::from_u32(range.end)) - .collect(), - splices: at - .iter() - .map(|&(at, arrival)| Splice { - at, - arrival: ArrivalIndex::from_u32(arrival), - }) - .collect(), - } - } - - /// Reads a spliced set's surviving splices. - fn splices_of(delivered: &DeliveredPoints) -> Vec<(u32, u32)> { - match delivered { - DeliveredPoints::Spliced { splices, .. } => splices - .iter() - .map(|splice| (splice.at, splice.arrival.as_u32())) - .collect(), - DeliveredPoints::Ranges(_) | DeliveredPoints::Positions(_) => { - panic!("the fixture holds the spliced shape") - } - } - } - - #[test] - fn a_withdrawn_base_row_shifts_the_splices_behind_it() { - // Merged order 10, A7, 11, 12: one run of four, arrival 7 at delivery index 1. - let mut delivered = spliced(&[10..13], &[(1, 7)]); - let mut runs = vec![4]; - - subtract_withdrawn( - &mut delivered, - &mut runs, - |row| matches!(row, ViewRow::Base(position) if position.as_u32() == 10), - ); - - // Withdrawing base 10 leaves A7, 11, 12 with the splice shifted to index 0. - assert_eq!(collected(&delivered), [7, 11, 12]); - assert_eq!(runs, [3]); - assert_eq!(splices_of(&delivered), [(0, 7)]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn a_withdrawn_arrival_leaves_the_splice_list_and_debits_its_run() { - // Merged order 0, 1 | A3, 5: the arrival opens the second run. - let mut delivered = spliced(&[0..2, 5..6], &[(2, 3)]); - let mut runs = vec![2, 2]; - - subtract_withdrawn(&mut delivered, &mut runs, |row| { - matches!(row, ViewRow::Arrival(_)) - }); - - assert_eq!(collected(&delivered), [0, 1, 5]); - assert_eq!(runs, [2, 1], "the arrival debits the run that owned it"); - assert_eq!(splices_of(&delivered), [] as [(u32, u32); 0]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn adjacent_base_and_arrival_withdrawals_shift_later_splices() { - // Merged order 10, A0, 11, 12, A1, 13: splices at delivery indices 1 and 4. - let mut delivered = spliced(&[10..14], &[(1, 0), (4, 1)]); - let mut runs = vec![6]; - - subtract_withdrawn(&mut delivered, &mut runs, |row| match row { - ViewRow::Base(position) => position.as_u32() == 11, - ViewRow::Arrival(index) => index.as_u32() == 0, - }); - - // Withdrawing base 11 and arrival 0 leaves 10, 12, A1, 13: A1 shifts from 4 to 2. - assert_eq!(collected(&delivered), [10, 12, 1, 13]); - assert_eq!(runs, [4]); - assert_eq!(splices_of(&delivered), [(2, 1)]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn a_trailing_splice_survives_the_last_base_row_withdrawal() { - // Merged order 0, A2: the arrival sits past the last base row. - let mut delivered = spliced(&[0..1], &[(1, 2)]); - let mut runs = vec![2]; - - subtract_withdrawn(&mut delivered, &mut runs, |row| { - matches!(row, ViewRow::Base(_)) - }); - - assert_eq!(collected(&delivered), [2]); - assert_eq!(runs, [1]); - assert_eq!(splices_of(&delivered), [(0, 2)]); - assert_partitioned(&delivered, &runs); - } - - #[test] - fn an_all_withdrawn_spliced_tile_subtracts_to_the_empty_shape() { - // Merged order A5, 2, 3: the arrival opens the delivery. - let mut delivered = spliced(&[2..4], &[(0, 5)]); - let mut runs = vec![3]; - - subtract_withdrawn(&mut delivered, &mut runs, |_| true); - - assert_eq!(collected(&delivered), [] as [u32; 0]); - assert_eq!(runs, [0]); - assert_eq!(splices_of(&delivered), [] as [(u32, u32); 0]); - assert_partitioned(&delivered, &runs); - } -} diff --git a/libs/@local/graph/atlas/tests/route_fixture.rs b/libs/@local/graph/atlas/tests/route_fixture.rs deleted file mode 100644 index bdf364bdebd..00000000000 --- a/libs/@local/graph/atlas/tests/route_fixture.rs +++ /dev/null @@ -1,761 +0,0 @@ -//! Captures and verifies the route-served scoped-tile fixture. -//! -//! The delivery contract wants one fixture whose bytes come through the served route rather -//! than from the encoder. A manifest resolved over a live store declares a nonzero delivery-cut -//! offset beside the authority token, and the data request presents that token. The TypeScript -//! client then decodes the checked-in response bytes under the declared schedule. The hand-built -//! corpus in `fixtures/wire/` pins the grammar. This fixture pins the composition that -//! transports it. -//! -//! The test stands where a hosting binary stands. It builds the router through -//! [`ServeCommand::run`] over the store the `HASH_GRAPH_PG_*` environment names and the -//! generation root `HASH_GRAPH_ATLAS_ROOT` names, supplies the request facilities a host -//! supplies - its credential verifier resolves the actor header directly - and then speaks to -//! the router as an HTTP client. No crate internals participate, so the captured bytes witness -//! the composition production serves, from the actor header and the sealed filter through token -//! admission to tile assembly. -//! -//! A fitted generation is not a standing property of every checkout, so the test runs only when -//! `ATLAS_ROUTE_FIXTURE` selects a mode and reports itself skipped otherwise: -//! -//! - `capture` re-captures the fixture against the live store and writes -//! `fixtures/wire/r1-scoped-route-tile.saltile` with its JSON sidecar. -//! - `verify` re-captures and requires byte equality with the checked-in fixture, refusing with a -//! re-bless instruction when the active generation is not the one the fixture records. -//! -//! `ATLAS_ROUTE_FIXTURE_ACTOR` names the requesting actor. The fixture's charter is nonzero cut -//! transport, so the capture refuses a resolution whose offset is zero. An actor whose view -//! attains the density band at the recorded schedule resolves that zero and cannot produce this -//! fixture. The sidecar records the served manifest declaration verbatim, and the TypeScript -//! conformance test derives its delivery cut from that declaration rather than from a constant. - -#![expect( - clippy::indexing_slicing, - reason = "an out-of-bounds read of a captured envelope is exactly this test's failure mode" -)] -#![expect( - clippy::wildcard_enum_match_arm, - reason = "the bounded reader refuses every shape the wire profile does not name" -)] -#![expect( - clippy::little_endian_bytes, - clippy::big_endian_bytes, - reason = "the envelope's integers are little-endian and CBOR arguments big-endian, per \ - contract" -)] -#![expect( - clippy::std_instead_of_alloc, - reason = "an integration test target is a std binary with no alloc crate of its own" -)] - -use core::{num::NonZeroU32, ops::ControlFlow}; -use std::{fs, path::PathBuf, sync::Arc}; - -use axum::{ - body::Body, - http::{HeaderMap, Method, Request, StatusCode, header::CONTENT_TYPE}, -}; -use clap::Parser; -use error_stack::Report; -use hash_graph_atlas::cli::{ - RootArgs, SecretString, ServeArgs, ServeCommand, ServeOptions, VisibilityLimits, -}; -use hash_graph_postgres_store::store::{ - DatabaseConnectionInfo, DatabasePoolConfig, DatabaseType, PostgresStorePool, - PostgresStoreSettings, -}; -use hash_middleware::{ - authentication::{ - provider::{AuthenticationProvider, Caller}, - request::{ - ACTOR_ID_HEADER, AuthenticationError, AuthenticationErrorKind, actor_id_from_header, - }, - }, - rate_limit::{ClientIpSource, RateLimitConfig, RateLimitMode}, -}; -use serde_json::{Value, json}; -use tokio_postgres::NoTls; -use tower::ServiceExt as _; -use type_system::principal::actor::{ActorId, UserId}; - -/// Resolves the actor header as a delegated user, standing where the deployment's credential -/// chain stands. -/// -/// The fixture's charter is transport, so the verifier trusts the header the test sends itself, -/// without the service-secret ceremony or the store's actor-kind lookup. The named actor is a -/// user in the arranged store. -struct HeaderDelegation; - -impl AuthenticationProvider for HeaderDelegation -where - C: Caller, -{ - fn authenticate( - &self, - headers: &HeaderMap, - ) -> impl Future>>>> + Send { - core::future::ready(match actor_id_from_header(headers) { - Ok(actor) => ControlFlow::Break(Ok(C::from_actor(ActorId::User(UserId::new(actor))))), - Err(error) if *error.kind() == AuthenticationErrorKind::MissingDelegatedActor => { - ControlFlow::Continue(()) - } - Err(error) => ControlFlow::Break(Err(Arc::new(Report::new(error)))), - }) - } -} - -/// The response header carrying the issued authority token. -const AUTHORITY_HEADER: &str = "atlas-authority"; - -/// The route serving the OpenAPI document. -const OPENAPI_PATH: &str = "/v1/atlas/openapi.json"; - -/// The fixture's basename under `fixtures/wire/`. -const FIXTURE_NAME: &str = "r1-scoped-route-tile"; - -/// The all-row filter document is the caller filter that admits every row its policies admit. -/// -/// Its presence is what makes the resolved scope mask-backed. A filtered request compiles -/// through the store's entity-query compiler regardless of how the actor's policies compile, -/// so the proof carries real masks rather than the unfiltered fast path. -const FILTER: &str = r#"{"all": []}"#; - -/// The environment contract, stated once for the skip report and the failure messages. -const ENV_CONTRACT: &str = "set ATLAS_ROUTE_FIXTURE=capture|verify with HASH_GRAPH_ATLAS_ROOT, \ - HASH_GRAPH_ATLAS_SECRET, ATLAS_ROUTE_FIXTURE_ACTOR, and the \ - HASH_GRAPH_PG_* store variables against a live store"; - -/// The hosting invocation flattens the same two flag groups as the `hash-graph atlas` binary. -#[derive(Debug, Parser)] -struct Invocation { - #[command(flatten)] - root: RootArgs, - #[command(flatten)] - serve: ServeArgs, -} - -/// One decoded CBOR value under the wire profile. -/// -/// The profile is deliberately small (`docs/wire.md` section 4): unsigned integers, byte and -/// text strings, arrays, integer-keyed maps, booleans, null, and f32. This reader exists so the -/// sidecar's expected values come from an implementation independent of the TypeScript decoder. -/// It reads exactly the profile and refuses everything else. -#[derive(Debug, Clone, PartialEq)] -enum Cbor { - Uint(u64), - Bytes(Vec), - Array(Vec), - Map(Vec<(u64, Self)>), - Bool(bool), - Null, - F32(f32), -} - -impl Cbor { - fn uint(&self) -> u64 { - match self { - Self::Uint(value) => *value, - other => panic!("expected an unsigned integer, read {other:?}"), - } - } - - fn array(&self) -> &[Self] { - match self { - Self::Array(items) => items, - other => panic!("expected an array, read {other:?}"), - } - } - - fn entry(&self, key: u64) -> &Self { - self.get(key) - .unwrap_or_else(|| panic!("the head carries no key {key}")) - } - - fn get(&self, key: u64) -> Option<&Self> { - match self { - Self::Map(entries) => entries - .iter() - .find(|(entry, _)| *entry == key) - .map(|(_, value)| value), - other => panic!("expected a map, read {other:?}"), - } - } -} - -/// Reads one value at `at`, returning it with the offset one past its end. -fn read_cbor(bytes: &[u8], at: usize) -> (Cbor, usize) { - let initial = bytes[at]; - let (major, info) = (initial >> 5, initial & 0x1F); - let (argument, mut next) = match info { - 0..=23 => (u64::from(info), at + 1), - 24 => (u64::from(bytes[at + 1]), at + 2), - 25 => { - let raw: [u8; 2] = bytes[at + 1..at + 3].try_into().expect("two bytes follow"); - (u64::from(u16::from_be_bytes(raw)), at + 3) - } - 26 => { - let raw: [u8; 4] = bytes[at + 1..at + 5].try_into().expect("four bytes follow"); - (u64::from(u32::from_be_bytes(raw)), at + 5) - } - 27 => { - let raw: [u8; 8] = bytes[at + 1..at + 9] - .try_into() - .expect("eight bytes follow"); - (u64::from_be_bytes(raw), at + 9) - } - _ => panic!("additional information {info} is outside the wire profile"), - }; - - match major { - 0 => (Cbor::Uint(argument), next), - 2 => { - let length = usize::try_from(argument).expect("byte strings fit in memory"); - ( - Cbor::Bytes(bytes[next..next + length].to_vec()), - next + length, - ) - } - 4 => { - let mut items = Vec::new(); - for _ in 0..argument { - let (item, after) = read_cbor(bytes, next); - items.push(item); - next = after; - } - (Cbor::Array(items), next) - } - 5 => { - let mut entries = Vec::new(); - for _ in 0..argument { - let (key, after_key) = read_cbor(bytes, next); - let (value, after_value) = read_cbor(bytes, after_key); - entries.push((key.uint(), value)); - next = after_value; - } - (Cbor::Map(entries), next) - } - 7 => match info { - 20 => (Cbor::Bool(false), next), - 21 => (Cbor::Bool(true), next), - 22 => (Cbor::Null, next), - 26 => { - let raw: [u8; 4] = bytes[at + 1..at + 5].try_into().expect("four bytes follow"); - (Cbor::F32(f32::from_be_bytes(raw)), at + 5) - } - other => panic!("simple value {other} is outside the wire profile"), - }, - other => panic!("major type {other} is outside the wire profile"), - } -} - -/// One parsed `SALTILET` envelope: the head map and the raw column sections. -struct Envelope { - head: Cbor, - positions: Vec, - row_ids: Vec, -} - -/// Parses and validates the envelope regions of `docs/wire.md` section 2. -/// -/// Asserts the fixed prefix, the directory's shape for the minimal-detail tile this fixture -/// pins (`TYPE_MASK` and `MASS` absent), and section alignment. -fn parse_envelope(bytes: &[u8]) -> Envelope { - assert_eq!(&bytes[..8], b"SALTILET", "the magic names a tile envelope"); - let field = |at: usize| u16::from_le_bytes([bytes[at], bytes[at + 1]]); - assert_eq!(field(8), 1, "wireVersion is 1"); - assert_eq!(field(10), 0, "flags are zero"); - let slots = usize::from(field(12)); - assert_eq!(slots, 5, "a tile envelope carries five slots"); - assert_eq!(field(14), 0, "reserved is zero"); - - let entry = |slot: usize| { - let at = 16 + slot * 8; - let word = |offset: usize| { - let raw: [u8; 4] = bytes[at + offset..at + offset + 4] - .try_into() - .expect("the directory carries eight bytes per slot"); - usize::try_from(u32::from_le_bytes(raw)).expect("offsets fit usize") - }; - (word(0), word(4)) - }; - - let (head_start, head_end) = entry(0); - assert_eq!(head_start, 16 + slots * 8, "HEAD starts the payload region"); - let (positions_start, positions_end) = entry(1); - assert!( - positions_start.is_multiple_of(8), - "POSITIONS starts 8-aligned" - ); - let (row_ids_start, row_ids_end) = entry(2); - assert_eq!( - entry(3), - (0, 0), - "TYPE_MASK is absent without coloredTypeIds" - ); - assert_eq!(entry(4), (0, 0), "MASS is absent under wireVersion 1"); - - let (head, consumed) = read_cbor(&bytes[head_start..head_end], 0); - assert_eq!(consumed, head_end - head_start, "the HEAD is one CBOR item"); - - Envelope { - head, - positions: bytes[positions_start..positions_end].to_vec(), - row_ids: bytes[row_ids_start..row_ids_end].to_vec(), - } -} - -/// Little-endian u32 words of a column section, the sidecar's bit-pattern form. -fn words_of(section: &[u8]) -> Vec { - section - .as_chunks::<4>() - .0 - .iter() - .map(|chunk| u32::from_le_bytes(*chunk)) - .collect() -} - -/// Sends one request through the router and returns the status, headers, and body. -async fn send( - router: &axum::Router, - request: Request, -) -> (StatusCode, axum::http::HeaderMap, Vec) { - let response = router - .clone() - .oneshot(request) - .await - .expect("the router is infallible"); - let (parts, body) = response.into_parts(); - let bytes = axum::body::to_bytes(body, usize::MAX) - .await - .expect("the response body is finite"); - (parts.status, parts.headers, bytes.to_vec()) -} - -/// The requesting actor `ATLAS_ROUTE_FIXTURE_ACTOR` names. -fn fixture_actor() -> String { - std::env::var("ATLAS_ROUTE_FIXTURE_ACTOR") - .unwrap_or_else(|_| panic!("ATLAS_ROUTE_FIXTURE_ACTOR names the actor; {ENV_CONTRACT}")) -} - -/// The served router over the store, root and secret the environment names. -/// -/// The delta consumer stays off, because every claim here is about the change under test and never -/// about live store state. Rate limiting observes, because a oneshot request carries no connection -/// info for the per-address key. -async fn served_router() -> axum::Router { - let store = |name: &str, fallback: &str| { - std::env::var(format!("HASH_GRAPH_PG_{name}")).unwrap_or_else(|_| fallback.to_owned()) - }; - let connection = DatabaseConnectionInfo::new( - DatabaseType::Postgres, - store("USER", "postgres"), - store("PASSWORD", "postgres"), - store("HOST", "localhost"), - store("PORT", "5432").parse().expect("the port is numeric"), - store("DATABASE", "graph"), - ); - let pool = PostgresStorePool::new( - &connection, - &DatabasePoolConfig::default(), - NoTls, - PostgresStoreSettings::default(), - ) - .await - .expect("the store the HASH_GRAPH_PG_* environment names is reachable"); - - let invocation = Invocation::parse_from(["route-fixture", "--no-delta"]); - let quota = |value| NonZeroU32::new(value).expect("the quota is non-zero"); - let facilities = ServeOptions { - provider: Arc::new(HeaderDelegation), - service_secret: SecretString::from("route-fixture-service-secret"), - rate_limit: RateLimitConfig { - rate_limit_mode: RateLimitMode::Observe, - client_ip_source: ClientIpSource::ConnectInfo, - rate_limit_gate_per_second: quota(10), - rate_limit_gate_burst: quota(50), - rate_limit_anonymous_per_hour: quota(60), - rate_limit_anonymous_burst: quota(50), - rate_limit_actor_per_hour: quota(6000), - rate_limit_actor_burst: quota(100), - }, - workflow: None, - pool: Arc::new(pool), - visibility: VisibilityLimits::default(), - }; - ServeCommand::new(invocation.root, invocation.serve) - .run(facilities) - .expect("the root holds an activated generation and a wire secret is configured") -} - -/// Every route the API router registers refuses a request without an actor, and the liveness -/// route alone answers without one. -/// -/// Path parameters take a placeholder, because the authentication layer answers before any -/// extractor reads the path. -#[tokio::test(flavor = "multi_thread")] -async fn routes_refuse_without_actor() { - if std::env::var_os("ATLAS_ROUTE_FIXTURE").is_none() { - eprintln!("routes_refuse_without_actor: skipped; {ENV_CONTRACT}"); - return; - } - let actor = fixture_actor(); - let router = served_router().await; - - let (status, _, body) = send( - &router, - Request::get(OPENAPI_PATH) - .header(ACTOR_ID_HEADER, &actor) - .body(Body::empty()) - .expect("the request builds"), - ) - .await; - assert_eq!(status, StatusCode::OK, "{}", String::from_utf8_lossy(&body)); - let document: Value = serde_json::from_slice(&body).expect("the OpenAPI document is JSON"); - - let mut operations: Vec<(Method, String)> = document["paths"] - .as_object() - .expect("the document lists its paths") - .iter() - .flat_map(|(path, operations)| { - operations - .as_object() - .into_iter() - .flatten() - .filter_map(|(method, _operation)| { - Method::from_bytes(method.to_ascii_uppercase().as_bytes()) - .ok() - .map(|method| (method, path.clone())) - }) - }) - .collect(); - operations.push((Method::GET, OPENAPI_PATH.to_owned())); - operations.push((Method::GET, "/v1/atlas/openapi".to_owned())); - assert!( - operations - .iter() - .any(|(_method, path)| path == "/v1/atlas/current"), - "the document lists no route the sweep knows; the sweep would be empty" - ); - - for (method, template) in &operations { - let path = template - .split('/') - .map(|segment| { - if segment.starts_with('{') { - "0" - } else { - segment - } - }) - .collect::>() - .join("/"); - let (status, headers, body) = send( - &router, - Request::builder() - .method(method.clone()) - .uri(&path) - .body(Body::empty()) - .expect("the request builds"), - ) - .await; - - assert_eq!( - status, - StatusCode::UNAUTHORIZED, - "{method} {template} answered without an actor: {}", - String::from_utf8_lossy(&body) - ); - assert_eq!( - headers - .get(CONTENT_TYPE) - .and_then(|value| value.to_str().ok()), - Some("application/problem+json"), - "{method} {template} refused outside the problem contract" - ); - let problem: Value = - serde_json::from_slice(&body).expect("the refusal is a problem document"); - assert_eq!( - problem["type"], "/problems/atlas/unauthenticated", - "{method} {template} refused for another cause: {problem}" - ); - } - - let (status, _, body) = send( - &router, - Request::get("/status") - .body(Body::empty()) - .expect("the request builds"), - ) - .await; - assert_eq!( - status, - StatusCode::OK, - "the liveness route requires an actor: {}", - String::from_utf8_lossy(&body) - ); -} - -/// Captures the fixture through the served route and verifies every law it pins. -/// -/// Skips, loudly, unless `ATLAS_ROUTE_FIXTURE` selects a mode: the fixture's bytes depend on a -/// fitted generation and a live store, which are operator-arranged rather than properties of -/// every checkout. The checked-in fixture with the TypeScript conformance test remains the -/// standing witness. This test is the honest re-derivation path. -#[expect( - clippy::too_many_lines, - reason = "the capture is one linear protocol run whose order is the contract it witnesses" -)] -#[tokio::test(flavor = "multi_thread")] -async fn route_served_scoped_tile_fixture() { - let Ok(mode) = std::env::var("ATLAS_ROUTE_FIXTURE") else { - eprintln!("route_served_scoped_tile_fixture: skipped; {ENV_CONTRACT}"); - return; - }; - assert!( - mode == "capture" || mode == "verify", - "ATLAS_ROUTE_FIXTURE selects capture or verify, not {mode:?}", - ); - let actor = fixture_actor(); - let router = served_router().await; - - // The generation under serve, from the route that names it. - let (status, _, body) = send( - &router, - Request::get("/v1/atlas/current") - .header(ACTOR_ID_HEADER, &actor) - .body(Body::empty()) - .expect("the request builds"), - ) - .await; - assert_eq!(status, StatusCode::OK, "{}", String::from_utf8_lossy(&body)); - let current: Value = serde_json::from_slice(&body).expect("current is JSON"); - let generation = current["generation"] - .as_str() - .expect("current names the generation") - .to_owned(); - - // The manifest resolves the filtered scope: the declaration and the token arrive together, - // which is the pairing the fixture exists to pin. - let (status, headers, body) = send( - &router, - Request::post(format!("/v1/atlas/generation/{generation}/manifest")) - .header(ACTOR_ID_HEADER, &actor) - .header(CONTENT_TYPE, "application/json") - .body(Body::from(FILTER)) - .expect("the request builds"), - ) - .await; - assert_eq!(status, StatusCode::OK, "{}", String::from_utf8_lossy(&body)); - let manifest: Value = serde_json::from_slice(&body).expect("the manifest is JSON"); - let token = headers - .get(AUTHORITY_HEADER) - .expect("the manifest issues an authority token") - .to_str() - .expect("the token is ASCII") - .to_owned(); - - let span = manifest["bucketSchedule"]["span"] - .as_u64() - .expect("the manifest declares the recorded span"); - assert!(span.is_power_of_two(), "the span is a power of two"); - let span_log2 = span.trailing_zeros(); - let offset = manifest["scopeSchedule"]["k"] - .as_u64() - .expect("the manifest declares the caller's offset"); - assert!( - offset >= 1, - "the fixture's charter is nonzero cut transport: this scope resolved k = 0, so the chosen \ - actor's view attains the density band at the recorded schedule and cannot produce it; \ - choose an actor whose view saturates instead", - ); - let cut_addend = span_log2 + u32::try_from(offset).expect("offsets are small"); - assert_eq!( - manifest["scopeSchedule"]["cut"].as_str(), - Some(format!("z+{cut_addend}").as_str()), - "the declared cut rule is the declared span and offset", - ); - - // The root tile under the sealed scope, all-defaults query. - let (status, headers, tile) = send( - &router, - Request::post(format!("/v1/atlas/tile/{generation}/plain/0/0/0")) - .header(ACTOR_ID_HEADER, &actor) - .header(AUTHORITY_HEADER, &token) - .body(Body::empty()) - .expect("the request builds"), - ) - .await; - assert_eq!(status, StatusCode::OK, "{}", String::from_utf8_lossy(&tile)); - assert_eq!( - headers - .get(CONTENT_TYPE) - .map(axum::http::HeaderValue::as_bytes), - Some(b"application/vnd.hash.saltile-v1".as_slice()), - "a tile answers as saltile bytes", - ); - - let envelope = parse_envelope(&tile); - let head = &envelope.head; - assert_eq!( - head.entry(0), - &Cbor::Bytes(hex_bytes(&generation)), - "the head echoes the served generation", - ); - assert_eq!(head.entry(1).uint(), 0, "the head echoes the plain variant"); - assert_eq!( - head.entry(2).array(), - &[Cbor::Uint(0), Cbor::Uint(0), Cbor::Uint(0)], - "the head echoes the root coordinate", - ); - assert_eq!(head.entry(3).uint(), 0, "the default mode is delta"); - - let delivered = head.entry(4).uint(); - assert!(delivered > 0, "a fixture of nothing witnesses nothing"); - assert!( - head.get(5).is_none(), - "key 5 is retired and no response emits it" - ); - assert_eq!(head.entry(6).uint(), 0, "a delta root starts at bucket 0"); - - let runs: Vec = head.entry(7).array().iter().map(Cbor::uint).collect(); - assert_eq!( - runs.len(), - usize::try_from(cut_addend + 1).expect("cuts are small"), - "the root carries one run per bucket through the declared cut z + {cut_addend}", - ); - assert_eq!( - runs.iter().sum::(), - delivered, - "sum(runs) = delivered is law in every response", - ); - - let global = head.entry(8); - let children = head.entry(9).uint(); - assert!(children <= 0xF, "children is a four-bit occupancy mask"); - assert_eq!( - head.entry(10), - &Cbor::Bool(false), - "minimal detail declares no trailer" - ); - - assert_eq!( - envelope.positions.len() as u64, - delivered * 8, - "POSITIONS carries one f32 xy pair per delivered point", - ); - assert_eq!( - envelope.row_ids.len() as u64, - delivered * 4, - "ROW_IDS carries one u32 per delivered point", - ); - - let bounds: Vec = global - .entry(1) - .array() - .iter() - .map(|value| match value { - Cbor::F32(bits) => json!(bits.to_bits()), - other => panic!("bounds carry f32 values, read {other:?}"), - }) - .collect(); - - let sidecar = json!({ - "golden": FIXTURE_NAME, - "layer": "tile", - "declaration": { - "generation": generation, - "wireVersion": manifest["wireVersion"], - "bucketSchedule": manifest["bucketSchedule"], - "scopeSchedule": manifest["scopeSchedule"], - }, - "request": { - "filter": serde_json::from_str::(FILTER).expect("the filter is JSON"), - "actor": actor, - "variant": "plain", - "coordinate": [0, 0, 0], - "detail": "minimal", - }, - "prefix": { - "magic": "SALTILET", - "wireVersion": 1, - "flags": 0, - "slotCount": 5, - "reserved": 0, - }, - "head": { - "generation": generation, - "variant": 0, - "coordinate": [0, 0, 0], - "mode": 0, - "delivered": delivered, - "firstBucket": 0, - "runs": runs, - "children": children, - "trailer": false, - "global": { - "visibleAtZoom": global.entry(0).uint(), - "boundsBits": bounds, - "minResolution": global.entry(2).uint(), - }, - }, - "positions": words_of(&envelope.positions), - "rowIds": words_of(&envelope.row_ids), - "typeMask": Value::Null, - "trailer": Value::Null, - "mass": Value::Null, - "appended": Value::Null, - }); - let sidecar_text = format!( - "{}\n", - serde_json::to_string_pretty(&sidecar).expect("sidecars are plain JSON"), - ); - - let dir = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("fixtures/wire"); - let bytes_path = dir.join(format!("{FIXTURE_NAME}.saltile")); - let sidecar_path = dir.join(format!("{FIXTURE_NAME}.json")); - - if mode == "capture" { - fs::write(&bytes_path, &tile).expect("the fixture directory is writable"); - fs::write(&sidecar_path, &sidecar_text).expect("the fixture directory is writable"); - eprintln!( - "captured {} and {}", - bytes_path.display(), - sidecar_path.display() - ); - return; - } - - let pinned_sidecar: Value = serde_json::from_slice( - &fs::read(&sidecar_path).expect("the checked-in sidecar exists; capture first"), - ) - .expect("the checked-in sidecar parses"); - assert_eq!( - pinned_sidecar["declaration"]["generation"].as_str(), - Some(generation.as_str()), - "the active generation is not the fixture's; re-capture with ATLAS_ROUTE_FIXTURE=capture \ - and re-run the TypeScript conformance test", - ); - let pinned = fs::read(&bytes_path).expect("the checked-in fixture exists; capture first"); - assert_eq!( - pinned, tile, - "the served route reproduces the checked-in bytes" - ); - assert_eq!( - serde_json::from_str::(&sidecar_text).expect("the fresh sidecar parses"), - pinned_sidecar, - "the re-derived sidecar agrees with the checked-in one", - ); -} - -/// The byte form of a lowercase hex string. -fn hex_bytes(hex: &str) -> Vec { - hex.as_bytes() - .as_chunks::<2>() - .0 - .iter() - .map(|pair| { - u8::from_str_radix(str::from_utf8(pair).expect("hex is ASCII"), 16) - .expect("the string is hexadecimal") - }) - .collect() -} From d252de5a61fd4ce2e3c65fab7e4250b0b10bd880 Mon Sep 17 00:00:00 2001 From: Bilal Mahmoud <7252775+indietyp@users.noreply.github.com> Date: Mon, 14 Sep 2026 16:26:09 +0200 Subject: [PATCH 2/3] chore: clean up atlas lint configuration Temporarily suppress lints in hash-graph-atlas to facilitate refactoring under BE-850. Replace specific dead-code handling with broader allowlist during consolidation work. --- Cargo.toml | 5 +++-- apps/hash-graph/src/subcommand/atlas.rs | 6 ++--- libs/@local/graph/atlas/src/lib.rs | 30 ++++++++++++++++++------- 3 files changed, 27 insertions(+), 14 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 99d1307bd39..d7b10d1dd84 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -378,8 +378,9 @@ nonstandard_style = { level = "warn", priority = -1 } unreachable_pub = "warn" unsafe_code = "deny" -# TODO: Enable `cargo` lints when supported -# [workspace.lints.cargo] +# TODO(BE-850): remove once the BE-815 stack has landed +[workspace.lints.cargo] +unused_dependencies = "allow" [workspace.lints.clippy] all = { level = "warn", priority = -1 } diff --git a/apps/hash-graph/src/subcommand/atlas.rs b/apps/hash-graph/src/subcommand/atlas.rs index 0d1025c126e..870ac2be570 100644 --- a/apps/hash-graph/src/subcommand/atlas.rs +++ b/apps/hash-graph/src/subcommand/atlas.rs @@ -2,16 +2,14 @@ use core::time::Duration; use clap::Parser; use error_stack::{Report, ResultExt as _}; -use hash_graph_atlas::cli::{self, PasswordString}; +use hash_graph_atlas::cli; use hash_graph_postgres_store::store::DatabaseConnectionInfo; use reqwest::Client; use tokio::time::timeout; use crate::{ error::{GraphError, HealthcheckError}, - subcommand::{ - HealthcheckArgs, wait_healthcheck, - }, + subcommand::{HealthcheckArgs, wait_healthcheck}, }; /// Address configuration for the atlas server. diff --git a/libs/@local/graph/atlas/src/lib.rs b/libs/@local/graph/atlas/src/lib.rs index 61884c8639e..797c71892a1 100644 --- a/libs/@local/graph/atlas/src/lib.rs +++ b/libs/@local/graph/atlas/src/lib.rs @@ -122,14 +122,28 @@ clippy::future_not_send, clippy::indexing_slicing )] -// Operator-command machinery is unconditional library code whose one consumer, the command -// shell, sits behind `cli`, so a build without `cli` marks that machinery dead rather than -// finding real rot. Bench machinery carries `cfg(any(test, feature = "bench"))` per item, so the -// dead-code lint is live on every other item in every unit with `cli` on. -#![cfg_attr(not(feature = "cli"), allow(dead_code))] -// The documentation's audience is the crate's developers. Module docs link private items on -// purpose, and readers view the docs under `--document-private-items`, where those links resolve. -#![allow(rustdoc::private_intra_doc_links)] +// TODO(BE-850): remove once all changes have landed +#![allow( + unused_crate_dependencies, + unused_features, + dead_code, + unreachable_pub, + unused_imports, + rustdoc::broken_intra_doc_links +)] +// #![cfg_attr( +// not(feature = "cli"), +// allow( +// dead_code, +// reason = "TODO(BE-804): the CLI is consolidated into one cohesive module" +// ) +// )] +#![allow( + rustdoc::private_intra_doc_links, + reason = "the crate is largely internal, for a user it makes more sense to read the full \ + docs, instead of just the outer public shell. Having arbitrary separation hurts \ + that exploration." +)] extern crate alloc; mod allocator; From e54bd1e4d26829c935dae2192b8a3212835aea9f Mon Sep 17 00:00:00 2001 From: Bilal Mahmoud <7252775+indietyp@users.noreply.github.com> Date: Mon, 14 Sep 2026 17:50:55 +0200 Subject: [PATCH 3/3] chore: remove integration test task from atlas --- libs/@local/graph/atlas/docs/task-dependencies.json | 9 --------- 1 file changed, 9 deletions(-) diff --git a/libs/@local/graph/atlas/docs/task-dependencies.json b/libs/@local/graph/atlas/docs/task-dependencies.json index a812fc20d18..3cddf7f8db2 100644 --- a/libs/@local/graph/atlas/docs/task-dependencies.json +++ b/libs/@local/graph/atlas/docs/task-dependencies.json @@ -44,15 +44,6 @@ "@rust/darwin-kperf-codegen" ] }, - "test:integration": { - "dependsOn": [], - "env": [ - "TEST_COVERAGE" - ], - "affectedBy": [ - "@rust/darwin-kperf-codegen" - ] - }, "test:miri": { "dependsOn": [], "affectedBy": [