diff --git a/protos/transaction/actions.proto b/protos/transaction/actions.proto new file mode 100644 index 00000000000..949d8c95ef1 --- /dev/null +++ b/protos/transaction/actions.proto @@ -0,0 +1,492 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +syntax = "proto3"; + +import "file.proto"; +import "table.proto"; +import "transaction/common.proto"; +import "google/protobuf/any.proto"; + +package lance.table; + +/* + * Action-based transactions (Transaction V2) — DRAFT. + * + * A `UserOperation` replaces the single legacy `Operation` (see + * transaction.proto) with an ordered list of granular `Action`s that commit + * atomically as one manifest change. This file is the wire draft only. + * + * STATUS + * ------ + * The messages and field numbers here are NOT yet a stable contract. Library + * support is read-side fail-closed only: a transaction carrying a + * `UserOperation` is rejected on load, and there is no write path. Apply, + * id translation, and conflict resolution are intentionally absent. + * + * DESIGN RATIONALE + * ---------------- + * The reasoning below is the durable record of why the wire format is shaped + * this way; it is deliberately kept in-tree rather than in an external tracker. + * + * 1. Actions are deltas, not post-images. Each action records *the change* to + * the manifest (add this file, tombstone this field), not the resulting + * state. Deltas are what make two things fall out uniformly with no special + * cases: composing several actions into one atomic commit, and merging or + * rebasing a branch (replay the deltas against the target). A few actions + * that carry non-derivable preconditions are the exception (see 8). + * + * 2. Minting vs. reference-stable actions. Minting actions allocate a new + * counter-based id (a field id, fragment id, or base id). Because the final + * id is not known until commit — and differs when the same action is + * replayed onto a different target during merge/rebase — a minting action + * names its new id with a `Local` token (see `Ref`) rather than a concrete + * value. Reference-stable actions touch already-committed ids at stable + * coordinates (a fragment's deletion vector, a config key); their target is + * unambiguous, so they may safely carry a post-image at rest. + * + * 3. One reference type, `Ref = Committed | Local`. `Committed` is a concrete, + * already-assigned id. `Local` is a placeholder minted by an earlier `Add*` + * action in the same operation. `Local` resolves to a freshly-allocated + * committed id at apply time, and re-resolves against the *target's* + * counters on merge/rebase — which is exactly what lets two independent + * `AddField`s on divergent branches become two distinct fields on merge. + * + * 4. Relocation by counter watermark. When an operation is replayed onto a + * newer version, minting actions are re-applied against the target's + * counters and their `Local` refs re-resolve; ids assigned above the + * read-version watermark are known to be freshly minted and are rewritten + * consistently across every action that referenced them. + * + * 5. Field-level schema. Schema changes are expressed per field + * (`AddField` / `DropField` / `AlterField`), never as a wholesale + * replacement, so concurrent schema edits to disjoint fields commute. + * + * 6. Identity-preserving `AlterField`. Altering a field is distinct from + * dropping it and adding a new one: the field keeps its id. Each mutable + * facet (name, type, nullability) is an independent optional, so two + * concurrent alters of *different* facets of the same field commute + * (e.g. a widening cast alongside a non-null -> nullable relaxation). + * + * 7. Index segments, not indices. The format has no first-class "index" + * separate from its segments; a new index is written as its first segment, + * and a logical index is the set of segments sharing a `name`. The actions + * mirror that: add/remove/adjust operate on a segment (by UUID). The + * resulting per-segment config duplication is a pre-existing limitation this + * draft does not attempt to fix. + * + * 8. Assertions carry only non-derivable preconditions. Most conflict checks + * are computed from the actions themselves. The exception is a fact that + * cannot be reconstructed from any post-image — e.g. the set of keys a + * merge-insert inserted — which is carried explicitly as an assertion + * (`AssertUniqueKeys`) and checked against concurrent commits. + * + * WHAT STAYS OFF THE WIRE + * ----------------------- + * Two classes of information are deliberately not serialized, and are + * recomputed at conflict time instead: + * - Conflict *footprints* (which fields/fragments/keys an action touches). + * These are a pure function of each action, so storing them would only risk + * drift. + * - Large derivable row-level deltas, e.g. a deletion's affected rows or an + * update's matched row offsets. These can be recomputed by diffing against + * the read-version state, and serializing them would bloat transactions + * with potentially huge integer lists. + * + * ADDITIVE-SAFETY + * --------------- + * The schema is meant to evolve additively: message identities and the core + * mutation fields are pinned, and anything discovered during implementation + * should arrive as an *added* optional field rather than a reshaping of what is + * here. + */ + +/* + * A reference to a counter-allocated identifier (field id, fragment id, or base + * id) that may not be committed yet. + * + * `committed` is an already-assigned id. `local` is a placeholder token minted + * by an `Add*` action earlier in the same `UserOperation`; it resolves to a + * freshly-allocated committed id at apply, and re-resolves against the target's + * counters on merge/rebase. `local` tokens are scoped to a single + * `UserOperation` and must be distinct within it (validated by the writer once + * the write path exists). + */ +message Ref { + oneof kind { + uint64 committed = 1; + uint32 local = 2; + } +} + +/* + * A user-facing, composable transaction: an ordered list of user actions that + * commit atomically as a single manifest change. + */ +message UserOperation { + // Human-readable description, e.g. "INSERT INTO t VALUES (1)". + string description = 1; + // Unique identifier for this operation (matches Transaction.uuid semantics). + string uuid = 2; + // The dataset version this operation was planned against. + uint64 read_version = 3; + // The ordered list of user actions applied by this operation. + repeated UserAction actions = 4; +} + +/* + * A single user-recognizable step within a UserOperation (e.g. "append batch", + * "rebuild index"). + * + * The description keeps the transaction history human-readable. When a range of + * transactions is squashed, each original UserOperation collapses into one + * UserAction so the readable sequence survives; the action lists are flattened + * when applied to the manifest. + */ +message UserAction { + // Human-readable description of this step. + string description = 1; + // The granular manifest changes this step expands to. + repeated Action actions = 2; +} + +/* + * A single granular change (a delta) or assertion. Legacy operations decompose + * into an ordered list of these. + * + * TODO: the `data_change` markers on the actions below are drafted per-action + * for maximum flexibility; revisit whether per-action or per-UserAction + * granularity is the right unit before the format stabilizes. + */ +message Action { + oneof action { + // -- Minting actions (allocate a new counter-based id via a Local token) -- + AddFragment add_fragment = 1; + AddField add_field = 3; + AddBase add_base = 4; + // -- Fragment / data-file deltas -- + AddDataFile add_data_file = 2; + TombstoneFieldData tombstone_field_data = 5; + RemoveFragment remove_fragment = 6; + // -- Reference-stable changes (committed coordinates; post-image at rest) -- + SetDeletionFile set_deletion_file = 7; + ConfigUpdate config_update = 8; + AddOverlays add_overlays = 9; + RefreshRowVersionMetadata refresh_row_version_metadata = 10; + UpdateCompactedSsTables update_compacted_sstables = 11; + // -- Schema (field-level; no wholesale SetSchema) -- + DropField drop_field = 12; + AlterField alter_field = 13; + // -- Index segments -- + AddIndexSegment add_index_segment = 14; + RemoveIndexSegment remove_index_segment = 15; + AdjustIndexCoverage adjust_index_coverage = 16; + // -- Wholesale-within-operation -- + ResetTable reset_table = 17; + ReserveFragmentIds reserve_fragment_ids = 18; + // -- Assertions (preconditions, not deltas) -- + AssertUniqueKeys assert_unique_keys = 19; + } +} + +/* + * Mint a new, empty fragment. + * + * Its data files arrive via AddDataFile actions referencing this fragment's + * `local` token. A freshly-minted fragment has no deletion vector; deletions + * are applied by a later operation via SetDeletionFile, once the fragment has a + * committed id. + */ +message AddFragment { + // Placeholder token for the fragment id, resolved to a committed id at apply. + uint32 local = 1; + // Number of physical rows (including rows later tombstoned). + uint64 physical_rows = 2; + /* + * Stable-row-id and row-version sequences, carried exactly as on + * DataFragment. Absent on datasets without stable row ids. + */ + oneof row_id_sequence { + bytes inline_row_ids = 3; + ExternalFile external_row_ids = 4; + } + oneof last_updated_at_version_sequence { + bytes inline_last_updated_at_versions = 5; + ExternalFile external_last_updated_at_versions = 6; + } + oneof created_at_version_sequence { + bytes inline_created_at_versions = 7; + ExternalFile external_created_at_versions = 8; + } + /* + * false => rearrangement only (rows unchanged, e.g. compaction); CDC and + * streaming consumers may skip it. Absent is treated as true (real change). + */ + optional bool data_change = 9; +} + +/* + * Add a data file to a fragment. + */ +message AddDataFile { + // The fragment to add the file to (Committed, or a same-op Local fragment). + Ref fragment = 1; + /* + * The data file. Its `fields` are left unset (-1) and stamped in at apply + * once `field_ids` resolve; `field_ids` below is the authority for the + * column -> field mapping. + */ + DataFile file = 2; + /* + * One entry per column in `file`, in order. Committed for existing fields, + * Local for fields minted by an AddField in the same operation. + */ + repeated Ref field_ids = 3; + // Data-change marker; see AddFragment.data_change. + optional bool data_change = 4; +} + +/* + * Mint a new schema field. + * + * A struct/list column that introduces several fields is expressed as several + * ordered AddField actions (parent before children, children referencing the + * parent via a Local ref). + */ +message AddField { + // Placeholder token for the new field id, resolved at apply. + uint32 local = 1; + /* + * The parent field: Committed to reparent under an existing field, Local for + * a sibling minted in the same operation. Absent => top-level column. + */ + optional Ref parent = 2; + // Field definition; its `id` and `parent_id` are ignored (see `local`, `parent`). + lance.file.Field def = 3; +} + +/* + * Mint a new base path. + */ +message AddBase { + // Placeholder token for the base id, resolved at apply. + uint32 local = 1; + // Base path; its `id` is ignored and stamped in at apply. + BasePath base = 2; +} + +/* + * Tombstone the data-file binding of one or more committed fields. + * + * Each field's slot in whatever data file currently backs it is set to -2, and + * any file left with no live fields is pruned at apply. This is how a column's + * data is dropped or superseded (data files have no id of their own; a live + * field is backed by exactly one file). A column re-encode is + * TombstoneFieldData(X) followed by AddDataFile(new file, field X). + */ +message TombstoneFieldData { + Ref fragment = 1; + // Committed field ids whose current backing is tombstoned. + repeated uint64 field_ids = 2; + // Data-change marker; see AddFragment.data_change. + optional bool data_change = 3; +} + +/* + * Remove a fragment entirely (all rows deleted, or replaced by compaction). + */ +message RemoveFragment { + Ref fragment = 1; + // Data-change marker; see AddFragment.data_change. + optional bool data_change = 2; +} + +/* + * Set (replace) a fragment's deletion file. + * + * Reference-stable: the fragment id is committed and physical row offsets are + * stable, so the post-image deletion file is safe to store. The newly-deleted + * rows (the delta used for rebase and conflict) are derived by diffing against + * the read-version deletion file and are not serialized here. + */ +message SetDeletionFile { + // The fragment, by committed id: unlike the sibling fragment actions this + // takes no Ref, because a fragment minted in the same operation has no rows + // to delete yet (see AddFragment). + uint64 fragment = 1; + DeletionFile deletion_file = 2; + // Data-change marker; see AddFragment.data_change. + optional bool data_change = 3; +} + +/* + * Apply config / metadata updates. + * + * Reference-stable key-merges over stable coordinates (config keys, field ids). + * Mirrors the new-style fields of the legacy UpdateConfig operation. + */ +message ConfigUpdate { + optional UpdateMap config = 1; + optional UpdateMap table_metadata = 2; + optional UpdateMap schema_metadata = 3; + /* + * Per-field metadata updates, keyed by Ref so a field minted in the same + * operation (Local) can receive metadata. + */ + repeated FieldMetadata field_metadata = 4; + + message FieldMetadata { + Ref field = 1; + UpdateMap updates = 2; + } +} + +/* + * Append overlay files to fragments (see DataOverlayFile in table.proto). + * + * Reference-stable: overlays target committed fragments/offsets/fields and are + * appended, never replacing existing overlays. + */ +message AddOverlays { + Ref fragment = 1; + repeated DataOverlayFile overlays = 2; + // Data-change marker; see AddFragment.data_change. + optional bool data_change = 3; +} + +/* + * Refresh row-version (created-at / last-updated-at) metadata for fragments + * whose columns were merged in place, mirroring the implicit refresh Merge + * performs on stable-row-id datasets. + */ +message RefreshRowVersionMetadata { + repeated uint64 fragment_ids = 1; +} + +/* + * Mark MemWAL SSTables as compacted into the base table. Covers + * UpdateMemWalState and the compaction bookkeeping of Update. + */ +message UpdateCompactedSsTables { + repeated CompactedSsTable compacted_sstables = 1; +} + +/* + * Remove a field from the schema (and, at apply, prune any data file left with + * no live fields). References an existing committed field id. + */ +message DropField { + uint64 field = 1; +} + +/* + * Alter facets of an existing field in place, preserving its identity (id). + * + * Each facet is optional: present means "change this facet", absent means + * "leave unchanged". Keying conflict on (field id, facet) lets independent + * facet changes to the same field commute (e.g. a cast concurrent with a + * nullability change). A cast additionally needs TombstoneFieldData plus a new + * AddDataFile to rewrite the data. New facets are added as optional fields. + */ +message AlterField { + uint64 field = 1; + optional string name = 2; + // New Arrow logical type (see Field.logical_type). The cast. + optional string logical_type = 3; + optional bool nullable = 4; +} + +/* + * Add an index segment. + * + * A brand-new index is expressed as its first segment; the set of segments + * sharing `name` constitutes one logical index. + */ +message AddIndexSegment { + UUID uuid = 1; + string name = 2; + // Indexed field ids (Committed, or Local for same-op minted fields). + repeated Ref fields = 3; + google.protobuf.Any index_details = 4; + optional int32 index_version = 5; + /* + * Fragments covered by this segment (resolved to a fragment_bitmap at apply, + * and remapped under fragment relocation). Committed or same-op Local. + */ + repeated Ref covered_fragments = 6; + /* + * Index files with sizes, when produced by the writer (e.g. a compaction that + * rewrites the segment). Empty when unavailable. + */ + repeated IndexFile files = 7; + // Data-change marker; see AddFragment.data_change. false marks a segment + // rebuild/compaction that does not reflect any change to the indexed data. + optional bool data_change = 8; +} + +/* + * Remove an index segment by uuid. + */ +message RemoveIndexSegment { + UUID uuid = 1; + // Data-change marker; see AddFragment.data_change. false marks removal of a + // segment superseded by compaction, not a change to the indexed data. + optional bool data_change = 2; +} + +/* + * Adjust the fragment coverage of an existing index segment without rewriting + * it. + * + * NOTE: coverage representation is an acknowledged open design area; this shape + * is provisional pending how much coverage change is derivable from fragment + * relocation versus stated explicitly here. + */ +message AdjustIndexCoverage { + UUID uuid = 1; + // Fragments to add to the segment's coverage (Committed or same-op Local). + repeated Ref add_fragments = 2; + // Committed fragment ids to drop from the segment's coverage. + repeated uint64 remove_fragments = 3; +} + +/* + * Reset the table to an empty state, in preparation for a fresh schema and data + * written by later actions in the same operation (the decomposition of a full + * Overwrite / CREATE OR REPLACE). + * + * Drops the entire schema, schema metadata, all fragments, and all indices; + * preserves table config, table metadata, and base paths (change those via + * ConfigUpdate / AddBase in the same operation). Its write footprint spans the + * whole schema, all fragments, and all indices, so it conflicts with any + * concurrent change to them (DROP TABLE-style exclusivity). + */ +message ResetTable {} + +/* + * Reserve a contiguous range of fragment ids from the counter, for a later + * (possibly distributed) writer to populate. Fragments written against a + * reserved range reference those ids as Committed. + */ +message ReserveFragmentIds { + uint32 count = 1; +} + +/* + * An assertion (precondition), not a delta: the keys this operation inserts + * must not collide with keys inserted by a concurrent commit. + * + * Carries a bloom / exact-set filter of the inserted key hashes (not derivable + * from any post-image); conflict = the two filters intersect. Home for + * merge-insert's strict primary-key conflict detection. + */ +message AssertUniqueKeys { + /* + * Field ids of the key columns (unenforced primary key). Committed, or Local + * for same-op minted fields. This is the authoritative key-column list; + * `filter.field_ids` (an artifact of the shared KeyExistenceFilter type) is + * ignored here and left empty. + */ + repeated Ref key_fields = 1; + KeyExistenceFilter filter = 2; +} diff --git a/protos/transaction/common.proto b/protos/transaction/common.proto new file mode 100644 index 00000000000..ad290bb485a --- /dev/null +++ b/protos/transaction/common.proto @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +syntax = "proto3"; + +package lance.table; + +/* + * Shared transaction building blocks, referenced by both the legacy + * operations in transaction.proto and the action-based transactions in + * actions.proto. + * + * These types are transaction *machinery* (metadata merges, conflict-detection + * filters), not persisted table state, so they live here rather than in + * table.proto. Kept in a dedicated file so both transaction models can import + * them without a circular dependency (transaction.proto imports actions.proto, + * so actions.proto must not import transaction.proto). + */ + +/* + * An entry for a map update. + * + * If `value` is not set, the key is removed from the map. + */ +message UpdateMapEntry { + // The key of the map entry to update. + string key = 1; + // The value to set for the key. Absent removes the key. + optional string value = 2; +} + +/* + * A set of updates to apply to a string map (table config, table metadata, + * schema metadata, or per-field metadata). + */ +message UpdateMap { + repeated UpdateMapEntry update_entries = 1; + /* + * If true, the map is replaced entirely with these entries. If false, the + * entries are merged into the existing map. + */ + bool replace = 2; +} + +/* + * Exact set of key hashes for conflict detection. + * + * Used when the number of inserted rows is small. + */ +message ExactKeySetFilter { + // 64-bit hashes of the inserted row keys. + repeated uint64 key_hashes = 1; +} + +/* + * Bloom filter for key existence tests. + * + * Used when the number of inserted rows is large. + */ +message BloomFilter { + // Bitset backing the bloom filter (SBBF format). + bytes bitmap = 1; + // Number of bits in the bitmap. + uint32 num_bits = 2; + /* + * Number of items the filter was sized for. Used for intersection validation: + * filters with different sizes cannot be compared. Default: 8192. + */ + uint64 number_of_items = 3; + /* + * False positive probability the filter was sized for. Used for intersection + * validation: filters with different parameters cannot be compared. + * Default: 0.00057. + */ + double probability = 4; +} + +/* + * A filter for checking key existence in the set of rows inserted by a + * merge-insert operation. + * + * Only created when the merge-insert's ON columns match the schema's unenforced + * primary key; its presence indicates strict primary-key conflict detection. + * Conflict detection intersects two filters, so the underlying representation + * is either an exact set (small row counts) or a Bloom filter (large counts). + */ +message KeyExistenceFilter { + // Field ids of the columns participating in the key (the unenforced primary key). + repeated int32 field_ids = 1; + // The underlying data structure storing the key hashes. + oneof data { + // Exact set of key hashes (small number of rows). + ExactKeySetFilter exact = 2; + // Bloom filter (large number of rows). + BloomFilter bloom = 3; + } +} diff --git a/protos/transaction.proto b/protos/transaction/transaction.proto similarity index 84% rename from protos/transaction.proto rename to protos/transaction/transaction.proto index c3c5e38190e..ef64b4102fd 100644 --- a/protos/transaction.proto +++ b/protos/transaction/transaction.proto @@ -5,6 +5,8 @@ syntax = "proto3"; import "file.proto"; import "table.proto"; +import "transaction/actions.proto"; +import "transaction/common.proto"; import "google/protobuf/any.proto"; package lance.table; @@ -187,46 +189,6 @@ message Transaction { optional string branch_name = 5; } - // Exact set of key hashes for conflict detection. - // Used when the number of inserted rows is small. - message ExactKeySetFilter { - // 64-bit hashes of the inserted row keys. - repeated uint64 key_hashes = 1; - } - - // Bloom filter for key existence tests. - // Used when the number of rows is large. - message BloomFilter { - // Bitset backing the bloom filter (SBBF format). - bytes bitmap = 1; - // Number of bits in the bitmap. - uint32 num_bits = 2; - // Number of items the filter was sized for. - // Used for intersection validation (filters with different sizes cannot be compared). - // Default: 8192 - uint64 number_of_items = 3; - // False positive probability the filter was sized for. - // Used for intersection validation (filters with different parameters cannot be compared). - // Default: 0.00057 - double probability = 4; - } - - // A filter for checking key existence in set of rows inserted by a merge insert operation. - // Only created when the merge insert's ON columns match the schema's unenforced primary key. - // The presence of this filter indicates strict primary key conflict detection should be used. - // Can use either an exact set (for small row counts) or a Bloom filter (for large row counts). - message KeyExistenceFilter { - // Field IDs of columns participating in the key (must match unenforced primary key). - repeated int32 field_ids = 1; - // The underlying data structure storing the key hashes. - oneof data { - // Exact set of key hashes (used for small number of rows). - ExactKeySetFilter exact = 2; - // Bloom filter (used for large number of rows). - BloomFilter bloom = 3; - } - } - // Serialized as sorted distinct local physical row offsets within the fragment (0-based). message UInt32List { repeated uint32 values = 1; @@ -271,21 +233,6 @@ message Transaction { REWRITE_COLUMNS = 1; } - // An entry for a map update. If value is not set, the key will be removed from the map. - message UpdateMapEntry { - // The key of the map entry to update. - string key = 1; - // The value to set for the key. - optional string value = 2; - } - - message UpdateMap { - repeated UpdateMapEntry update_entries = 1; - // If true, the map will be replaced entirely with the new entries. - // If false, the new entries will be merged with the existing map. - bool replace = 2; - } - // An operation that updates the table config, table metadata, schema metadata, // or field metadata. message UpdateConfig { @@ -369,6 +316,9 @@ message Transaction { Clone clone = 113; UpdateBases update_bases = 114; DataOverlay data_overlay = 115; + // Action-based transaction (Transaction V2). See actions.proto. + // DRAFT: currently rejected on load; no write path. + UserOperation user_operation = 116; } // Fields 200/202 (`blob_append` / `blob_overwrite`) previously represented blob dataset ops. diff --git a/python/src/transaction.rs b/python/src/transaction.rs index e710a36df02..18800cf0bd9 100644 --- a/python/src/transaction.rs +++ b/python/src/transaction.rs @@ -13,7 +13,7 @@ use lance::dataset::transaction::{ use lance::datatypes::Schema; use lance_table::format::overlay::{DataOverlayFile, OverlayCoverage}; use lance_table::format::{BasePath, DataFile, Fragment, IndexFile, IndexMetadata}; -use pyo3::exceptions::PyValueError; +use pyo3::exceptions::{PyNotImplementedError, PyValueError}; use pyo3::types::PySet; use pyo3::{Bound, FromPyObject, PyAny, PyResult, Python}; use pyo3::{intern, prelude::*}; @@ -732,6 +732,11 @@ impl<'py> IntoPyObject<'py> for PyLance<&Operation> { base_op.call0() } } + // Action-based operations have no Python surface yet. An explicit arm + // keeps them off the `todo!()` below, which would panic. + Operation::UserOperation(..) => Err(PyNotImplementedError::new_err( + "action-based transactions are not supported from the Python binding", + )), _ => todo!(), } } diff --git a/rust/lance-index/src/lib.rs b/rust/lance-index/src/lib.rs index c95e2ce609c..52fa9e5a5de 100644 --- a/rust/lance-index/src/lib.rs +++ b/rust/lance-index/src/lib.rs @@ -79,9 +79,7 @@ pub struct IndexMetadata { pub distance_type: String, } -pub fn is_system_index(index_meta: &lance_table::format::IndexMetadata) -> bool { - index_meta.name == FRAG_REUSE_INDEX_NAME || index_meta.name == MEM_WAL_INDEX_NAME -} +pub use lance_table::system_index::is_system_index; pub fn infer_system_index_type( index_meta: &lance_table::format::IndexMetadata, diff --git a/rust/lance-table/build.rs b/rust/lance-table/build.rs index 03216636b30..103109adfdb 100644 --- a/rust/lance-table/build.rs +++ b/rust/lance-table/build.rs @@ -19,7 +19,9 @@ fn main() -> Result<()> { prost_build.compile_protos( &[ "./protos/table.proto", - "./protos/transaction.proto", + "./protos/transaction/common.proto", + "./protos/transaction/actions.proto", + "./protos/transaction/transaction.proto", "./protos/rowids.proto", ], &["./protos"], diff --git a/rust/lance-table/src/format.rs b/rust/lance-table/src/format.rs index 5a5db7919d3..cb707211b2b 100644 --- a/rust/lance-table/src/format.rs +++ b/rust/lance-table/src/format.rs @@ -6,6 +6,7 @@ use uuid::Uuid; mod fragment; mod index; +pub mod key_existence; mod manifest; pub mod overlay; mod transaction; @@ -17,8 +18,8 @@ pub use fragment::*; pub use index::{IndexFile, IndexMetadata, index_metadata_codec, list_index_files_with_sizes}; pub use manifest::{ - BasePath, DETACHED_VERSION_MASK, DataStorageFormat, Manifest, SelfDescribingFileReader, - WriterVersion, is_detached_version, + BasePath, DETACHED_VERSION_MASK, DataStorageFormat, Manifest, ManifestBuildConfig, + SelfDescribingFileReader, WriterVersion, is_detached_version, }; pub use transaction::Transaction; diff --git a/rust/lance-table/src/format/key_existence.rs b/rust/lance-table/src/format/key_existence.rs new file mode 100644 index 00000000000..da88deb62ed --- /dev/null +++ b/rust/lance-table/src/format/key_existence.rs @@ -0,0 +1,714 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Key existence tracking for merge insert conflict detection. +//! +//! A merge insert records the join keys it inserted into a bloom filter, which +//! is carried in the transaction so a concurrent commit can detect whether it +//! inserted any of the same keys. The filter is serialized into the transaction +//! protobuf, so it lives at the table layer next to [`crate::format::pb`]. + +use std::collections::HashSet; +use std::collections::hash_map::DefaultHasher; +use std::hash::{Hash, Hasher}; + +use crate::format::pb; +use arrow_array::cast::AsArray; +use arrow_array::{ + Array, BinaryArray, LargeBinaryArray, LargeListArray, LargeStringArray, ListArray, RecordBatch, + StringArray, StructArray, +}; +use arrow_schema::DataType; +use lance_core::Result; +use lance_core::deepsize::DeepSizeOf; +use lance_core::utils::bloomfilter::sbbf::{Sbbf, SbbfBuilder}; + +// Default bloom filter config: 8192 items @ 0.00057 fpp -> 16KiB filter +pub const BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS: u64 = 8192; +pub const BLOOM_FILTER_DEFAULT_PROBABILITY: f64 = 0.00057; + +/// Key value for conflict detection. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub enum KeyValue { + String(String), + Int64(i64), + UInt64(u64), + Binary(Vec), + List(Vec), + Struct(Vec), + Composite(Vec), +} + +impl KeyValue { + pub fn to_bytes(&self) -> Vec { + match self { + Self::String(s) => s.as_bytes().to_vec(), + Self::Int64(i) => i.to_le_bytes().to_vec(), + Self::UInt64(u) => u.to_le_bytes().to_vec(), + Self::Binary(b) => b.clone(), + Self::List(values) | Self::Struct(values) | Self::Composite(values) => { + let mut result = Vec::new(); + for value in values { + result.extend_from_slice(&value.to_bytes()); + result.push(0); + } + result + } + } + } + + pub fn hash_value(&self) -> u64 { + let mut hasher = DefaultHasher::new(); + self.to_bytes().hash(&mut hasher); + hasher.finish() + } +} + +/// Builder for KeyExistenceFilter using Split Block Bloom Filter. +#[derive(Debug, Clone)] +pub struct KeyExistenceFilterBuilder { + sbbf: Sbbf, + field_ids: Vec, + item_count: usize, +} + +impl KeyExistenceFilterBuilder { + pub fn new(field_ids: Vec) -> Self { + let sbbf = SbbfBuilder::new() + .expected_items(BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS) + .false_positive_probability(BLOOM_FILTER_DEFAULT_PROBABILITY) + .build() + .expect("Failed to build SBBF"); + Self { + sbbf, + field_ids, + item_count: 0, + } + } + + pub fn insert(&mut self, key: KeyValue) -> Result<()> { + self.sbbf.insert(&key.to_bytes()[..]); + self.item_count += 1; + Ok(()) + } + + pub fn contains(&self, key: &KeyValue) -> bool { + self.sbbf.check(&key.to_bytes()[..]) + } + + pub fn might_intersect(&self, other: &Self) -> Result { + self.sbbf + .might_intersect(&other.sbbf) + .map_err(|e| lance_core::Error::invalid_input(e.to_string())) + } + + pub fn field_ids(&self) -> &[i32] { + &self.field_ids + } + + pub fn estimated_size_bytes(&self) -> usize { + self.sbbf.size_bytes() + } + + pub fn len(&self) -> usize { + self.item_count + } + + pub fn is_empty(&self) -> bool { + self.item_count == 0 + } + + pub fn build(&self) -> KeyExistenceFilter { + KeyExistenceFilter { + field_ids: self.field_ids.clone(), + filter: FilterType::Bloom { + bitmap: self.sbbf.to_bytes(), + num_bits: (self.sbbf.size_bytes() as u32) * 8, + number_of_items: BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS, + probability: BLOOM_FILTER_DEFAULT_PROBABILITY, + }, + } + } +} + +impl From<&KeyExistenceFilterBuilder> for pb::KeyExistenceFilter { + fn from(builder: &KeyExistenceFilterBuilder) -> Self { + Self { + field_ids: builder.field_ids.clone(), + data: Some(pb::key_existence_filter::Data::Bloom(pb::BloomFilter { + bitmap: builder.sbbf.to_bytes(), + num_bits: (builder.sbbf.size_bytes() as u32) * 8, + number_of_items: BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS, + probability: BLOOM_FILTER_DEFAULT_PROBABILITY, + })), + } + } +} + +/// Filter type for key existence data. +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub enum FilterType { + ExactSet(HashSet), + Bloom { + bitmap: Vec, + num_bits: u32, + number_of_items: u64, + probability: f64, + }, +} + +/// Tracks keys of inserted rows for conflict detection. +/// Only created when ON columns match the schema's unenforced primary key. +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub struct KeyExistenceFilter { + pub field_ids: Vec, + pub filter: FilterType, +} + +impl KeyExistenceFilter { + pub fn from_bloom_filter(bloom: &KeyExistenceFilterBuilder) -> Self { + bloom.build() + } + + /// Check if two filters intersect. Returns (has_intersection, might_be_false_positive). + /// Errors if bloom filter configs don't match. + pub fn intersects(&self, other: &Self) -> Result<(bool, bool)> { + match (&self.filter, &other.filter) { + (FilterType::ExactSet(a), FilterType::ExactSet(b)) => { + Ok((a.iter().any(|h| b.contains(h)), false)) + } + (FilterType::ExactSet(_), FilterType::Bloom { .. }) + | (FilterType::Bloom { .. }, FilterType::ExactSet(_)) => { + // Can't compare different hash schemes, assume intersection + Ok((true, true)) + } + ( + FilterType::Bloom { + bitmap: a_bits, + number_of_items: a_num_items, + probability: a_prob, + .. + }, + FilterType::Bloom { + bitmap: b_bits, + number_of_items: b_num_items, + probability: b_prob, + .. + }, + ) => { + if a_num_items != b_num_items || (a_prob - b_prob).abs() > f64::EPSILON { + return Err(lance_core::Error::invalid_input(format!( + "Bloom filter config mismatch: ({}, {}) vs ({}, {})", + a_num_items, a_prob, b_num_items, b_prob + ))); + } + let has = Sbbf::bytes_might_intersect(a_bits, b_bits) + .map_err(|e| lance_core::Error::invalid_input(e.to_string()))?; + Ok((has, has)) + } + } + } +} + +impl From<&KeyExistenceFilter> for pb::KeyExistenceFilter { + fn from(filter: &KeyExistenceFilter) -> Self { + match &filter.filter { + FilterType::ExactSet(hashes) => Self { + field_ids: filter.field_ids.clone(), + data: Some(pb::key_existence_filter::Data::Exact( + pb::ExactKeySetFilter { + key_hashes: hashes.iter().copied().collect(), + }, + )), + }, + FilterType::Bloom { + bitmap, + num_bits, + number_of_items, + probability, + } => Self { + field_ids: filter.field_ids.clone(), + data: Some(pb::key_existence_filter::Data::Bloom(pb::BloomFilter { + bitmap: bitmap.clone(), + num_bits: *num_bits, + number_of_items: *number_of_items, + probability: *probability, + })), + }, + } + } +} + +impl TryFrom<&pb::KeyExistenceFilter> for KeyExistenceFilter { + type Error = lance_core::Error; + + fn try_from(message: &pb::KeyExistenceFilter) -> Result { + let filter = match message.data.as_ref() { + Some(pb::key_existence_filter::Data::Exact(exact)) => { + FilterType::ExactSet(exact.key_hashes.iter().copied().collect()) + } + Some(pb::key_existence_filter::Data::Bloom(b)) => { + // Use defaults for backwards compatibility + let number_of_items = if b.number_of_items == 0 { + BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS + } else { + b.number_of_items + }; + let probability = if b.probability == 0.0 { + BLOOM_FILTER_DEFAULT_PROBABILITY + } else { + b.probability + }; + FilterType::Bloom { + bitmap: b.bitmap.clone(), + num_bits: b.num_bits, + number_of_items, + probability, + } + } + None => FilterType::ExactSet(HashSet::new()), + }; + Ok(Self { + field_ids: message.field_ids.clone(), + filter, + }) + } +} + +/// Extract key value from a batch row. Returns None if null or unsupported type. +pub fn extract_key_value_from_batch( + batch: &RecordBatch, + row_idx: usize, + on_columns: &[String], +) -> Option { + let mut parts: Vec = Vec::with_capacity(on_columns.len()); + + for col_name in on_columns { + let (col_idx, _) = batch.schema().column_with_name(col_name)?; + let column = batch.column(col_idx); + + if column.is_null(row_idx) { + return None; + } + + let key_part = extract_key_value(column, row_idx)?; + parts.push(key_part); + } + + if parts.is_empty() { + None + } else if parts.len() == 1 { + Some(parts.into_iter().next().unwrap()) + } else { + Some(KeyValue::Composite(parts)) + } +} + +fn extract_key_value(array: &dyn Array, row_idx: usize) -> Option { + let v = match array.data_type() { + DataType::Utf8 => { + let arr = array.as_any().downcast_ref::()?; + KeyValue::String(arr.value(row_idx).to_string()) + } + DataType::LargeUtf8 => { + let arr = array.as_any().downcast_ref::()?; + KeyValue::String(arr.value(row_idx).to_string()) + } + DataType::UInt64 => { + let arr = array.as_primitive::(); + KeyValue::UInt64(arr.value(row_idx)) + } + DataType::Int64 => { + let arr = array.as_primitive::(); + KeyValue::Int64(arr.value(row_idx)) + } + DataType::UInt32 => { + let arr = array.as_primitive::(); + KeyValue::UInt64(arr.value(row_idx) as u64) + } + DataType::Int32 => { + let arr = array.as_primitive::(); + KeyValue::Int64(arr.value(row_idx) as i64) + } + DataType::Binary => { + let arr = array.as_any().downcast_ref::()?; + KeyValue::Binary(arr.value(row_idx).to_vec()) + } + DataType::LargeBinary => { + let arr = array.as_any().downcast_ref::()?; + KeyValue::Binary(arr.value(row_idx).to_vec()) + } + DataType::List(_) => { + let list_array = array.as_any().downcast_ref::().unwrap(); + let values = list_array.value(row_idx); + + let mut elements = Vec::with_capacity(values.len()); + for i in 0..values.len() { + if values.is_null(i) { + return None; + } + let element = extract_key_value(&values, i)?; + elements.push(element); + } + KeyValue::List(elements) + } + DataType::LargeList(_) => { + let list_array = array.as_any().downcast_ref::().unwrap(); + let values = list_array.value(row_idx); + + let mut elements = Vec::with_capacity(values.len()); + for i in 0..values.len() { + if values.is_null(i) { + return None; + } + let element = extract_key_value(&values, i)?; + elements.push(element); + } + KeyValue::List(elements) + } + DataType::Struct(_) => { + let struct_array = array.as_any().downcast_ref::()?; + let mut elements = Vec::with_capacity(struct_array.num_columns()); + for i in 0..struct_array.num_columns() { + let child = struct_array.column(i); + if child.is_null(row_idx) { + return None; + } + let field_value = extract_key_value(child.as_ref(), row_idx)?; + elements.push(field_value); + } + KeyValue::Struct(elements) + } + _ => return None, + }; + Some(v) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::sync::Arc; + + use arrow_array::builder::{Int32Builder, ListBuilder, StringBuilder}; + use arrow_array::{Int32Array, RecordBatch, StringArray, StructArray}; + use arrow_schema::{Field, Schema}; + + #[test] + fn test_extract_key_value_from_batch_list_int() { + let values_builder = Int32Builder::new(); + let mut list_builder = ListBuilder::new(values_builder); + + list_builder.append_value([Some(1), Some(2)]); + list_builder.append_value([Some(3), Some(4), Some(5)]); + + let list_array = list_builder.finish(); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + list_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) + .expect("batch should be valid"); + + let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("first row should produce a key"); + let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) + .expect("second row should produce a key"); + + match &key0 { + KeyValue::List(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::Int64(1)); + assert_eq!(values[1], KeyValue::Int64(2)); + } + other => panic!("expected list key, got {:?}", other), + } + + match &key1 { + KeyValue::List(values) => { + assert_eq!(values.len(), 3); + assert_eq!(values[0], KeyValue::Int64(3)); + assert_eq!(values[1], KeyValue::Int64(4)); + assert_eq!(values[2], KeyValue::Int64(5)); + } + other => panic!("expected list key, got {:?}", other), + } + + assert_ne!( + key0.hash_value(), + key1.hash_value(), + "different list values should hash differently", + ); + } + + #[test] + fn test_extract_key_value_from_batch_empty_list() { + let values_builder = Int32Builder::new(); + let mut list_builder = ListBuilder::new(values_builder); + + list_builder.append_value(std::iter::empty::>()); + + let list_array = list_builder.finish(); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + list_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) + .expect("batch should be valid"); + + let key = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("empty list should still produce a key"); + + match key { + KeyValue::List(values) => { + assert!(values.is_empty(), "expected empty list"); + } + other => panic!("expected list key, got {:?}", other), + } + } + + #[test] + fn test_extract_key_value_from_batch_list_utf8() { + let values_builder = StringBuilder::new(); + let mut list_builder = ListBuilder::new(values_builder); + + list_builder.append_value([Some("a"), Some("bc")]); + list_builder.append_value([Some("de")]); + + let list_array = list_builder.finish(); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + list_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) + .expect("batch should be valid"); + + let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("first row should produce a key"); + let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) + .expect("second row should produce a key"); + + match &key0 { + KeyValue::List(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::String("a".to_string())); + assert_eq!(values[1], KeyValue::String("bc".to_string())); + } + other => panic!("expected list key, got {:?}", other), + } + + match &key1 { + KeyValue::List(values) => { + assert_eq!(values.len(), 1); + assert_eq!(values[0], KeyValue::String("de".to_string())); + } + other => panic!("expected list key, got {:?}", other), + } + + assert_ne!( + key0.hash_value(), + key1.hash_value(), + "different list values should hash differently", + ); + } + + #[test] + fn test_extract_key_value_from_batch_list_with_null_child() { + let values_builder = Int32Builder::new(); + let mut list_builder = ListBuilder::new(values_builder); + + list_builder.append_value([Some(1), Some(2)]); + list_builder.append_value([Some(3), None]); + + let list_array = list_builder.finish(); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + list_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) + .expect("batch should be valid"); + + let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("first row should produce a key"); + let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]); + + match &key0 { + KeyValue::List(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::Int64(1)); + assert_eq!(values[1], KeyValue::Int64(2)); + } + other => panic!("expected list key, got {:?}", other), + } + + assert!( + key1.is_none(), + "list row with a null child should not produce a key", + ); + } + + #[test] + fn test_extract_key_value_from_batch_struct_int() { + let a_values = Int32Array::from(vec![1, 3]); + let b_values = Int32Array::from(vec![2, 4]); + + let struct_array = StructArray::from(vec![ + ( + Arc::new(Field::new("a", arrow_schema::DataType::Int32, false)), + Arc::new(a_values) as Arc, + ), + ( + Arc::new(Field::new("b", arrow_schema::DataType::Int32, false)), + Arc::new(b_values) as Arc, + ), + ]); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + struct_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(struct_array)]) + .expect("batch should be valid"); + + let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("first row should produce a key"); + let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) + .expect("second row should produce a key"); + + match &key0 { + KeyValue::Struct(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::Int64(1)); + assert_eq!(values[1], KeyValue::Int64(2)); + } + other => panic!("expected struct key, got {:?}", other), + } + + match &key1 { + KeyValue::Struct(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::Int64(3)); + assert_eq!(values[1], KeyValue::Int64(4)); + } + other => panic!("expected struct key, got {:?}", other), + } + + assert_ne!( + key0.hash_value(), + key1.hash_value(), + "different struct values should hash differently", + ); + } + + #[test] + fn test_extract_key_value_from_batch_struct_utf8() { + let first_names = StringArray::from(vec!["alice", "bob"]); + let last_names = StringArray::from(vec!["smith", "jones"]); + + let struct_array = StructArray::from(vec![ + ( + Arc::new(Field::new("first", arrow_schema::DataType::Utf8, false)), + Arc::new(first_names) as Arc, + ), + ( + Arc::new(Field::new("last", arrow_schema::DataType::Utf8, false)), + Arc::new(last_names) as Arc, + ), + ]); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + struct_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(struct_array)]) + .expect("batch should be valid"); + + let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("first row should produce a key"); + let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) + .expect("second row should produce a key"); + + match &key0 { + KeyValue::Struct(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::String("alice".to_string())); + assert_eq!(values[1], KeyValue::String("smith".to_string())); + } + other => panic!("expected struct key, got {:?}", other), + } + + match &key1 { + KeyValue::Struct(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::String("bob".to_string())); + assert_eq!(values[1], KeyValue::String("jones".to_string())); + } + other => panic!("expected struct key, got {:?}", other), + } + + assert_ne!( + key0.hash_value(), + key1.hash_value(), + "different struct values should hash differently", + ); + } + + #[test] + fn test_extract_key_value_from_batch_struct_with_null_child() { + let a_values = Int32Array::from(vec![Some(1), None]); + let b_values = Int32Array::from(vec![Some(2), Some(3)]); + + let struct_array = StructArray::from(vec![ + ( + Arc::new(Field::new("a", arrow_schema::DataType::Int32, true)), + Arc::new(a_values) as Arc, + ), + ( + Arc::new(Field::new("b", arrow_schema::DataType::Int32, true)), + Arc::new(b_values) as Arc, + ), + ]); + + let schema = Arc::new(Schema::new(vec![Field::new( + "id", + struct_array.data_type().clone(), + false, + )])); + + let batch = RecordBatch::try_new(schema, vec![Arc::new(struct_array)]) + .expect("batch should be valid"); + + let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) + .expect("first row should produce a key"); + let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]); + + match &key0 { + KeyValue::Struct(values) => { + assert_eq!(values.len(), 2); + assert_eq!(values[0], KeyValue::Int64(1)); + assert_eq!(values[1], KeyValue::Int64(2)); + } + other => panic!("expected struct key, got {:?}", other), + } + + assert!( + key1.is_none(), + "struct row with a null child should not produce a key", + ); + } +} diff --git a/rust/lance-table/src/format/manifest.rs b/rust/lance-table/src/format/manifest.rs index 83c9643e600..4ab30dddd57 100644 --- a/rust/lance-table/src/format/manifest.rs +++ b/rust/lance-table/src/format/manifest.rs @@ -639,6 +639,32 @@ impl From for DataStorageFormat { } } +/// Options controlling how a new [`Manifest`] is assembled from a transaction. +/// +/// The timestamp arrives already resolved to nanoseconds since the Unix epoch. +/// Callers own the clock so that a caller wanting a mockable one keeps it: the +/// `lance` crate mocks `SystemTime` under `cfg(test)`, which only takes effect in +/// that crate. +#[derive(Debug, Clone)] +pub struct ManifestBuildConfig { + /// Recompute the manifest's feature flags from the fragments and settings + /// below. False leaves whatever flags the previous manifest carried. + pub auto_set_feature_flags: bool, + /// Value for the new manifest's timestamp, in nanoseconds since the Unix epoch. + pub timestamp_nanos: u128, + /// Request the stable row id feature. The flag is also inherited from the + /// previous manifest, so false does not turn it off for a dataset that has it. + pub use_stable_row_ids: bool, + /// Overwrite only: force the legacy (true) or v2 (false) file format. `None` + /// keeps the format the dataset already had. + pub use_legacy_format: Option, + /// Overwrite only: force this storage format, taking precedence over + /// `use_legacy_format`. `None` keeps the format the dataset already had. + pub storage_format: Option, + /// Skip writing a detached transaction file for this commit. + pub disable_transaction_file: bool, +} + #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum VersionPart { Major, diff --git a/rust/lance-table/src/format/overlay.rs b/rust/lance-table/src/format/overlay.rs index 5e729411502..385afd39d9c 100644 --- a/rust/lance-table/src/format/overlay.rs +++ b/rust/lance-table/src/format/overlay.rs @@ -33,6 +33,8 @@ //! removed, so the overlay's other fields — and its coverage positions — stay //! intact (see [`tombstone_overlay_fields`]). +pub mod staleness; + use std::sync::Arc; use lance_core::Error; diff --git a/rust/lance-table/src/format/overlay/staleness.rs b/rust/lance-table/src/format/overlay/staleness.rs new file mode 100644 index 00000000000..dd5ff5d5e25 --- /dev/null +++ b/rust/lance-table/src/format/overlay/staleness.rs @@ -0,0 +1,413 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Which rows an overlay makes stale with respect to an index. +//! +//! An overlay supplies replacement values for some `(row, field)` cells without +//! rewriting the base data. An index built before that overlay was committed still +//! reflects the old values, so those rows must be excluded from the index's results +//! and re-evaluated against current values on the flat path. +//! +//! Deciding which rows are affected needs only fragment and index metadata — the +//! overlay coverage bitmaps, the overlay `committed_version`, and the indexed field +//! ids — so it lives here rather than in the read path that consumes it. + +use std::collections::HashMap; + +use lance_core::Result; +use lance_core::datatypes::Schema; +use roaring::RoaringBitmap; + +use crate::format::overlay::DataOverlayFile; +use crate::format::{Fragment, IndexMetadata}; + +/// The physical offsets within a fragment whose value for an indexed field may be +/// stale relative to an index built at `index_version`, and so must be excluded +/// from that index's results and re-evaluated against current values on the flat +/// path. +/// +/// The set is the union, over every overlay whose `committed_version` is newer +/// than `index_version`, of that overlay's coverage **restricted to the indexed +/// fields**. The restriction makes exclusion field-aware: an overlay that touches +/// only non-indexed fields contributes nothing. An overlay whose +/// `committed_version <= index_version` is already incorporated by the index and +/// is ignored. +pub fn overlay_exclusion_offsets( + overlays: &[DataOverlayFile], + indexed_field_ids: &[i32], + index_version: u64, + schema: &Schema, +) -> Result { + let mut excluded = RoaringBitmap::new(); + for overlay in overlays { + if overlay.committed_version <= index_version { + continue; + } + for (field_pos, field_id) in overlay.data_file.fields.iter().enumerate() { + let overlay_ancestry = schema.field_ancestry_by_id(*field_id); + let affects_index = indexed_field_ids.iter().any(|indexed_field_id| { + indexed_field_id == field_id + || overlay_ancestry.as_ref().is_some_and(|ancestry| { + ancestry + .iter() + .any(|ancestor| ancestor.id == *indexed_field_id) + }) + || schema + .field_ancestry_by_id(*indexed_field_id) + .is_some_and(|ancestry| { + ancestry.iter().any(|ancestor| ancestor.id == *field_id) + }) + }); + if affects_index { + excluded |= &*overlay.coverage_for_field(field_pos)?; + } + } + } + Ok(excluded) +} + +// Stale row offsets contributed by one fragment's overlays for a given index version. +// Applies a cheap version gate first: if every overlay predates the segment it is already +// incorporated by the index, so there is nothing stale and the field/bitmap work is skipped. +fn stale_offsets_for_fragment( + fragment: &Fragment, + fields: &[i32], + index_version: u64, + schema: &Schema, +) -> Result { + if fragment + .overlays + .iter() + .all(|o| o.committed_version <= index_version) + { + return Ok(RoaringBitmap::new()); + } + overlay_exclusion_offsets(&fragment.overlays, fields, index_version, schema) +} + +// A missing `fragment_bitmap` means the index predates fragment-bitmap tracking; treat it as +// covering every fragment (matching `DatasetPreFilter::new`) so overlay-stale rows can't slip +// through unmasked. Only skip fragments explicitly absent from a present bitmap. +fn covers_fragment(coverage: Option<&RoaringBitmap>, frag_id: u32) -> bool { + coverage.is_none_or(|c| c.contains(frag_id)) +} + +/// Index by fragment id the fragments that carry at least one overlay. Overlays are rare, so +/// this is empty on the common path, letting callers skip index loading entirely; when non-empty +/// it bounds the stale-collection loops to `O(overlaid fragments)`. +pub fn overlaid_fragments(fragments: &[Fragment]) -> HashMap { + fragments + .iter() + .filter(|f| !f.overlays.is_empty()) + .map(|f| (f.id as u32, f)) + .collect() +} + +/// Insert into `stale` the ids of fragments covered by `segment` whose index entries may be +/// stale because an overlay committed after the segment was built touches a field the segment +/// indexes. Field-aware and version-gated via [`overlay_exclusion_offsets`]. +/// +/// `overlaid_frags` holds only the fragments that actually carry overlays (rare), so the loop is +/// `O(overlaid_frags)` rather than `O(fragments the segment covers)`. +pub fn collect_overlay_stale_frags( + segment: &IndexMetadata, + overlaid_frags: &HashMap, + stale: &mut RoaringBitmap, + schema: &Schema, +) -> Result<()> { + let coverage = segment.fragment_bitmap.as_ref(); + for (&frag_id, fragment) in overlaid_frags { + if stale.contains(frag_id) || !covers_fragment(coverage, frag_id) { + continue; + } + if !stale_offsets_for_fragment(fragment, &segment.fields, segment.dataset_version, schema)? + .is_empty() + { + stale.insert(frag_id); + } + } + Ok(()) +} + +/// Like [`collect_overlay_stale_frags`] but with row-level granularity: instead of marking the +/// whole fragment stale, it computes exactly which row offsets within each covered fragment are +/// stale and accumulates them into `stale` (fragment_id → stale row offsets). +/// +/// Used by the scalar and vector paths to block only the affected rows from index results and +/// re-evaluate only those rows on the flat path, keeping overhead proportional to the number of +/// overlaid rows rather than the whole fragment size. +pub fn collect_overlay_stale_rows_for_segment( + segment: &IndexMetadata, + overlaid_frags: &HashMap, + stale: &mut HashMap, + schema: &Schema, +) -> Result<()> { + let coverage = segment.fragment_bitmap.as_ref(); + for (&frag_id, fragment) in overlaid_frags { + if !covers_fragment(coverage, frag_id) { + continue; + } + let excluded = + stale_offsets_for_fragment(fragment, &segment.fields, segment.dataset_version, schema)?; + if !excluded.is_empty() { + *stale.entry(frag_id).or_default() |= &excluded; + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::DataFile; + use crate::format::overlay::OverlayCoverage; + + fn bitmap(offsets: impl IntoIterator) -> RoaringBitmap { + RoaringBitmap::from_iter(offsets) + } + + fn flat_test_schema() -> Schema { + use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; + + let mut schema = Schema::try_from(&ArrowSchema::new( + (0..5) + .map(|id| ArrowField::new(format!("field_{id}"), DataType::Int32, true)) + .collect::>(), + )) + .unwrap(); + schema.set_field_id(None); + schema + } + + /// `outer: struct>`, for the ancestry checks. + fn nested_struct_schema() -> Schema { + use arrow_schema::{DataType, Field as ArrowField, Fields, Schema as ArrowSchema}; + + let mid = Fields::from(vec![ + ArrowField::new("a", DataType::Int32, true), + ArrowField::new("b", DataType::Int32, true), + ]); + let outer_fields = + Fields::from(vec![ArrowField::new("middle", DataType::Struct(mid), true)]); + let mut schema = Schema::try_from(&ArrowSchema::new(vec![ArrowField::new( + "outer", + DataType::Struct(outer_fields), + true, + )])) + .unwrap(); + schema.set_field_id(None); + schema + } + + /// A dense overlay covering `offsets` for `field_ids`, committed at `version`. + fn dense_overlay( + field_ids: Vec, + offsets: impl IntoIterator, + version: u64, + ) -> DataOverlayFile { + DataOverlayFile { + data_file: DataFile::new_legacy_from_fields("o.lance", field_ids, None), + coverage: OverlayCoverage::dense(bitmap(offsets)), + committed_version: version, + } + } + + #[test] + fn test_exclusion_offsets_version_gate() { + let schema = flat_test_schema(); + // index built at version 5; only overlays committed > 5 are excluded. + let overlays = vec![ + dense_overlay(vec![3], [0, 1], 4), + dense_overlay(vec![3], [2, 7], 6), + ]; + let excluded = overlay_exclusion_offsets(&overlays, &[3], 5, &schema).unwrap(); + assert_eq!(excluded, bitmap([2, 7])); + // An overlay exactly at the index version is already incorporated. + let overlays = vec![dense_overlay(vec![3], [9], 5)]; + assert!( + overlay_exclusion_offsets(&overlays, &[3], 5, &schema) + .unwrap() + .is_empty() + ); + } + + #[test] + fn test_exclusion_offsets_is_field_aware() { + let schema = flat_test_schema(); + // An overlay touching only an unrelated field excludes nothing. + let overlays = vec![dense_overlay(vec![2], [0, 1, 2], 9)]; + assert!( + overlay_exclusion_offsets(&overlays, &[3], 1, &schema) + .unwrap() + .is_empty() + ); + // The union spans only the indexed fields the overlay actually carries. + let overlays = vec![dense_overlay(vec![2, 3], [4], 9)]; + assert_eq!( + overlay_exclusion_offsets(&overlays, &[3], 1, &schema).unwrap(), + bitmap([4]) + ); + } + + #[test] + fn test_exclusion_offsets_matches_nested_fields() { + let schema = nested_struct_schema(); + let outer = &schema.fields[0]; + let middle = &outer.children[0]; + let a = &middle.children[0]; + let b = &middle.children[1]; + + let overlays = vec![dense_overlay(vec![a.id], [1], 9)]; + assert_eq!( + overlay_exclusion_offsets(&overlays, &[outer.id], 1, &schema).unwrap(), + bitmap([1]) + ); + assert!( + overlay_exclusion_offsets(&overlays, &[b.id], 1, &schema) + .unwrap() + .is_empty() + ); + + let overlays = vec![dense_overlay(vec![middle.id], [2], 9)]; + assert_eq!( + overlay_exclusion_offsets(&overlays, &[a.id], 1, &schema).unwrap(), + bitmap([2]) + ); + } + + #[test] + fn test_exclusion_offsets_sparse_per_field() { + let schema = flat_test_schema(); + // Sparse overlay: field 2 covers {2,3}, field 4 covers {1}. + let overlay = DataOverlayFile { + data_file: DataFile::new_legacy_from_fields("o.lance", vec![2, 4], None), + coverage: OverlayCoverage::sparse(vec![bitmap([2, 3]), bitmap([1])]), + committed_version: 9, + }; + let overlays = vec![overlay]; + // Only the bitmap for the indexed field (4) contributes. + assert_eq!( + overlay_exclusion_offsets(&overlays, &[4], 1, &schema).unwrap(), + bitmap([1]) + ); + assert_eq!( + overlay_exclusion_offsets(&overlays, &[2], 1, &schema).unwrap(), + bitmap([2, 3]) + ); + } + + #[test] + fn test_exclusion_offsets_unions_multiple_overlays() { + let schema = flat_test_schema(); + let overlays = vec![ + dense_overlay(vec![3], [1], 6), + dense_overlay(vec![3], [4, 5], 7), + ]; + assert_eq!( + overlay_exclusion_offsets(&overlays, &[3], 1, &schema).unwrap(), + bitmap([1, 4, 5]) + ); + } + + /// An index segment covering `fields`, built at `dataset_version`, with the given + /// fragment coverage (`None` = legacy index predating fragment-bitmap tracking). + fn segment( + fields: Vec, + dataset_version: u64, + fragment_bitmap: Option, + ) -> IndexMetadata { + IndexMetadata { + uuid: uuid::Uuid::new_v4(), + name: "idx".into(), + fields, + dataset_version, + fragment_bitmap, + index_details: None, + index_version: 0, + created_at: None, + base_id: None, + files: None, + } + } + + fn fragment_with_overlay(id: u64, overlay: DataOverlayFile) -> Fragment { + let mut fragment = Fragment::new(id); + fragment.overlays.push(overlay); + fragment + } + + #[test] + fn test_collect_frags_missing_bitmap_covers_all() { + let schema = flat_test_schema(); + // A segment with no fragment_bitmap (legacy index predating bitmap tracking) must treat + // every overlaid fragment as covered so stale rows can't leak past the index unmasked. + let fragment = fragment_with_overlay(3, dense_overlay(vec![3], [1, 2], 9)); + let overlaid: HashMap = HashMap::from([(3u32, &fragment)]); + + let mut stale = RoaringBitmap::new(); + collect_overlay_stale_frags(&segment(vec![3], 1, None), &overlaid, &mut stale, &schema) + .unwrap(); + assert_eq!(stale, bitmap([3]), "missing bitmap must cover fragment 3"); + + // A present bitmap that excludes fragment 3 leaves it untouched. + let mut stale = RoaringBitmap::new(); + collect_overlay_stale_frags( + &segment(vec![3], 1, Some(bitmap([0]))), + &overlaid, + &mut stale, + &schema, + ) + .unwrap(); + assert!( + stale.is_empty(), + "fragment absent from bitmap is not covered" + ); + + // A present bitmap that includes fragment 3 marks it stale. + let mut stale = RoaringBitmap::new(); + collect_overlay_stale_frags( + &segment(vec![3], 1, Some(bitmap([3]))), + &overlaid, + &mut stale, + &schema, + ) + .unwrap(); + assert_eq!(stale, bitmap([3])); + } + + #[test] + fn test_collect_rows_missing_bitmap_covers_all() { + let schema = flat_test_schema(); + // Same covers-all guarantee at row-level granularity. + let fragment = fragment_with_overlay(3, dense_overlay(vec![3], [1, 2], 9)); + let overlaid: HashMap = HashMap::from([(3u32, &fragment)]); + + let mut stale = HashMap::new(); + collect_overlay_stale_rows_for_segment( + &segment(vec![3], 1, None), + &overlaid, + &mut stale, + &schema, + ) + .unwrap(); + assert_eq!( + stale.get(&3), + Some(&bitmap([1, 2])), + "missing bitmap must cover fragment 3" + ); + + // A present bitmap that excludes fragment 3 yields no stale rows. + let mut stale = HashMap::new(); + collect_overlay_stale_rows_for_segment( + &segment(vec![3], 1, Some(bitmap([0]))), + &overlaid, + &mut stale, + &schema, + ) + .unwrap(); + assert!( + stale.is_empty(), + "fragment absent from bitmap contributes no rows" + ); + } +} diff --git a/rust/lance-table/src/lib.rs b/rust/lance-table/src/lib.rs index 89b424adc61..1a008740439 100644 --- a/rust/lance-table/src/lib.rs +++ b/rust/lance-table/src/lib.rs @@ -6,4 +6,5 @@ pub mod format; pub mod io; pub mod rowids; pub mod system_index; +pub mod transaction; pub mod utils; diff --git a/rust/lance-table/src/system_index.rs b/rust/lance-table/src/system_index.rs index 021c01a5e52..c315b1af326 100644 --- a/rust/lance-table/src/system_index.rs +++ b/rust/lance-table/src/system_index.rs @@ -13,3 +13,12 @@ pub mod frag_reuse; pub mod mem_wal; + +use crate::format::IndexMetadata; +use frag_reuse::FRAG_REUSE_INDEX_NAME; +use mem_wal::MEM_WAL_INDEX_NAME; + +/// Whether `index_meta` describes one of the system indices defined in this module. +pub fn is_system_index(index_meta: &IndexMetadata) -> bool { + index_meta.name == FRAG_REUSE_INDEX_NAME || index_meta.name == MEM_WAL_INDEX_NAME +} diff --git a/rust/lance-table/src/system_index/mem_wal.rs b/rust/lance-table/src/system_index/mem_wal.rs index d3f5a157e94..4feebdea7d5 100644 --- a/rust/lance-table/src/system_index/mem_wal.rs +++ b/rust/lance-table/src/system_index/mem_wal.rs @@ -2,13 +2,14 @@ // SPDX-FileCopyrightText: Copyright The Lance Authors use std::collections::HashMap; +use std::sync::Arc; -use lance_core::Error; use lance_core::deepsize::DeepSizeOf; +use lance_core::{Error, Result}; use serde::{Deserialize, Serialize}; use uuid::Uuid; -use crate::format::pb; +use crate::format::{IndexMetadata, pb}; pub const MEM_WAL_INDEX_NAME: &str = "__lance_mem_wal"; @@ -432,3 +433,102 @@ impl MemWalIndex { caught_up_gen.is_none_or(|generation| generation >= compacted_gen) } } + +// Reading and updating the `IndexMetadata` entry that carries the details above. + +/// Load MemWalIndexDetails from an IndexMetadata. +pub fn load_mem_wal_index_details(index: IndexMetadata) -> Result { + if let Some(details_any) = index.index_details.as_ref() { + if !details_any.type_url.ends_with("MemWalIndexDetails") { + return Err(Error::index(format!( + "Index details is not for the MemWAL index, but {}", + details_any.type_url + ))); + } + + Ok(MemWalIndexDetails::try_from( + details_any.to_msg::()?, + )?) + } else { + Err(Error::index("Index details not found for the MemWAL index")) + } +} + +/// Open the MemWAL index from its metadata. +pub fn open_mem_wal_index(index: IndexMetadata) -> Result> { + Ok(Arc::new(MemWalIndex::new(load_mem_wal_index_details( + index, + )?))) +} + +/// Update `compacted_sstables` in the MemWAL index. +/// +/// This is called during merge-insert commits to atomically record which +/// SSTables have been compacted into the base table. +pub fn update_mem_wal_index_compacted_sstables( + indices: &mut Vec, + dataset_version: u64, + new_compacted_sstables: Vec, +) -> Result<()> { + if new_compacted_sstables.is_empty() { + return Ok(()); + } + + let pos = indices + .iter() + .position(|idx| idx.name == MEM_WAL_INDEX_NAME); + + let new_meta = if let Some(pos) = pos { + let current_meta = indices.remove(pos); + let mut details = load_mem_wal_index_details(current_meta)?; + + // Update compacted_sstables - for each shard, keep the higher generation + for new_sstable in new_compacted_sstables { + if let Some(existing) = details + .compacted_sstables + .iter_mut() + .find(|sstable| sstable.shard_id == new_sstable.shard_id) + { + if new_sstable.generation > existing.generation { + existing.generation = new_sstable.generation; + } + } else { + details.compacted_sstables.push(new_sstable); + } + } + + new_mem_wal_index_meta(dataset_version, details)? + } else { + // Create a MemWAL index containing only compaction progress. + let details = MemWalIndexDetails { + compacted_sstables: new_compacted_sstables, + ..Default::default() + }; + new_mem_wal_index_meta(dataset_version, details)? + }; + + indices.push(new_meta); + Ok(()) +} + +/// Create a new MemWAL index metadata entry. +pub fn new_mem_wal_index_meta( + dataset_version: u64, + details: MemWalIndexDetails, +) -> Result { + Ok(IndexMetadata { + uuid: Uuid::new_v4(), + name: MEM_WAL_INDEX_NAME.to_string(), + fields: vec![], + dataset_version, + fragment_bitmap: None, + index_details: Some(Arc::new(prost_types::Any::from_msg( + &pb::MemWalIndexDetails::from(&details), + )?)), + index_version: 0, + created_at: Some(chrono::Utc::now()), + base_id: None, + // Memory WAL index is inline (no files) + files: None, + }) +} diff --git a/rust/lance-table/src/transaction.rs b/rust/lance-table/src/transaction.rs new file mode 100644 index 00000000000..283540012c8 --- /dev/null +++ b/rust/lance-table/src/transaction.rs @@ -0,0 +1,54 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Transaction definitions for updating datasets +//! +//! Prior to creating a new manifest, a transaction must be created representing +//! the changes being made to the dataset. By representing them as incremental +//! changes, we can detect whether concurrent operations are compatible with +//! one another. We can also rebuild manifests when retrying committing a +//! manifest. +//! +//! For more details please refer to the +//! [Transaction Specification](https://lance.org/format/table/transaction/#transaction-types). +//! +//! The work splits along these lines: +//! +//! ```text +//! builder Transaction: an operation plus the version it was based on +//! operation the vocabulary of changes an operation can describe +//! update_map incremental edits to the manifest's string maps +//! validate pre-commit checks against the manifest being replaced +//! manifest_build applying an operation to produce the next manifest +//! index_maintenance how that narrows or drops index metadata +//! row_version how it assigns row ids and per-row version metadata +//! conflicts whether two operations collide, for the commit retry path +//! action the composable action vocabulary, and its own apply path +//! proto the persisted protobuf encoding of all of the above +//! ``` + +mod action; +mod builder; +mod conflicts; +mod index_maintenance; +mod manifest_build; +mod operation; +mod proto; +mod row_version; +mod update_map; +mod validate; + +#[cfg(test)] +pub(crate) mod test_support; + +pub use action::mask::{Conflict, Key, ManifestMask, Region, Scope}; +pub use action::{Action, AddBase, Ref, UserAction, UserOperation}; +pub use builder::{Transaction, TransactionBuilder}; +pub use operation::{ + DataOverlayGroup, DataReplacementGroup, Operation, RewriteGroup, RewrittenIndex, UpdateMode, + UpdatedFragmentOffsets, +}; +pub use update_map::{ + UpdateMap, UpdateMapEntry, translate_config_updates, translate_schema_metadata_updates, +}; +pub use validate::validate_operation; diff --git a/rust/lance-table/src/transaction/action.rs b/rust/lance-table/src/transaction/action.rs new file mode 100644 index 00000000000..1f94bf42078 --- /dev/null +++ b/rust/lance-table/src/transaction/action.rs @@ -0,0 +1,298 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Action-based transactions: composable manifest deltas. +//! +//! A [`UserOperation`] is an ordered list of [`UserAction`] steps, each expanding +//! to granular [`Action`] deltas. An action records the *change* to the manifest +//! rather than a post-image, which is what lets several of them commit atomically +//! and lets two branches be merged. Minted identifiers travel as [`Ref::Local`] +//! tokens and are allocated when the operation is applied, so two concurrent +//! operations can never collide on an id. +//! +//! The wire format is `protos/transaction/actions.proto`, which holds the design +//! rationale and is authoritative for the vocabulary. +//! +//! The split of responsibility here is deliberate: +//! +//! ```text +//! action the list of actions, and a router with one line per variant +//! action/mask the manifest regions an action writes, and how they collide +//! action/refs Local/Committed references and the apply-time bindings +//! action/apply assembling a manifest from an operation's actions +//! action/translate the legacy Operation vocabulary expressed as actions +//! action/ one module per action: its struct, wire form, logic, tests +//! ``` +//! +//! Every method on [`Action`] is a one-line-per-variant match. If a router method +//! grows a body, the body belongs in the action's own module. + +pub mod add_base; +pub mod apply; +pub mod mask; +pub mod refs; +pub mod translate; + +pub use add_base::AddBase; +pub use refs::Ref; + +use crate::format::{Manifest, pb}; +use mask::ManifestMask; +use refs::TxnContext; + +use lance_core::deepsize::DeepSizeOf; +use lance_core::{Error, Result}; + +/// A single granular change to the manifest. +#[derive(Debug, Clone, PartialEq, DeepSizeOf)] +pub enum Action { + AddBase(AddBase), +} + +impl Action { + pub(crate) fn apply(&self, manifest: &mut Manifest, ctx: &mut TxnContext) -> Result<()> { + match self { + Self::AddBase(action) => action.apply(manifest, ctx), + } + } + + /// The manifest regions this action writes. Conflict is intersection; see + /// [`ManifestMask`]. + pub(crate) fn writes(&self) -> ManifestMask { + match self { + Self::AddBase(action) => action.writes(), + } + } +} + +/// A single user-recognizable step within a [`UserOperation`], e.g. "append +/// batch". The description keeps the transaction history readable when a range of +/// transactions is squashed. +#[derive(Debug, Clone, PartialEq, DeepSizeOf)] +pub struct UserAction { + pub description: String, + pub actions: Vec, +} + +/// A user-facing, composable transaction: an ordered list of steps that commit +/// atomically as a single manifest change. +#[derive(Debug, Clone, PartialEq, DeepSizeOf)] +pub struct UserOperation { + /// Human-readable description, e.g. `"INSERT INTO t VALUES (1)"`. + pub description: String, + /// Unique identifier for this operation, matching `Transaction::uuid` semantics. + pub uuid: String, + /// The dataset version this operation was planned against. + pub read_version: u64, + pub actions: Vec, +} + +impl UserOperation { + /// Every action in the operation, in application order. + pub(crate) fn actions(&self) -> impl Iterator { + self.actions.iter().flat_map(|step| step.actions.iter()) + } + + /// The union of the operation's actions' footprints. + pub(crate) fn writes(&self) -> ManifestMask { + self.actions() + .fold(ManifestMask::default(), |mask, action| { + mask.union(action.writes()) + }) + } +} + +impl TryFrom for Action { + type Error = Error; + + fn try_from(message: pb::Action) -> Result { + match message.action { + Some(pb::action::Action::AddBase(action)) => Ok(Self::AddBase(action.try_into()?)), + // An unrecognized arm — including one proto3 decodes to `None` because + // it was written by a newer version — must error, never be skipped. + // `load_and_sort_new_transactions` collects concurrent transactions with + // `try_collect`, so failing here aborts the commit; parsing leniently + // would instead drop a concurrent action out of conflict detection and + // let two colliding commits both succeed. Do NOT make this lenient. + Some(other) => Err(Error::not_supported(format!( + "this version of Lance does not support the transaction action {other:?}" + ))), + None => Err(Error::not_supported( + "Action message did not contain a recognized action; \ + it was written by a newer version of Lance", + )), + } + } +} + +impl From<&Action> for pb::Action { + fn from(action: &Action) -> Self { + let action = match action { + Action::AddBase(action) => pb::action::Action::AddBase(action.into()), + }; + Self { + action: Some(action), + } + } +} + +impl TryFrom for UserAction { + type Error = Error; + + fn try_from(message: pb::UserAction) -> Result { + Ok(Self { + description: message.description, + actions: message + .actions + .into_iter() + .map(Action::try_from) + .collect::>>()?, + }) + } +} + +impl From<&UserAction> for pb::UserAction { + fn from(step: &UserAction) -> Self { + Self { + description: step.description.clone(), + actions: step.actions.iter().map(pb::Action::from).collect(), + } + } +} + +impl TryFrom for UserOperation { + type Error = Error; + + fn try_from(message: pb::UserOperation) -> Result { + Ok(Self { + description: message.description, + uuid: message.uuid, + read_version: message.read_version, + actions: message + .actions + .into_iter() + .map(UserAction::try_from) + .collect::>>()?, + }) + } +} + +impl From<&UserOperation> for pb::UserOperation { + fn from(operation: &UserOperation) -> Self { + Self { + description: operation.description.clone(), + uuid: operation.uuid.clone(), + read_version: operation.read_version, + actions: operation.actions.iter().map(pb::UserAction::from).collect(), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn add_base(local: u32, name: Option<&str>, path: &str) -> Action { + Action::AddBase(AddBase { + local, + name: name.map(str::to_string), + is_dataset_root: false, + path: path.to_string(), + }) + } + + fn two_step_operation() -> UserOperation { + UserOperation { + description: "ALTER TABLE t ADD BASE".to_string(), + uuid: "9f0b8f4e-0000-4000-8000-000000000000".to_string(), + read_version: 3, + actions: vec![ + UserAction { + description: "register warm base".to_string(), + actions: vec![add_base(0, Some("warm"), "s3://bucket/warm")], + }, + UserAction { + description: "register unnamed base".to_string(), + actions: vec![add_base(1, None, "/local/cold")], + }, + ], + } + } + + #[test] + fn test_user_operation_roundtrips() { + let operation = two_step_operation(); + let message = pb::UserOperation::from(&operation); + assert_eq!(UserOperation::try_from(message).unwrap(), operation); + } + + #[test] + fn test_action_without_arm_is_rejected() { + let err = Action::try_from(pb::Action { action: None }).unwrap_err(); + assert!( + matches!(err, Error::NotSupported { .. }), + "expected NotSupported, got: {err:?}" + ); + assert!(err.to_string().contains("newer version of Lance")); + } + + #[test] + fn test_user_operation_containing_undecodable_action_is_rejected() { + // Fail-closed must propagate through the envelope: a `UserOperation` whose + // action list holds one undecodable entry is itself undecodable. + let message = pb::UserOperation { + description: "op".to_string(), + uuid: "u".to_string(), + read_version: 1, + actions: vec![pb::UserAction { + description: "step".to_string(), + actions: vec![pb::Action { action: None }], + }], + }; + let err = UserOperation::try_from(message).unwrap_err(); + assert!(matches!(err, Error::NotSupported { .. })); + } + + #[test] + fn test_unsupported_action_arm_is_rejected() { + // An arm this version does not implement yet must be rejected too, not + // silently ignored. + let message = pb::Action { + action: Some(pb::action::Action::ResetTable(pb::ResetTable {})), + }; + let err = Action::try_from(message).unwrap_err(); + assert!(matches!(err, Error::NotSupported { .. })); + assert!(err.to_string().contains("does not support")); + } + + #[test] + fn test_writes_unions_across_steps() { + let mask = two_step_operation().writes(); + assert_eq!( + mask.conflicts_with(&add_base(0, Some("warm"), "s3://elsewhere").writes()), + Some(mask::Conflict::Incompatible) + ); + assert_eq!( + mask.conflicts_with(&add_base(0, None, "/local/cold").writes()), + Some(mask::Conflict::Incompatible) + ); + assert_eq!( + mask.conflicts_with(&add_base(0, Some("other"), "/local/other").writes()), + None + ); + } + + #[test] + fn test_writes_of_empty_operation_is_disjoint_from_everything() { + let empty = UserOperation { + description: String::new(), + uuid: String::new(), + read_version: 0, + actions: vec![], + }; + assert_eq!( + empty.writes().conflicts_with(&ManifestMask::everything()), + None + ); + } +} diff --git a/rust/lance-table/src/transaction/action/add_base.rs b/rust/lance-table/src/transaction/action/add_base.rs new file mode 100644 index 00000000000..f9d2913132e --- /dev/null +++ b/rust/lance-table/src/transaction/action/add_base.rs @@ -0,0 +1,304 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Minting a new base path. + +use super::mask::{Key, ManifestMask, Region}; +use super::refs::TxnContext; +use crate::format::pb; +use crate::format::{BasePath, Manifest}; +use lance_core::deepsize::DeepSizeOf; +use lance_core::{Error, Result}; + +/// Mint a new base path. +/// +/// The `BasePath` fields are flattened rather than embedded so the meaningless +/// `id` — which `actions.proto` documents as ignored — has no in-memory +/// representation; [`AddBase::apply`] constructs the `BasePath` with the id it +/// mints. +#[derive(Debug, Clone, PartialEq, DeepSizeOf)] +pub struct AddBase { + /// Token for the base id this action mints, resolved at apply. Must be + /// distinct from every other local token in the same operation. + pub local: u32, + /// Optional human-readable alias, matching `BasePath::name`. `None` means the + /// base is addressed only by path. + pub name: Option, + /// Whether this base is a dataset root rather than a plain file directory. + pub is_dataset_root: bool, + /// The full URI, e.g. `"s3://bucket/path"`, matching `BasePath::path`. Not an + /// `object_store::Path`, which is store-relative and percent-normalized and + /// so would drop the scheme and bucket. + pub path: String, +} + +impl AddBase { + pub(super) fn apply(&self, manifest: &mut Manifest, ctx: &mut TxnContext) -> Result<()> { + // A name or path already in use cannot be freed by retrying, so the + // collision check runs before anything is minted. + if let Some(existing) = manifest + .base_paths + .values() + .find(|base| (self.name.is_some() && base.name == self.name) || base.path == self.path) + { + return Err(Error::invalid_input(format!( + "Conflict detected: Base path with name '{:?}' or path '{}' already exists. Existing: name='{:?}', path='{}'", + self.name, self.path, existing.name, existing.path + ))); + } + + // Base ids are *not* a manifest counter: there is no `max_base_id` field in + // `Manifest` or `table.proto`, so the next id is derived from the current + // maximum key. Nothing in the tree removes a base, so ids happen to be + // monotone — but only by that accident. Relocation-by-watermark + // (actions.proto principle 4) assumes a never-reused counter, so OSS-1529 + // needs either a real counter field or a written-down never-remove + // invariant. Within one lineage, which is all this does, deriving from the + // max is equivalent. + let minted = match manifest.base_paths.keys().max() { + Some(&max) => max.checked_add(1).ok_or_else(|| { + Error::invalid_input(format!( + "cannot mint a base id: the maximum base id {max} is already u32::MAX" + )) + })?, + None => 1, + }; + + manifest.base_paths.insert( + minted, + BasePath::new( + minted, + self.path.clone(), + self.name.clone(), + self.is_dataset_root, + ), + ); + ctx.bind_base(self.local, minted) + } + + pub(super) fn writes(&self) -> ManifestMask { + let mut keys = vec![Key::Path(self.path.clone())]; + keys.extend(self.name.clone().map(Key::Name)); + ManifestMask::default().with_keys(Region::Bases, keys) + } +} + +impl TryFrom for AddBase { + type Error = Error; + + fn try_from(message: pb::AddBase) -> Result { + let base = message + .base + .ok_or_else(|| Error::invalid_input("AddBase message did not contain a base path"))?; + Ok(Self { + local: message.local, + name: base.name, + is_dataset_root: base.is_dataset_root, + path: base.path, + }) + } +} + +impl From<&AddBase> for pb::AddBase { + fn from(action: &AddBase) -> Self { + Self { + local: action.local, + base: Some(pb::BasePath { + // Ignored on read; the id is stamped in at apply. + id: 0, + name: action.name.clone(), + is_dataset_root: action.is_dataset_root, + path: action.path.clone(), + }), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::DataStorageFormat; + use lance_core::datatypes::Schema; + use std::collections::HashMap; + use std::sync::Arc; + + fn manifest_with_bases(bases: Vec) -> Manifest { + let mut manifest = Manifest::new( + Schema::default(), + Arc::new(Vec::new()), + DataStorageFormat::default(), + HashMap::new(), + ); + manifest.base_paths = bases.into_iter().map(|base| (base.id, base)).collect(); + manifest + } + + fn add_base(local: u32, name: Option<&str>, path: &str) -> AddBase { + AddBase { + local, + name: name.map(str::to_string), + is_dataset_root: false, + path: path.to_string(), + } + } + + #[test] + fn test_apply_mints_first_id_on_empty_base_map() { + let mut manifest = manifest_with_bases(vec![]); + let mut ctx = TxnContext::default(); + + add_base(0, Some("warm"), "s3://bucket/warm") + .apply(&mut manifest, &mut ctx) + .unwrap(); + + assert_eq!(manifest.base_paths.len(), 1); + let base = &manifest.base_paths[&1]; + assert_eq!(base.id, 1); + assert_eq!(base.name.as_deref(), Some("warm")); + assert_eq!(base.path, "s3://bucket/warm"); + assert!(!base.is_dataset_root); + } + + #[test] + fn test_apply_mints_above_existing_max() { + // Not `len() + 1`: the id comes from the maximum existing key, so a gap in + // the map does not cause a collision. + let mut manifest = manifest_with_bases(vec![ + BasePath::new(1, "s3://bucket/a".into(), Some("a".into()), false), + BasePath::new(7, "s3://bucket/b".into(), Some("b".into()), false), + ]); + let mut ctx = TxnContext::default(); + + add_base(0, Some("c"), "s3://bucket/c") + .apply(&mut manifest, &mut ctx) + .unwrap(); + + assert_eq!(manifest.base_paths[&8].name.as_deref(), Some("c")); + } + + #[test] + fn test_two_actions_mint_distinct_ids_and_bind_both() { + let mut manifest = manifest_with_bases(vec![]); + let mut ctx = TxnContext::default(); + + add_base(0, Some("warm"), "s3://bucket/warm") + .apply(&mut manifest, &mut ctx) + .unwrap(); + add_base(1, Some("cold"), "s3://bucket/cold") + .apply(&mut manifest, &mut ctx) + .unwrap(); + + assert_eq!(ctx.bound_base(0), Some(1)); + assert_eq!(ctx.bound_base(1), Some(2)); + assert_eq!(manifest.base_paths.len(), 2); + } + + #[test] + fn test_apply_rejects_duplicate_name() { + let mut manifest = manifest_with_bases(vec![BasePath::new( + 1, + "s3://bucket/a".into(), + Some("warm".into()), + false, + )]); + let err = add_base(0, Some("warm"), "s3://bucket/other") + .apply(&mut manifest, &mut TxnContext::default()) + .unwrap_err(); + + assert!( + matches!(err, Error::InvalidInput { .. }), + "expected InvalidInput, got: {err:?}" + ); + assert!(err.to_string().contains("already exists")); + assert_eq!(manifest.base_paths.len(), 1); + } + + #[test] + fn test_apply_rejects_duplicate_path() { + let mut manifest = manifest_with_bases(vec![BasePath::new( + 1, + "s3://bucket/a".into(), + Some("warm".into()), + false, + )]); + let err = add_base(0, Some("cold"), "s3://bucket/a") + .apply(&mut manifest, &mut TxnContext::default()) + .unwrap_err(); + + assert!(matches!(err, Error::InvalidInput { .. })); + assert!(err.to_string().contains("already exists")); + } + + #[test] + fn test_unnamed_bases_do_not_collide_on_name() { + // Two bases may both be unnamed: `None` is the absence of a claim, not a + // name that can be taken. (The legacy apply path compares `name == name` + // unconditionally and so wrongly rejects this; see the PR description.) + let mut manifest = + manifest_with_bases(vec![BasePath::new(1, "s3://bucket/a".into(), None, false)]); + add_base(0, None, "s3://bucket/b") + .apply(&mut manifest, &mut TxnContext::default()) + .unwrap(); + + assert_eq!(manifest.base_paths.len(), 2); + assert_eq!(manifest.base_paths[&2].path, "s3://bucket/b"); + } + + #[test] + fn test_writes_claims_name_and_path() { + let named = add_base(0, Some("warm"), "s3://bucket/warm").writes(); + assert_eq!( + named.conflicts_with(&add_base(1, Some("warm"), "s3://bucket/other").writes()), + Some(super::super::mask::Conflict::Incompatible) + ); + assert_eq!( + named.conflicts_with(&add_base(1, Some("cold"), "s3://bucket/warm").writes()), + Some(super::super::mask::Conflict::Incompatible) + ); + assert_eq!( + named.conflicts_with(&add_base(1, Some("cold"), "s3://bucket/cold").writes()), + None + ); + } + + #[test] + fn test_writes_of_unnamed_base_claims_only_path() { + let unnamed = add_base(0, None, "s3://bucket/a").writes(); + assert_eq!( + unnamed.conflicts_with(&add_base(1, None, "s3://bucket/b").writes()), + None + ); + assert_eq!( + unnamed.conflicts_with(&add_base(1, None, "s3://bucket/a").writes()), + Some(super::super::mask::Conflict::Incompatible) + ); + } + + #[test] + fn test_proto_roundtrips() { + for action in [ + add_base(0, Some("warm"), "s3://bucket/warm"), + add_base(3, None, "/local/path"), + AddBase { + local: 1, + name: Some("root".into()), + is_dataset_root: true, + path: "s3://bucket/root".into(), + }, + ] { + let message = pb::AddBase::from(&action); + assert_eq!(AddBase::try_from(message).unwrap(), action); + } + } + + #[test] + fn test_proto_without_base_is_rejected() { + let err = AddBase::try_from(pb::AddBase { + local: 0, + base: None, + }) + .unwrap_err(); + assert!(matches!(err, Error::InvalidInput { .. })); + assert!(err.to_string().contains("did not contain a base path")); + } +} diff --git a/rust/lance-table/src/transaction/action/apply.rs b/rust/lance-table/src/transaction/action/apply.rs new file mode 100644 index 00000000000..1ca7ee84470 --- /dev/null +++ b/rust/lance-table/src/transaction/action/apply.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Assembling the next manifest from an operation's actions. +//! +//! This is the action vocabulary's counterpart to [`manifest_build`]. It stays +//! separate rather than threading actions through that function's two match +//! phases: the legacy assembly is a frozen compatibility surface, and an action +//! applies to the manifest directly instead of being folded into a post-image. +//! +//! [`manifest_build`]: crate::transaction::manifest_build + +use super::UserOperation; +use super::refs::TxnContext; +use crate::feature_flags::apply_feature_flags; +use crate::format::{IndexMetadata, Manifest, ManifestBuildConfig}; +use lance_core::{Error, Result}; + +impl UserOperation { + /// Build the manifest this operation produces from the one it applies to. + /// + /// `tag` is the enclosing transaction's tag. Actions are applied in order, so + /// a later action can resolve a `Local` token minted by an earlier one. + pub(crate) fn build_manifest( + &self, + current_manifest: Option<&Manifest>, + current_indices: Vec, + tag: Option<&str>, + transaction_file_path: &str, + config: &ManifestBuildConfig, + ) -> Result<(Manifest, Vec)> { + let Some(current_manifest) = current_manifest else { + return Err(Error::not_supported( + "an action-based operation cannot create a dataset; \ + it can only be applied to an existing one", + )); + }; + + // An identity baseline: every change to schema, fragments and bases comes + // from an action, so nothing here pre-empts one. + let mut manifest = Manifest::new_from_previous( + current_manifest, + current_manifest.schema.clone(), + current_manifest.fragments.clone(), + ); + + manifest.tag = tag.map(str::to_string); + + if config.auto_set_feature_flags { + // Mirrors the legacy path: the flag is inherited, so a config built + // with the default `use_stable_row_ids = false` does not clear it. + let use_stable_row_ids = + config.use_stable_row_ids || current_manifest.uses_stable_row_ids(); + apply_feature_flags( + &mut manifest, + use_stable_row_ids, + config.disable_transaction_file, + )?; + } + manifest.set_timestamp(config.timestamp_nanos); + manifest.update_max_fragment_id(); + + let mut ctx = TxnContext::default(); + for action in self.actions() { + action.apply(&mut manifest, &mut ctx)?; + } + + manifest.transaction_file = Some(transaction_file_path.to_string()); + + Ok((manifest, current_indices)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::{BasePath, DataStorageFormat}; + use crate::transaction::action::{Action, AddBase, UserAction}; + use crate::transaction::test_support::default_build_config; + use lance_core::datatypes::Schema; + use std::collections::HashMap; + use std::sync::Arc; + + fn operation(actions: Vec) -> UserOperation { + UserOperation { + description: "test".to_string(), + uuid: "u".to_string(), + read_version: 1, + actions: vec![UserAction { + description: "step".to_string(), + actions, + }], + } + } + + fn add_base(local: u32, name: Option<&str>, path: &str) -> Action { + Action::AddBase(AddBase { + local, + name: name.map(str::to_string), + is_dataset_root: false, + path: path.to_string(), + }) + } + + fn current() -> Manifest { + let mut manifest = Manifest::new( + Schema::default(), + Arc::new(Vec::new()), + DataStorageFormat::default(), + HashMap::new(), + ); + manifest.base_paths = HashMap::from_iter([( + 1, + BasePath::new(1, "s3://bucket/a".into(), Some("a".into()), false), + )]); + manifest + } + + #[test] + fn test_build_manifest_applies_actions_and_stamps_scaffolding() { + let config = ManifestBuildConfig { + timestamp_nanos: 42, + ..default_build_config() + }; + let (manifest, indices) = operation(vec![add_base(0, Some("warm"), "s3://bucket/warm")]) + .build_manifest( + Some(¤t()), + vec![], + Some("v7"), + "_txn/abc.txn", + &config, + ) + .unwrap(); + + assert_eq!(manifest.base_paths.len(), 2); + assert_eq!(manifest.base_paths[&2].name.as_deref(), Some("warm")); + // The pre-existing base is carried forward untouched. + assert_eq!(manifest.base_paths[&1].path, "s3://bucket/a"); + assert_eq!(manifest.version, current().version + 1); + assert_eq!(manifest.tag.as_deref(), Some("v7")); + assert_eq!(manifest.timestamp_nanos, 42); + assert_eq!(manifest.transaction_file.as_deref(), Some("_txn/abc.txn")); + assert!(indices.is_empty()); + } + + #[test] + fn test_build_manifest_applies_actions_in_order_across_steps() { + let operation = UserOperation { + description: "test".to_string(), + uuid: "u".to_string(), + read_version: 1, + actions: vec![ + UserAction { + description: "first".to_string(), + actions: vec![add_base(0, Some("warm"), "s3://bucket/warm")], + }, + UserAction { + description: "second".to_string(), + actions: vec![add_base(1, Some("cold"), "s3://bucket/cold")], + }, + ], + }; + let (manifest, _) = operation + .build_manifest( + Some(¤t()), + vec![], + None, + "_txn/abc.txn", + &default_build_config(), + ) + .unwrap(); + + // Ids are minted in application order from the existing maximum. + assert_eq!(manifest.base_paths[&2].name.as_deref(), Some("warm")); + assert_eq!(manifest.base_paths[&3].name.as_deref(), Some("cold")); + } + + #[test] + fn test_build_manifest_preserves_indices() { + let indices = vec![IndexMetadata { + uuid: uuid::Uuid::new_v4(), + name: "idx".to_string(), + fields: vec![0], + dataset_version: 1, + fragment_bitmap: None, + index_details: None, + index_version: 0, + created_at: None, + base_id: None, + files: None, + }]; + let (_, built) = operation(vec![add_base(0, Some("warm"), "s3://bucket/warm")]) + .build_manifest( + Some(¤t()), + indices.clone(), + None, + "_txn/abc.txn", + &default_build_config(), + ) + .unwrap(); + + assert_eq!(built.len(), 1); + assert_eq!(built[0].name, indices[0].name); + } + + #[test] + fn test_build_manifest_rejects_dataset_creation() { + let err = operation(vec![add_base(0, Some("warm"), "s3://bucket/warm")]) + .build_manifest(None, vec![], None, "_txn/abc.txn", &default_build_config()) + .unwrap_err(); + + assert!( + matches!(err, Error::NotSupported { .. }), + "expected NotSupported, got: {err:?}" + ); + } + + #[test] + fn test_build_manifest_propagates_action_failure() { + // A colliding name aborts the whole assembly; no partial manifest escapes. + let err = operation(vec![add_base(0, Some("a"), "s3://bucket/other")]) + .build_manifest( + Some(¤t()), + vec![], + None, + "_txn/abc.txn", + &default_build_config(), + ) + .unwrap_err(); + + assert!(matches!(err, Error::InvalidInput { .. })); + } + + #[test] + fn test_build_manifest_rejects_duplicate_local_tokens_at_apply() { + let err = operation(vec![ + add_base(0, Some("warm"), "s3://bucket/warm"), + add_base(0, Some("cold"), "s3://bucket/cold"), + ]) + .build_manifest( + Some(¤t()), + vec![], + None, + "_txn/abc.txn", + &default_build_config(), + ) + .unwrap_err(); + + assert!(err.to_string().contains("local base token 0")); + } +} diff --git a/rust/lance-table/src/transaction/action/mask.rs b/rust/lance-table/src/transaction/action/mask.rs new file mode 100644 index 00000000000..8af2a9818d0 --- /dev/null +++ b/rust/lance-table/src/transaction/action/mask.rs @@ -0,0 +1,303 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! The manifest regions an operation writes, and whether two footprints collide. +//! +//! Conflict detection in the action vocabulary is a *footprint* comparison rather +//! than a pairwise match: every action and every legacy [`Operation`] declares the +//! regions of the manifest it writes, and two transactions conflict exactly when +//! their footprints intersect. The intersection is computed once, here, so the +//! cost of a new action is one declaration instead of a row and a column in an +//! N-by-N table. +//! +//! [`Operation`]: crate::transaction::Operation + +use std::collections::{BTreeMap, BTreeSet}; + +/// A region of the manifest an operation can write. +/// +/// Deliberately coarse. Over-declaring a region costs a spurious retry, but an +/// *unlisted* region is a silently permitted conflict, so every region is defined +/// here before any action needs it and refined to key granularity per slice. +/// +/// Id counters are not regions and never will be. Every operation that needs an +/// id is expressed as *adding to* the counter, and additions to a monotone +/// counter cannot conflict: each side mints from whatever the counter holds when +/// it applies, so two concurrent minting operations get disjoint ranges by +/// construction. Making a counter a region would manufacture exactly the +/// conflicts the minting design exists to avoid — it is why two `Append`s +/// commute. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum Region { + /// `Manifest::base_paths`. + Bases, + /// The field structure of `Manifest::schema`. + Schema, + /// The metadata maps on `Manifest::schema` and its fields. + SchemaMetadata, + /// `Manifest::fragments`, including their data files and deletion files. + Fragments, + /// The index metadata list committed alongside the manifest. + Indices, + /// `Manifest::config`. + Config, + /// `Manifest::table_metadata`. + TableMetadata, + /// MemWAL bookkeeping: which SSTable generations are compacted into the + /// base table. + Generations, +} + +impl Region { + /// Every region. Used to build the footprint of operations that replace the + /// manifest wholesale. + pub const ALL: [Self; 8] = [ + Self::Bases, + Self::Schema, + Self::SchemaMetadata, + Self::Fragments, + Self::Indices, + Self::Config, + Self::TableMetadata, + Self::Generations, + ]; +} + +/// A named claim inside a region, for regions where an operation can say more +/// than "I touch this". +/// +/// Keys are compared only within one region, so the same string under two +/// regions is two unrelated claims. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)] +pub enum Key { + /// A base path's user-visible name (`BasePath::name`). + Name(String), + /// A base path's full URI (`BasePath::path`). + Path(String), + /// An entry in one of the manifest's string maps. + Entry(String), +} + +/// How much of a region an operation writes. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Scope { + /// The whole region. Intersects any other claim on it. + All, + /// Only these named claims. + Keys(BTreeSet), +} + +/// Whether two intersecting footprints can be resolved by retrying the commit. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Conflict { + /// Re-applying against the newer manifest may succeed. + Retryable, + /// No retry can succeed, because both sides claim the same name. + Incompatible, +} + +/// The set of manifest regions an operation writes. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct ManifestMask { + regions: BTreeMap, +} + +impl ManifestMask { + /// Every region, in full — the footprint of an operation that replaces the + /// manifest wholesale. + pub fn everything() -> Self { + Self { + regions: Region::ALL.iter().map(|r| (*r, Scope::All)).collect(), + } + } + + /// Claim a whole region. + pub fn with_all(mut self, region: Region) -> Self { + self.regions.insert(region, Scope::All); + self + } + + /// Claim named keys within a region, merging with any claim already made on + /// it. A region already claimed in full stays claimed in full. + pub fn with_keys(mut self, region: Region, keys: impl IntoIterator) -> Self { + match self + .regions + .entry(region) + .or_insert_with(|| Scope::Keys(BTreeSet::new())) + { + Scope::All => {} + Scope::Keys(existing) => existing.extend(keys), + } + self + } + + pub fn union(mut self, other: Self) -> Self { + for (region, scope) in other.regions { + match scope { + Scope::All => self = self.with_all(region), + Scope::Keys(keys) => self = self.with_keys(region, keys), + } + } + self + } + + /// Whether this footprint touches any of `regions`, at any granularity. + pub fn intersects_any(&self, regions: &[Region]) -> bool { + regions + .iter() + .any(|region| self.regions.contains_key(region)) + } + + /// The verdict for two footprints, or `None` if they are disjoint. + /// + /// Exact-key overlap is incompatible: key-level claims are exclusive, and no + /// retry frees a base name or a config key that another commit has taken. + /// Any coarser overlap is retryable, because a retry re-runs apply against + /// the newer manifest and apply re-validates. + pub fn conflicts_with(&self, other: &Self) -> Option { + let mut verdict = None; + for (region, scope) in &self.regions { + let Some(other_scope) = other.regions.get(region) else { + continue; + }; + match (scope, other_scope) { + (Scope::Keys(ours), Scope::Keys(theirs)) => { + if !ours.is_disjoint(theirs) { + return Some(Conflict::Incompatible); + } + } + _ => verdict = Some(Conflict::Retryable), + } + } + verdict + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn keys(region: Region, keys: impl IntoIterator) -> ManifestMask { + ManifestMask::default().with_keys(region, keys) + } + + #[test] + fn test_disjoint_regions_do_not_conflict() { + let bases = ManifestMask::default().with_all(Region::Bases); + let fragments = ManifestMask::default().with_all(Region::Fragments); + assert_eq!(bases.conflicts_with(&fragments), None); + assert_eq!(fragments.conflicts_with(&bases), None); + } + + #[test] + fn test_distinct_keys_in_one_region_do_not_conflict() { + let warm = keys(Region::Bases, [Key::Name("warm".into())]); + let cold = keys(Region::Bases, [Key::Name("cold".into())]); + assert_eq!(warm.conflicts_with(&cold), None); + } + + #[test] + fn test_exact_key_overlap_is_incompatible() { + let warm = keys(Region::Bases, [Key::Name("warm".into())]); + assert_eq!( + warm.conflicts_with(&warm.clone()), + Some(Conflict::Incompatible) + ); + } + + #[test] + fn test_key_variants_are_distinct_claims() { + // A base *named* "s3://bucket/a" does not claim the base *at* that path. + let by_name = keys(Region::Bases, [Key::Name("s3://bucket/a".into())]); + let by_path = keys(Region::Bases, [Key::Path("s3://bucket/a".into())]); + assert_eq!(by_name.conflicts_with(&by_path), None); + } + + #[test] + fn test_same_key_string_in_different_regions_does_not_conflict() { + let config = keys(Region::Config, [Key::Entry("k".into())]); + let table_metadata = keys(Region::TableMetadata, [Key::Entry("k".into())]); + assert_eq!(config.conflicts_with(&table_metadata), None); + } + + #[test] + fn test_coarse_overlap_is_retryable() { + let all_bases = ManifestMask::default().with_all(Region::Bases); + let one_base = keys(Region::Bases, [Key::Name("warm".into())]); + assert_eq!( + all_bases.conflicts_with(&one_base), + Some(Conflict::Retryable) + ); + assert_eq!( + one_base.conflicts_with(&all_bases), + Some(Conflict::Retryable) + ); + assert_eq!( + all_bases.conflicts_with(&all_bases.clone()), + Some(Conflict::Retryable) + ); + } + + #[test] + fn test_incompatible_wins_over_retryable() { + // Overlapping keys in one region and coarse overlap in another: the + // stricter verdict must win regardless of region iteration order. + let a = keys(Region::Bases, [Key::Name("warm".into())]).with_all(Region::Fragments); + let b = keys(Region::Bases, [Key::Name("warm".into())]).with_all(Region::Fragments); + assert_eq!(a.conflicts_with(&b), Some(Conflict::Incompatible)); + } + + #[test] + fn test_everything_conflicts_with_any_claim() { + for region in Region::ALL { + let mask = ManifestMask::default().with_all(region); + assert!( + ManifestMask::everything().conflicts_with(&mask).is_some(), + "everything() must cover {region:?}" + ); + } + } + + #[test] + fn test_with_all_subsumes_keys() { + let mask = keys(Region::Bases, [Key::Name("warm".into())]).with_all(Region::Bases); + // Now coarse, so an unrelated name still overlaps — and retryably, not + // as an exact-key claim. + let other = keys(Region::Bases, [Key::Name("cold".into())]); + assert_eq!(mask.conflicts_with(&other), Some(Conflict::Retryable)); + } + + #[test] + fn test_with_keys_does_not_downgrade_all() { + let mask = ManifestMask::default() + .with_all(Region::Bases) + .with_keys(Region::Bases, [Key::Name("warm".into())]); + let other = keys(Region::Bases, [Key::Name("cold".into())]); + assert_eq!(mask.conflicts_with(&other), Some(Conflict::Retryable)); + } + + #[test] + fn test_union_merges_keys_and_regions() { + let a = keys(Region::Bases, [Key::Name("warm".into())]); + let b = keys(Region::Bases, [Key::Name("cold".into())]).with_all(Region::Fragments); + let merged = a.union(b); + + assert_eq!( + merged.conflicts_with(&keys(Region::Bases, [Key::Name("warm".into())])), + Some(Conflict::Incompatible) + ); + assert_eq!( + merged.conflicts_with(&keys(Region::Bases, [Key::Name("cold".into())])), + Some(Conflict::Incompatible) + ); + assert!(merged.intersects_any(&[Region::Fragments])); + } + + #[test] + fn test_intersects_any() { + let mask = keys(Region::Bases, [Key::Name("warm".into())]); + assert!(mask.intersects_any(&[Region::Fragments, Region::Bases])); + assert!(!mask.intersects_any(&[Region::Fragments, Region::Indices])); + assert!(!mask.intersects_any(&[])); + } +} diff --git a/rust/lance-table/src/transaction/action/refs.rs b/rust/lance-table/src/transaction/action/refs.rs new file mode 100644 index 00000000000..1cacf9685a6 --- /dev/null +++ b/rust/lance-table/src/transaction/action/refs.rs @@ -0,0 +1,143 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! References to counter-allocated ids, and the state that resolves them. + +use crate::format::pb; +use lance_core::deepsize::DeepSizeOf; +use lance_core::{Error, Result}; +use std::collections::HashMap; + +/// A reference to a counter-allocated identifier (field id, fragment id, or base +/// id) that may not be committed yet. +/// +/// `Committed` is an already-assigned id. `Local` is a placeholder token minted by +/// an `Add*` action earlier in the same [`UserOperation`], resolved to a freshly +/// allocated id when the operation is applied. +/// +/// Local tokens are scoped to one `UserOperation`, which is what makes minting +/// conflict-free: two concurrent operations each mint from the counter as they +/// apply, so their ids cannot collide however the tokens were numbered. It is +/// also what lets two branches that each add a field be merged — a placeholder +/// shared by several actions names the same not-yet-minted id, which a bare +/// sentinel value could not express. +/// +/// [`UserOperation`]: super::UserOperation +#[derive(Debug, Clone, Copy, PartialEq, Eq, DeepSizeOf)] +pub enum Ref { + Committed(u64), + Local(u32), +} + +impl TryFrom for Ref { + type Error = Error; + + fn try_from(message: pb::Ref) -> Result { + match message.kind { + Some(pb::r#ref::Kind::Committed(id)) => Ok(Self::Committed(id)), + Some(pb::r#ref::Kind::Local(token)) => Ok(Self::Local(token)), + // Fail closed, for the reason given on `Action`'s decode: an + // unrecognized reference kind must abort the commit rather than be + // dropped from conflict detection. + None => Err(Error::invalid_input( + "Ref message did not contain a kind; it was written by a newer version of Lance", + )), + } + } +} + +impl From<&Ref> for pb::Ref { + fn from(reference: &Ref) -> Self { + let kind = match reference { + Ref::Committed(id) => pb::r#ref::Kind::Committed(*id), + Ref::Local(token) => pb::r#ref::Kind::Local(*token), + }; + Self { kind: Some(kind) } + } +} + +/// State accumulated while applying one [`UserOperation`]'s actions in order. +/// +/// A minting action records the id it allocated here, so a later action naming +/// the same `Local` token can resolve it. It grows a field per action family as +/// the vocabulary lands; today only base ids are minted. +/// +/// There is deliberately no resolver yet: nothing in this slice *reads* a base +/// binding, and `AddDataFile` is the first action that will. Recording the +/// binding here is still not optional — minting without it would leave a seam +/// the next slice has to reopen, and it is where the "local tokens must be +/// distinct" rule that `actions.proto` promises is enforced at apply. +/// +/// [`UserOperation`]: super::UserOperation +#[derive(Debug, Default)] +pub struct TxnContext { + bases: HashMap, +} + +impl TxnContext { + /// Record that `local` names the freshly minted base `id`. + pub(crate) fn bind_base(&mut self, local: u32, id: u32) -> Result<()> { + if let Some(existing) = self.bases.insert(local, id) { + return Err(Error::invalid_input(format!( + "local base token {local} is already bound to base id {existing}; \ + local tokens must be distinct within a UserOperation" + ))); + } + Ok(()) + } + + #[cfg(test)] + pub(crate) fn bound_base(&self, local: u32) -> Option { + self.bases.get(&local).copied() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_ref_roundtrips() { + for reference in [Ref::Committed(u64::MAX), Ref::Local(7)] { + let message = pb::Ref::from(&reference); + assert_eq!(Ref::try_from(message).unwrap(), reference); + } + } + + #[test] + fn test_ref_without_kind_is_rejected() { + let err = Ref::try_from(pb::Ref { kind: None }).unwrap_err(); + assert!( + matches!(err, Error::InvalidInput { .. }), + "expected InvalidInput, got: {err:?}" + ); + assert!(err.to_string().contains("did not contain a kind")); + } + + #[test] + fn test_bind_records_the_minted_id_under_its_token() { + let mut ctx = TxnContext::default(); + ctx.bind_base(3, 9).unwrap(); + ctx.bind_base(4, 10).unwrap(); + assert_eq!(ctx.bound_base(3), Some(9)); + assert_eq!(ctx.bound_base(4), Some(10)); + assert_eq!(ctx.bound_base(5), None); + } + + #[test] + fn test_duplicate_local_token_is_rejected() { + let mut ctx = TxnContext::default(); + ctx.bind_base(0, 1).unwrap(); + let err = ctx.bind_base(0, 2).unwrap_err(); + assert!( + matches!(err, Error::InvalidInput { .. }), + "expected InvalidInput, got: {err:?}" + ); + assert!(err.to_string().contains("already bound to base id 1")); + } + + #[test] + fn test_unbound_token_is_absent() { + assert_eq!(TxnContext::default().bound_base(4), None); + } +} diff --git a/rust/lance-table/src/transaction/action/translate.rs b/rust/lance-table/src/transaction/action/translate.rs new file mode 100644 index 00000000000..8b2c1521232 --- /dev/null +++ b/rust/lance-table/src/transaction/action/translate.rs @@ -0,0 +1,208 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Expressing a legacy [`Operation`] as actions. +//! +//! This is a parity-testing device and the eventual write-path bridge: it lets a +//! legacy operation and its action decomposition be applied to the same manifest +//! and compared, which is what proves the action path is not quietly diverging. +//! +//! It is deliberately *not* used on the conflict path. Reducing everything to +//! action-versus-action there would need a translation for every operation (this +//! translates one), some operations cannot translate without the prior manifest at +//! their own read version, and it would mean allocating an action list for every +//! concurrent transaction just to learn which regions it touched. Footprints +//! ([`ManifestMask`]) answer that question directly. +//! +//! [`ManifestMask`]: super::mask::ManifestMask + +use super::{Action, AddBase}; +use crate::transaction::Operation; +use lance_core::{Error, Result}; + +impl TryFrom<&Operation> for Vec { + type Error = Error; + + fn try_from(operation: &Operation) -> Result { + match operation { + Operation::UpdateBases { new_bases } => new_bases + .iter() + .enumerate() + .map(|(index, base)| { + if base.id != 0 { + // The caller pre-assigned an id, which the action model does + // not express: ids are minted at apply. Rejecting is the only + // faithful answer — translating would silently drop the id. + return Err(Error::invalid_input(format!( + "cannot express UpdateBases entry with pre-assigned base id {} \ + as an AddBase action; base ids are minted when the action is applied", + base.id + ))); + } + let local = u32::try_from(index).map_err(|_| { + Error::invalid_input(format!( + "UpdateBases carries {} bases, more than a local token can address", + new_bases.len() + )) + })?; + Ok(Action::AddBase(AddBase { + local, + name: base.name.clone(), + is_dataset_root: base.is_dataset_root, + path: base.path.clone(), + })) + }) + .collect(), + other => Err(Error::not_supported(format!( + "expressing the {} operation as actions is not supported yet", + other.name() + ))), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::{BasePath, DataStorageFormat, Manifest}; + use crate::transaction::action::{UserAction, UserOperation}; + use crate::transaction::test_support::default_build_config; + use crate::transaction::{Transaction, TransactionBuilder}; + use lance_core::datatypes::Schema; + use std::collections::HashMap; + use std::sync::Arc; + + const TXN_PATH: &str = "_transactions/parity.txn"; + + fn current() -> Manifest { + let mut manifest = Manifest::new( + Schema::default(), + Arc::new(Vec::new()), + DataStorageFormat::default(), + HashMap::new(), + ); + manifest.base_paths = HashMap::from_iter([ + ( + 1, + BasePath::new(1, "s3://bucket/a".into(), Some("a".into()), false), + ), + ( + 4, + BasePath::new(4, "s3://bucket/b".into(), Some("b".into()), false), + ), + ]); + manifest + } + + fn unassigned(name: Option<&str>, path: &str) -> BasePath { + BasePath::new(0, path.to_string(), name.map(str::to_string), false) + } + + #[test] + fn test_update_bases_translates_to_one_add_base_per_entry() { + let operation = Operation::UpdateBases { + new_bases: vec![ + unassigned(Some("warm"), "s3://bucket/warm"), + unassigned(None, "/local/cold"), + ], + }; + let actions = Vec::::try_from(&operation).unwrap(); + + assert_eq!( + actions, + vec![ + Action::AddBase(AddBase { + local: 0, + name: Some("warm".to_string()), + is_dataset_root: false, + path: "s3://bucket/warm".to_string(), + }), + Action::AddBase(AddBase { + local: 1, + name: None, + is_dataset_root: false, + path: "/local/cold".to_string(), + }), + ] + ); + } + + #[test] + fn test_preassigned_base_id_is_rejected() { + let operation = Operation::UpdateBases { + new_bases: vec![BasePath::new(9, "s3://bucket/x".into(), None, false)], + }; + let err = Vec::::try_from(&operation).unwrap_err(); + + assert!( + matches!(err, Error::InvalidInput { .. }), + "expected InvalidInput, got: {err:?}" + ); + assert!(err.to_string().contains("pre-assigned base id 9")); + } + + #[test] + fn test_other_operations_are_not_supported_yet() { + let operation = Operation::ReserveFragments { num_fragments: 1 }; + let err = Vec::::try_from(&operation).unwrap_err(); + + assert!( + matches!(err, Error::NotSupported { .. }), + "expected NotSupported, got: {err:?}" + ); + assert!(err.to_string().contains("ReserveFragments")); + } + + /// The most valuable test in the slice: a legacy `UpdateBases` and its action + /// decomposition must produce the same manifest. `transaction_file` and + /// `timestamp_nanos` are excluded because they differ by construction. + #[test] + fn test_round_trip_parity_with_legacy_update_bases() { + let current = current(); + let config = default_build_config(); + + let legacy_operation = Operation::UpdateBases { + new_bases: vec![ + unassigned(Some("warm"), "s3://bucket/warm"), + unassigned(None, "/local/cold"), + ], + }; + let actions = Vec::::try_from(&legacy_operation).unwrap(); + + let legacy_transaction: Transaction = + TransactionBuilder::new(current.version, legacy_operation).build(); + let (legacy, legacy_indices) = legacy_transaction + .build_manifest(Some(¤t), vec![], TXN_PATH, &config) + .unwrap(); + + let user_operation = UserOperation { + description: "ALTER TABLE t ADD BASE".to_string(), + uuid: "0197f6bd-0000-4000-8000-000000000000".to_string(), + read_version: current.version, + actions: vec![UserAction { + description: "register bases".to_string(), + actions, + }], + }; + let (v2, v2_indices) = user_operation + .build_manifest(Some(¤t), vec![], None, TXN_PATH, &config) + .unwrap(); + + assert_eq!(v2.base_paths, legacy.base_paths); + assert_eq!(v2.schema, legacy.schema); + assert_eq!(v2.fragments, legacy.fragments); + assert_eq!(v2.version, legacy.version); + assert_eq!(v2.next_row_id, legacy.next_row_id); + assert_eq!(v2.max_fragment_id, legacy.max_fragment_id); + assert_eq!(v2.reader_feature_flags, legacy.reader_feature_flags); + assert_eq!(v2.writer_feature_flags, legacy.writer_feature_flags); + assert_eq!(v2.config, legacy.config); + assert_eq!(v2.table_metadata, legacy.table_metadata); + assert_eq!(v2.data_storage_format, legacy.data_storage_format); + assert_eq!(v2_indices.len(), legacy_indices.len()); + + // The ids the two paths mint must agree, not merely the count. + assert_eq!(v2.base_paths[&5].name.as_deref(), Some("warm")); + assert_eq!(v2.base_paths[&6].path, "/local/cold"); + } +} diff --git a/rust/lance-table/src/transaction/builder.rs b/rust/lance-table/src/transaction/builder.rs new file mode 100644 index 00000000000..1240de4d882 --- /dev/null +++ b/rust/lance-table/src/transaction/builder.rs @@ -0,0 +1,90 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! The transaction itself: an operation plus the version it was based on. + +use crate::transaction::Operation; +use lance_core::deepsize::DeepSizeOf; +use std::collections::HashMap; +use std::sync::Arc; +use uuid::Uuid; + +/// A change to a dataset that can be retried +/// +/// This contains enough information to be able to build the next manifest, +/// given the current manifest. +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub struct Transaction { + /// The version of the table this transaction is based off of. If this is + /// the first transaction, this should be 0. + pub read_version: u64, + pub uuid: String, + pub operation: Operation, + pub tag: Option, + pub transaction_properties: Option>>, +} + +/// Add TransactionBuilder for flexibly setting option without using `mut` +pub struct TransactionBuilder { + read_version: u64, + // uuid is optional for builder since it can autogenerate + uuid: Option, + operation: Operation, + tag: Option, + transaction_properties: Option>>, +} + +impl TransactionBuilder { + pub fn new(read_version: u64, operation: Operation) -> Self { + Self { + read_version, + uuid: None, + operation, + tag: None, + transaction_properties: None, + } + } + + pub fn uuid(mut self, uuid: String) -> Self { + self.uuid = Some(uuid); + self + } + + pub fn tag(mut self, tag: Option) -> Self { + self.tag = tag; + self + } + + pub fn transaction_properties( + mut self, + transaction_properties: Option>>, + ) -> Self { + self.transaction_properties = transaction_properties; + self + } + + pub fn build(self) -> Transaction { + let uuid = self + .uuid + .unwrap_or_else(|| Uuid::new_v4().hyphenated().to_string()); + Transaction { + read_version: self.read_version, + uuid, + operation: self.operation, + tag: self.tag, + transaction_properties: self.transaction_properties, + } + } +} + +impl Transaction { + pub fn new_from_version(read_version: u64, operation: Operation) -> Self { + TransactionBuilder::new(read_version, operation).build() + } + + pub fn new(read_version: u64, operation: Operation, tag: Option) -> Self { + TransactionBuilder::new(read_version, operation) + .tag(tag) + .build() + } +} diff --git a/rust/lance-table/src/transaction/conflicts.rs b/rust/lance-table/src/transaction/conflicts.rs new file mode 100644 index 00000000000..643bafe5618 --- /dev/null +++ b/rust/lance-table/src/transaction/conflicts.rs @@ -0,0 +1,1220 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Deciding whether two operations describe the same change or touch the same +//! metadata. +//! +//! The commit path uses these when it retries against a newer version: equality +//! tells it whether the operation it is holding is the one already committed, and +//! the metadata checks tell it whether a concurrent operation wrote keys it +//! depends on. +//! +//! `PartialEq` is hand-written rather than derived because several operations +//! carry `Vec` fields whose order is not meaningful. + +use crate::transaction::action::mask::{Key, ManifestMask, Region}; +use crate::transaction::{Operation, UpdateMap}; +use std::collections::HashSet; + +/// Fold an [`UpdateMap`]'s footprint into `mask`. A replacing update claims the +/// whole region; an incremental one claims only the keys it edits. +fn map_writes(mask: ManifestMask, region: Region, updates: Option<&UpdateMap>) -> ManifestMask { + let Some(updates) = updates else { + return mask; + }; + if updates.replace { + return mask.with_all(region); + } + mask.with_keys( + region, + updates + .update_entries + .iter() + .map(|entry| Key::Entry(entry.key.clone())), + ) +} + +impl PartialEq for Operation { + fn eq(&self, other: &Self) -> bool { + // Many of the operations contain `Vec` where the order of the + // elements don't matter. So we need to compare them in a way that + // ignores the order of the elements. + // TODO: we can make it so the vecs are always constructed in order. + // Then we can use `==` instead of `compare_vec`. + fn compare_vec(a: &[T], b: &[T]) -> bool { + a.len() == b.len() && a.iter().all(|f| b.contains(f)) + } + match (self, other) { + (Self::Append { fragments: a }, Self::Append { fragments: b }) => compare_vec(a, b), + ( + Self::Clone { + is_shallow: a_is_shallow, + ref_name: a_ref_name, + ref_version: a_ref_version, + ref_path: a_source_path, + branch_name: a_branch_name, + }, + Self::Clone { + is_shallow: b_is_shallow, + ref_name: b_ref_name, + ref_version: b_ref_version, + ref_path: b_source_path, + branch_name: b_branch_name, + }, + ) => { + a_is_shallow == b_is_shallow + && a_ref_name == b_ref_name + && a_ref_version == b_ref_version + && a_source_path == b_source_path + && a_branch_name == b_branch_name + } + ( + Self::Delete { + updated_fragments: a_updated, + deleted_fragment_ids: a_deleted, + predicate: a_predicate, + }, + Self::Delete { + updated_fragments: b_updated, + deleted_fragment_ids: b_deleted, + predicate: b_predicate, + }, + ) => { + compare_vec(a_updated, b_updated) + && compare_vec(a_deleted, b_deleted) + && a_predicate == b_predicate + } + ( + Self::Overwrite { + fragments: a_fragments, + schema: a_schema, + config_upsert_values: a_config, + initial_bases: a_initial, + }, + Self::Overwrite { + fragments: b_fragments, + schema: b_schema, + config_upsert_values: b_config, + initial_bases: b_initial, + }, + ) => { + compare_vec(a_fragments, b_fragments) + && a_schema == b_schema + && a_config == b_config + && a_initial == b_initial + } + ( + Self::CreateIndex { + new_indices: a_new, + removed_indices: a_removed, + }, + Self::CreateIndex { + new_indices: b_new, + removed_indices: b_removed, + }, + ) => compare_vec(a_new, b_new) && compare_vec(a_removed, b_removed), + ( + Self::Rewrite { + groups: a_groups, + rewritten_indices: a_indices, + frag_reuse_index: a_frag_reuse_index, + }, + Self::Rewrite { + groups: b_groups, + rewritten_indices: b_indices, + frag_reuse_index: b_frag_reuse_index, + }, + ) => { + compare_vec(a_groups, b_groups) + && compare_vec(a_indices, b_indices) + && a_frag_reuse_index == b_frag_reuse_index + } + ( + Self::Merge { + fragments: a_fragments, + schema: a_schema, + }, + Self::Merge { + fragments: b_fragments, + schema: b_schema, + }, + ) => compare_vec(a_fragments, b_fragments) && a_schema == b_schema, + (Self::Restore { version: a }, Self::Restore { version: b }) => a == b, + ( + Self::ReserveFragments { num_fragments: a }, + Self::ReserveFragments { num_fragments: b }, + ) => a == b, + ( + Self::Update { + removed_fragment_ids: a_removed, + updated_fragments: a_updated, + new_fragments: a_new, + fields_modified: a_fields, + compacted_sstables: a_compacted_sstables, + fields_for_preserving_frag_bitmap: a_fields_for_preserving_frag_bitmap, + update_mode: a_update_mode, + inserted_rows_filter: a_inserted_rows_filter, + updated_fragment_offsets: a_updated_fragment_offsets, + }, + Self::Update { + removed_fragment_ids: b_removed, + updated_fragments: b_updated, + new_fragments: b_new, + fields_modified: b_fields, + compacted_sstables: b_compacted_sstables, + fields_for_preserving_frag_bitmap: b_fields_for_preserving_frag_bitmap, + update_mode: b_update_mode, + inserted_rows_filter: b_inserted_rows_filter, + updated_fragment_offsets: b_updated_fragment_offsets, + }, + ) => { + compare_vec(a_removed, b_removed) + && compare_vec(a_updated, b_updated) + && compare_vec(a_new, b_new) + && compare_vec(a_fields, b_fields) + && compare_vec(a_compacted_sstables, b_compacted_sstables) + && compare_vec( + a_fields_for_preserving_frag_bitmap, + b_fields_for_preserving_frag_bitmap, + ) + && a_update_mode == b_update_mode + && a_inserted_rows_filter == b_inserted_rows_filter + && a_updated_fragment_offsets == b_updated_fragment_offsets + } + (Self::Project { schema: a }, Self::Project { schema: b }) => a == b, + ( + Self::UpdateConfig { + config_updates: a_config, + table_metadata_updates: a_table_metadata, + schema_metadata_updates: a_schema, + field_metadata_updates: a_field, + }, + Self::UpdateConfig { + config_updates: b_config, + table_metadata_updates: b_table_metadata, + schema_metadata_updates: b_schema, + field_metadata_updates: b_field, + }, + ) => { + a_config == b_config + && a_table_metadata == b_table_metadata + && a_schema == b_schema + && a_field == b_field + } + ( + Self::DataReplacement { replacements: a }, + Self::DataReplacement { replacements: b }, + ) => a.len() == b.len() && a.iter().all(|r| b.contains(r)), + // Handle all remaining combinations. + // We spell out all combinations explicitly to prevent + // us accidentally handling a new case in the wrong way. + (Self::Append { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Append { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Delete { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Overwrite { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::CreateIndex { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Rewrite { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Merge { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Restore { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::ReserveFragments { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Update { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Project { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::UpdateConfig { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::DataReplacement { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::UpdateMemWalState { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + ( + Self::UpdateMemWalState { + compacted_sstables: a_compacted, + }, + Self::UpdateMemWalState { + compacted_sstables: b_compacted, + }, + ) => compare_vec(a_compacted, b_compacted), + (Self::Clone { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::UpdateBases { new_bases: a }, Self::UpdateBases { new_bases: b }) => { + compare_vec(a, b) + } + + (Self::UpdateBases { .. }, Self::Append { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Delete { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Overwrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::CreateIndex { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Rewrite { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Merge { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Restore { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::ReserveFragments { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Update { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Project { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::UpdateConfig { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::DataReplacement { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::UpdateMemWalState { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateBases { .. }, Self::Clone { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + + (Self::Append { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Delete { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Overwrite { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::CreateIndex { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Rewrite { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Merge { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Restore { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::ReserveFragments { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Update { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Project { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateConfig { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataReplacement { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::UpdateMemWalState { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::Clone { .. }, Self::UpdateBases { .. }) => { + std::mem::discriminant(self) == std::mem::discriminant(other) + } + (Self::DataOverlay { groups: a }, Self::DataOverlay { groups: b }) => compare_vec(a, b), + (Self::DataOverlay { .. }, _) | (_, Self::DataOverlay { .. }) => false, + (Self::UserOperation(a), Self::UserOperation(b)) => a == b, + (Self::UserOperation(..), _) | (_, Self::UserOperation(..)) => false, + } + } +} + +impl Operation { + /// The manifest regions this operation writes. + /// + /// This restates as a footprint what the `check_*_txn` matrix in + /// `conflict_resolver` decides pairwise, and it is what lets an action-based + /// operation be checked against a legacy one without translating every legacy + /// operation into actions: both sides declare where they write, and conflict is + /// intersection. `test_writes_agrees_with_conflict_oracle` asserts it never + /// *under*-declares relative to that matrix, which is the unsafe direction — + /// over-declaring only costs a retry. + pub fn writes(&self) -> ManifestMask { + let mask = ManifestMask::default; + match self { + Self::UpdateBases { new_bases } => mask().with_keys( + Region::Bases, + new_bases.iter().flat_map(|base| { + base.name + .clone() + .map(Key::Name) + .into_iter() + .chain([Key::Path(base.path.clone())]) + }), + ), + // Overwrite replaces the manifest, Restore replaces it with an older + // one without merging anything forward, and Clone resets the table to + // another dataset's state. + Self::Overwrite { .. } | Self::Restore { .. } | Self::Clone { .. } => { + ManifestMask::everything() + } + Self::UpdateConfig { + config_updates, + table_metadata_updates, + schema_metadata_updates, + field_metadata_updates, + } => { + let mut mask = mask(); + mask = map_writes(mask, Region::Config, config_updates.as_ref()); + mask = map_writes(mask, Region::TableMetadata, table_metadata_updates.as_ref()); + if schema_metadata_updates.is_some() { + // Any two schema-metadata updates conflict, matching + // `modifies_same_metadata`, so this is not refined to keys. + mask = mask.with_all(Region::SchemaMetadata); + } + // Field metadata shares the schema-metadata region, keyed by field + // id: two operations touching different fields commute. + mask.with_keys( + Region::SchemaMetadata, + field_metadata_updates + .keys() + .map(|field_id| Key::Entry(field_id.to_string())), + ) + } + Self::Append { .. } | Self::ReserveFragments { .. } => { + // ReserveFragments depends on the fragment *list*, not on the id + // counter: Overwrite and Restore conflict with it because they reset + // the fragment space and so invalidate the reservation. + mask().with_all(Region::Fragments) + } + Self::Delete { .. } + | Self::Update { .. } + | Self::Rewrite { .. } + | Self::DataReplacement { .. } + | Self::DataOverlay { .. } => { + mask().with_all(Region::Fragments).with_all(Region::Indices) + } + Self::Project { .. } | Self::Merge { .. } => mask() + .with_all(Region::Schema) + .with_all(Region::Fragments) + .with_all(Region::Indices), + Self::CreateIndex { .. } => mask().with_all(Region::Indices), + Self::UpdateMemWalState { .. } => mask() + .with_all(Region::Generations) + .with_all(Region::Indices), + Self::UserOperation(operation) => operation.writes(), + } + } + + /// Returns the config keys that have been upserted by this operation. + fn get_upsert_config_keys(&self) -> Vec { + match self { + Self::Overwrite { + config_upsert_values: Some(upsert_values), + .. + } => { + let vec: Vec = upsert_values.keys().cloned().collect(); + vec + } + Self::UpdateConfig { + config_updates: Some(config_updates), + .. + } => config_updates + .update_entries + .iter() + .filter_map(|entry| { + if entry.value.is_some() { + Some(entry.key.clone()) + } else { + None + } + }) + .collect(), + _ => Vec::::new(), + } + } + + /// Returns the config keys that have been deleted by this operation. + fn get_delete_config_keys(&self) -> Vec { + match self { + Self::UpdateConfig { + config_updates: Some(config_updates), + .. + } => config_updates + .update_entries + .iter() + .filter_map(|entry| { + if entry.value.is_none() { + Some(entry.key.clone()) + } else { + None + } + }) + .collect(), + _ => Vec::::new(), + } + } + + pub fn modifies_same_metadata(&self, other: &Self) -> bool { + match (self, other) { + ( + Self::UpdateConfig { + table_metadata_updates, + schema_metadata_updates, + field_metadata_updates, + .. + }, + Self::UpdateConfig { + table_metadata_updates: other_table_metadata, + schema_metadata_updates: other_schema_metadata, + field_metadata_updates: other_field_metadata, + .. + }, + ) => { + if Self::update_maps_conflict( + table_metadata_updates.as_ref(), + other_table_metadata.as_ref(), + ) { + return true; + } + if schema_metadata_updates.is_some() && other_schema_metadata.is_some() { + return true; + } + if !field_metadata_updates.is_empty() && !other_field_metadata.is_empty() { + for field in field_metadata_updates.keys() { + if other_field_metadata.contains_key(field) { + return true; + } + } + } + false + } + _ => false, + } + } + + fn update_maps_conflict(left: Option<&UpdateMap>, right: Option<&UpdateMap>) -> bool { + let (Some(left), Some(right)) = (left, right) else { + return false; + }; + if left.replace || right.replace { + return true; + } + let left_keys = left + .update_entries + .iter() + .map(|entry| entry.key.as_str()) + .collect::>(); + right + .update_entries + .iter() + .any(|entry| left_keys.contains(entry.key.as_str())) + } + + /// Check whether another operation upserts a key that is referenced by another operation + pub fn upsert_key_conflict(&self, other: &Self) -> bool { + let self_upsert_keys = self.get_upsert_config_keys(); + let other_upsert_keys = other.get_upsert_config_keys(); + + let self_delete_keys = self.get_delete_config_keys(); + let other_delete_keys = other.get_delete_config_keys(); + + self_upsert_keys + .iter() + .any(|x| other_upsert_keys.contains(x) || other_delete_keys.contains(x)) + || other_upsert_keys + .iter() + .any(|x| self_upsert_keys.contains(x) || self_delete_keys.contains(x)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::BasePath; + use crate::transaction::test_support::overlay_with_field; + use crate::transaction::{ + Action, AddBase, Conflict, DataOverlayGroup, UpdateMapEntry, UserAction, UserOperation, + }; + use std::collections::HashMap; + + fn table_metadata_update(entries: Vec<(&str, Option<&str>)>, replace: bool) -> Operation { + Operation::UpdateConfig { + config_updates: None, + table_metadata_updates: Some(UpdateMap { + update_entries: entries.into_iter().map(UpdateMapEntry::from).collect(), + replace, + }), + schema_metadata_updates: None, + field_metadata_updates: HashMap::new(), + } + } + + #[test] + fn test_table_metadata_conflicts_on_same_key() { + let left = table_metadata_update(vec![("key", Some("1"))], false); + let same_key = table_metadata_update(vec![("key", Some("2"))], false); + let different_key = table_metadata_update(vec![("other", Some("2"))], false); + let replace = table_metadata_update(vec![("other", Some("2"))], true); + + assert!(left.modifies_same_metadata(&same_key)); + assert!(!left.modifies_same_metadata(&different_key)); + assert!(left.modifies_same_metadata(&replace)); + } + + #[test] + fn test_data_overlay_operation_eq() { + let overlay = |field: i32| Operation::DataOverlay { + groups: vec![DataOverlayGroup { + fragment_id: 0, + overlays: vec![overlay_with_field(field, 1)], + }], + }; + // Reflexive and value-based (the arm previously returned false for self). + assert_eq!(overlay(1), overlay(1)); + assert_ne!(overlay(1), overlay(2)); + // Not equal to a different operation kind (previously returned true vs Rewrite). + let rewrite = Operation::Rewrite { + groups: vec![], + rewritten_indices: vec![], + frag_reuse_index: None, + }; + assert_ne!(overlay(1), rewrite); + } + + fn user_operation(name: &str) -> Operation { + Operation::UserOperation(UserOperation { + description: "ALTER TABLE t ADD BASE".to_string(), + uuid: "u".to_string(), + read_version: 1, + actions: vec![UserAction { + description: "register base".to_string(), + actions: vec![Action::AddBase(AddBase { + local: 0, + name: Some(name.to_string()), + is_dataset_root: false, + path: format!("s3://bucket/{name}"), + })], + }], + }) + } + + #[test] + fn test_user_operation_equality() { + assert_eq!(user_operation("warm"), user_operation("warm")); + assert_ne!(user_operation("warm"), user_operation("cold")); + + for other in [ + Operation::Append { fragments: vec![] }, + Operation::Restore { version: 1 }, + Operation::ReserveFragments { num_fragments: 1 }, + Operation::UpdateBases { new_bases: vec![] }, + ] { + assert_ne!(user_operation("warm"), other); + assert_ne!(other, user_operation("warm")); + } + } + + #[test] + fn test_user_operation_writes_is_the_union_of_its_actions() { + let mask = user_operation("warm").writes(); + assert_eq!( + mask.conflicts_with(&user_operation("warm").writes()), + Some(Conflict::Incompatible) + ); + assert_eq!(mask.conflicts_with(&user_operation("cold").writes()), None); + // A base-only operation commutes with an append. + assert_eq!( + mask.conflicts_with(&Operation::Append { fragments: vec![] }.writes()), + None + ); + } + + #[test] + fn test_update_bases_writes_claims_name_and_path() { + let warm = Operation::UpdateBases { + new_bases: vec![BasePath::new( + 0, + "s3://bucket/warm".into(), + Some("warm".into()), + false, + )], + }; + // The legacy operation and its action form declare the same footprint, + // which is what lets the two vocabularies be compared at all. + assert_eq!(warm.writes(), user_operation("warm").writes()); + } + + #[test] + fn test_restore_writes_everything() { + // Restore replaces the manifest wholesale without merging base_paths + // forward, so it must conflict with a concurrent base addition. + assert_eq!( + Operation::Restore { version: 1 } + .writes() + .conflicts_with(&user_operation("warm").writes()), + Some(Conflict::Retryable) + ); + } + + #[test] + fn test_update_config_writes_is_key_level() { + let a = table_metadata_update(vec![("a", Some("1"))], false); + let b = table_metadata_update(vec![("b", Some("1"))], false); + assert_eq!(a.writes().conflicts_with(&b.writes()), None); + assert_eq!( + a.writes().conflicts_with(&a.writes()), + Some(Conflict::Incompatible) + ); + + // A replacing update claims the whole region, so it collides with any key. + let replace = table_metadata_update(vec![("c", Some("1"))], true); + assert_eq!( + replace.writes().conflicts_with(&a.writes()), + Some(Conflict::Retryable) + ); + } +} diff --git a/rust/lance-table/src/transaction/index_maintenance.rs b/rust/lance-table/src/transaction/index_maintenance.rs new file mode 100644 index 00000000000..df499e4e20d --- /dev/null +++ b/rust/lance-table/src/transaction/index_maintenance.rs @@ -0,0 +1,1027 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Keeping index metadata honest about what the new fragment list contains. +//! +//! An index entry claims coverage of a set of fragments and fields. Any operation +//! that rewrites data can invalidate part of that claim, so a commit has to either +//! narrow the entry's fragment bitmap, drop the fields it no longer describes, or +//! drop the index. Getting this wrong does not fail the commit -- it silently +//! returns stale rows from the index -- so each rule here is paired with a test. + +use crate::format::overlay::staleness::collect_overlay_stale_frags; +use crate::format::{Fragment, IndexMetadata}; +use crate::system_index::frag_reuse::FRAG_REUSE_INDEX_NAME; +use crate::system_index::is_system_index; +use crate::transaction::{RewriteGroup, RewrittenIndex, Transaction}; +use lance_core::datatypes::Schema; +use lance_core::{Error, Result}; +use roaring::RoaringBitmap; +use std::collections::{HashMap, HashSet}; + +impl Transaction { + pub(super) fn register_pure_rewrite_rows_update_frags_in_indices( + indices: &mut [IndexMetadata], + pure_update_frag_ids: &[u64], + original_fragment_ids: &[u64], + fields_for_preserving_frag_bitmap: &[u32], + original_overlaid_frags: &HashMap, + schema: &Schema, + ) -> Result<()> { + if pure_update_frag_ids.is_empty() { + return Ok(()); + } + + let value_updated_field_set = fields_for_preserving_frag_bitmap + .iter() + .collect::>(); + + for index in indices.iter_mut() { + let index_covers_modified_field = index.fields.iter().any(|field_id| { + value_updated_field_set.contains(&u32::try_from(*field_id).unwrap()) + }); + if index_covers_modified_field { + continue; + } + let Some(fragment_bitmap) = index.fragment_bitmap.as_ref() else { + continue; + }; + + // Check that all the original fragments containing the updated rows are covered by + // the index. If not, some updated rows were not indexed, so we cannot index them. + let index_covers_all_original_fragments = original_fragment_ids + .iter() + .all(|&fragment_id| fragment_bitmap.contains(fragment_id as u32)); + if !index_covers_all_original_fragments { + continue; + } + + // A rewrite materializes overlays. If any of those overlays touched the + // column being indexed then the rewrite will modify that column. As a + // result, that index will no longer cover the fragment and it does not + // count as a pure rewrite and we must exclude it from the index's fragment + // bitmap. + let mut overlay_stale = RoaringBitmap::new(); + collect_overlay_stale_frags( + index, + original_overlaid_frags, + &mut overlay_stale, + schema, + )?; + if !overlay_stale.is_empty() { + continue; + } + + if let Some(fragment_bitmap) = index.fragment_bitmap.as_mut() { + for fragment_id in pure_update_frag_ids.iter().map(|f| *f as u32) { + fragment_bitmap.insert(fragment_id); + } + } + } + Ok(()) + } + + /// If an operation modifies one or more fields in a fragment then we need to remove + /// that fragment from any indices that cover one of the modified fields. + pub(super) fn prune_updated_fields_from_indices( + indices: &mut [IndexMetadata], + updated_fragments: &[Fragment], + fields_modified: &[u32], + ) { + if fields_modified.is_empty() { + return; + } + + // If we modified any fields in the fragments then we need to remove those fragments + // from the index if the index covers one of those modified fields. + let fields_modified_set = fields_modified.iter().collect::>(); + for index in indices.iter_mut() { + if index + .fields + .iter() + .any(|field_id| fields_modified_set.contains(&u32::try_from(*field_id).unwrap())) + && let Some(fragment_bitmap) = &mut index.fragment_bitmap + { + for fragment_id in updated_fragments.iter().map(|f| f.id as u32) { + fragment_bitmap.remove(fragment_id); + } + } + } + } + + /// Map each (non-tombstoned) field id in a fragment to the path of the data + /// file that backs it. + fn fragment_field_paths(frag: &Fragment) -> HashMap { + let mut map = HashMap::new(); + for file in &frag.files { + for &field_id in file.fields.iter() { + if field_id >= 0 { + map.insert(field_id, file.path.as_str()); + } + } + } + map + } + + /// A `Merge` can rewrite a column's data *in place* -- the field stays in the + /// schema but its backing data file changes (the overlay fragment carries a new + /// file for the field and tombstones its old field id). `retain_relevant_indices` + /// only drops indices for *removed* fields, so without this the index keeps + /// covering the rewritten fragments with stale entries. Remove each such fragment + /// from any index covering a field whose backing data file changed. + pub(super) fn prune_merge_rewritten_fields_from_indices( + indices: &mut [IndexMetadata], + prev_fragments: &[Fragment], + new_fragments: &[Fragment], + ) { + let prev_by_id: HashMap = + prev_fragments.iter().map(|f| (f.id, f)).collect(); + for new_frag in new_fragments { + let Some(prev) = prev_by_id.get(&new_frag.id) else { + continue; // brand-new fragment: nothing stale to prune + }; + let prev_paths = Self::fragment_field_paths(prev); + let new_paths = Self::fragment_field_paths(new_frag); + // Fields still present whose backing file path changed == rewritten data. + let changed: Vec = prev_paths + .iter() + .filter(|(field_id, prev_path)| { + new_paths + .get(*field_id) + .is_some_and(|new_path| new_path != *prev_path) + }) + .map(|(field_id, _)| *field_id as u32) + .collect(); + if changed.is_empty() { + continue; + } + Self::prune_updated_fields_from_indices( + indices, + std::slice::from_ref(new_frag), + &changed, + ); + } + } + + /// After a `Rewrite` fully compacts a fragment, its data overlays are baked + /// into the new fragment's base data. An index built *before* one of those + /// overlays (`overlay.committed_version > index.dataset_version`) indexed the + /// stale pre-overlay values -- and unlike a live overlay, the compacted + /// fragment no longer signals that staleness to the query path. Drop each + /// rewritten (new) fragment from the coverage of any index covering a field + /// such an overlay supplied, so those rows fall back to a flat scan. + pub(super) fn prune_overlay_stale_fields_from_indices( + indices: &mut [IndexMetadata], + groups: &[RewriteGroup], + ) { + for group in groups { + // field id -> newest overlay committed_version supplying that field + let mut overlaid_field_versions: HashMap = HashMap::new(); + for old_frag in &group.old_fragments { + for overlay in &old_frag.overlays { + for &field_id in overlay.data_file.fields.iter() { + if field_id < 0 { + // Tombstoned (obsolete) overlay field: supplies nothing. + continue; + } + let entry = overlaid_field_versions.entry(field_id).or_insert(0); + *entry = (*entry).max(overlay.committed_version); + } + } + } + if overlaid_field_versions.is_empty() { + continue; + } + + let new_fragment_ids = group + .new_fragments + .iter() + .map(|f| f.id as u32) + .collect::>(); + for index in indices.iter_mut() { + let is_stale = index.fields.iter().any(|field_id| { + overlaid_field_versions + .get(field_id) + .is_some_and(|&overlay_version| overlay_version > index.dataset_version) + }); + if is_stale && let Some(fragment_bitmap) = &mut index.fragment_bitmap { + for new_id in &new_fragment_ids { + fragment_bitmap.remove(*new_id); + } + } + } + } + } + + fn is_vector_index(index: &IndexMetadata) -> bool { + if let Some(details) = &index.index_details { + details.type_url.ends_with("VectorIndexDetails") + } else { + false + } + } + + pub(super) fn retain_relevant_indices( + indices: &mut Vec, + schema: &Schema, + fragments: &[Fragment], + ) { + let field_ids = schema + .fields_pre_order() + .map(|f| f.id) + .collect::>(); + + // Remove indices for fields no longer in schema + indices.retain(|existing_index| { + existing_index + .fields + .iter() + .all(|field_id| field_ids.contains(field_id)) + || is_system_index(existing_index) + }); + + // Fragment bitmaps record which fragments the index was originally built for. + // Operations like updates and data replacement prune these bitmaps, and + // effective_fragment_bitmap intersects with existing fragments at query time. + + // Apply retention logic for indices with empty bitmaps per index name + // (except for fragment reuse indices which are always kept) + let mut indices_by_name: std::collections::HashMap> = + std::collections::HashMap::new(); + + // Group indices by name + for index in indices.iter() { + if index.name != FRAG_REUSE_INDEX_NAME { + indices_by_name + .entry(index.name.clone()) + .or_default() + .push(index); + } + } + + // Build a set of UUIDs to keep based on retention rules + let mut uuids_to_keep = std::collections::HashSet::new(); + + let existing_fragments = fragments + .iter() + .map(|f| f.id as u32) + .collect::(); + + // For each group of indices with the same name + for (_, same_name_indices) in indices_by_name { + if same_name_indices.len() > 1 { + // Separate empty and non-empty indices + let (empty_indices, non_empty_indices): (Vec<_>, Vec<_>) = + same_name_indices.iter().partition(|index| { + index + .effective_fragment_bitmap(&existing_fragments) + .as_ref() + .is_none_or(|bitmap| bitmap.is_empty()) + }); + + if non_empty_indices.is_empty() { + // All indices are empty - for scalar indices, keep only the first (oldest) one + // For vector indices, remove all of them + let mut sorted_indices = empty_indices; + sorted_indices.sort_by_key(|index: &&IndexMetadata| index.dataset_version); // Sort by ascending dataset_version + + // Keep only the first (oldest) if it's not a vector index + if let Some(oldest) = sorted_indices.first() + && !Self::is_vector_index(oldest) + { + uuids_to_keep.insert(oldest.uuid); + } + } else { + // At least one index has non-empty bitmap - keep all non-empty indices + for index in non_empty_indices { + uuids_to_keep.insert(index.uuid); + } + } + } else { + // Single index - keep it unless it's an empty vector index + if let Some(index) = same_name_indices.first() { + let is_empty = index + .effective_fragment_bitmap(&existing_fragments) + .as_ref() + .is_none_or(|bitmap| bitmap.is_empty()); + let is_vector = Self::is_vector_index(index); + + // Keep the index unless it's an empty vector index + if !is_empty || !is_vector { + uuids_to_keep.insert(index.uuid); + } + } + } + } + + // Use Vec::retain to safely remove indices + indices.retain(|index| { + index.name == FRAG_REUSE_INDEX_NAME || uuids_to_keep.contains(&index.uuid) + }); + } + + pub(super) fn recalculate_fragment_bitmap( + old: &RoaringBitmap, + groups: &[RewriteGroup], + ) -> Result { + let mut new_bitmap = old.clone(); + for group in groups { + let any_in_index = group + .old_fragments + .iter() + .any(|frag| old.contains(frag.id as u32)); + let all_in_index = group + .old_fragments + .iter() + .all(|frag| old.contains(frag.id as u32)); + // Any rewrite group may or may not be covered by the index. However, if any fragment + // in a rewrite group was previously covered by the index then all fragments in the rewrite + // group must have been previously covered by the index. plan_compaction takes care of + // this for us so this should be safe to assume. + if any_in_index { + if all_in_index { + for frag_id in group.old_fragments.iter().map(|frag| frag.id as u32) { + new_bitmap.remove(frag_id); + } + new_bitmap.extend(group.new_fragments.iter().map(|frag| frag.id as u32)); + } else { + return Err(Error::invalid_input( + "The compaction plan included a rewrite group that was a split of indexed and non-indexed data", + )); + } + } + } + Ok(new_bitmap) + } + + pub(super) fn handle_rewrite_indices( + indices: &mut [IndexMetadata], + rewritten_indices: &[RewrittenIndex], + groups: &[RewriteGroup], + ) -> Result<()> { + let mut modified_indices = HashSet::new(); + + for rewritten_index in rewritten_indices { + if !modified_indices.insert(rewritten_index.old_id) { + return Err(Error::invalid_input(format!( + "An invalid compaction plan must have been generated because multiple tasks modified the same index: {}", + rewritten_index.old_id + ))); + } + + // Skip indices that no longer exist (may have been removed by concurrent operation) + let Some(index) = indices + .iter_mut() + .find(|idx| idx.uuid == rewritten_index.old_id) + else { + continue; + }; + + index.fragment_bitmap = Some(Self::recalculate_fragment_bitmap( + index.fragment_bitmap.as_ref().ok_or_else(|| { + Error::invalid_input(format!( + "Cannot rewrite index {} which did not store fragment bitmap", + index.uuid + )) + })?, + groups, + )?); + index.uuid = rewritten_index.new_id; + // Update file sizes to match the new index files. When not available + // (e.g., from older writers), clear the old file sizes to avoid + // using stale sizes from the pre-remap index. + index.files = rewritten_index.new_index_files.clone(); + } + Ok(()) + } + + pub(super) fn handle_rewrite_fragments( + final_fragments: &mut Vec, + groups: &[RewriteGroup], + fragment_id: &mut u64, + version: u64, + _next_row_id: Option<&u64>, + ) -> Result<()> { + for group in groups { + // If the old fragments are contiguous, find the range + let replace_range = { + let start = final_fragments + .iter() + .enumerate() + .find(|(_, f)| f.id == group.old_fragments[0].id) + .ok_or_else(|| { + Error::commit_conflict_source( + version, + format!( + "dataset does not contain a fragment a rewrite operation wants to replace: id={}", + group.old_fragments[0].id + ) + .into(), + ) + })? + .0; + + // Verify old_fragments matches contiguous range + let mut i = 1; + loop { + if i == group.old_fragments.len() { + break Some(start..start + i); + } + if final_fragments[start + i].id != group.old_fragments[i].id { + break None; + } + i += 1; + } + }; + + let new_fragments = Self::fragments_with_ids(group.new_fragments.clone(), fragment_id) + .collect::>(); + + // Version metadata for rewritten fragments is handled by the compaction code + // (recalc_versions_for_rewritten_fragments) which preserves version information + // from the original fragments. We don't modify it here. + + if let Some(replace_range) = replace_range { + // Efficiently path using slice + final_fragments.splice(replace_range, new_fragments); + } else { + // Slower path for non-contiguous ranges + for fragment in group.old_fragments.iter() { + final_fragments.retain(|f| f.id != fragment.id); + } + final_fragments.extend(new_fragments); + } + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::transaction::test_support::overlay_with_field; + use uuid::Uuid; + + #[test] + fn test_rewrite_fragments() { + let existing_fragments: Vec = (0..10).map(Fragment::new).collect(); + + let mut final_fragments = existing_fragments; + let rewrite_groups = vec![ + // Since these are contiguous, they will be put in the same location + // as 1 and 2. + RewriteGroup { + old_fragments: vec![Fragment::new(1), Fragment::new(2)], + // These two fragments were previously reserved + new_fragments: vec![Fragment::new(15), Fragment::new(16)], + }, + // These are not contiguous, so they will be inserted at the end. + RewriteGroup { + old_fragments: vec![Fragment::new(5), Fragment::new(8)], + // We pretend this id was not reserved. Does not happen in practice today + // but we want to leave the door open. + new_fragments: vec![Fragment::new(0)], + }, + ]; + + let mut fragment_id = 20; + let version = 0; + + Transaction::handle_rewrite_fragments( + &mut final_fragments, + &rewrite_groups, + &mut fragment_id, + version, + None, + ) + .unwrap(); + + assert_eq!(fragment_id, 21); + + let expected_fragments: Vec = vec![ + Fragment::new(0), + Fragment::new(15), + Fragment::new(16), + Fragment::new(3), + Fragment::new(4), + Fragment::new(6), + Fragment::new(7), + Fragment::new(9), + Fragment::new(20), + ]; + + assert_eq!(final_fragments, expected_fragments); + } + + #[test] + fn test_retain_indices_removes_missing_fields() { + let schema = create_test_schema(&[1, 2]); + let fragments = vec![Fragment::new(1), Fragment::new(2)]; + + let mut indices = vec![ + create_test_index("idx1", 1, 1, Some(RoaringBitmap::from_iter([1])), false), + create_test_index("idx2", 2, 1, Some(RoaringBitmap::from_iter([1])), false), + create_test_index("idx3", 99, 1, Some(RoaringBitmap::from_iter([1])), false), // Field doesn't exist + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + assert_eq!(indices.len(), 2); + assert!(indices.iter().all(|idx| idx.fields[0] != 99)); + } + + #[test] + fn test_retain_indices_keeps_system_indices() { + use crate::system_index::mem_wal::MEM_WAL_INDEX_NAME; + + let schema = create_test_schema(&[1, 2]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_system_index(FRAG_REUSE_INDEX_NAME, 99), // Field doesn't exist but should be kept + create_system_index(MEM_WAL_INDEX_NAME, 99), // Field doesn't exist but should be kept + create_test_index("regular_idx", 99, 1, Some(RoaringBitmap::new()), false), // Should be removed + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + assert_eq!(indices.len(), 2); + assert!(indices.iter().any(|idx| idx.name == FRAG_REUSE_INDEX_NAME)); + assert!(indices.iter().any(|idx| idx.name == MEM_WAL_INDEX_NAME)); + } + + #[test] + fn test_retain_indices_keeps_fragment_reuse_index() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_system_index(FRAG_REUSE_INDEX_NAME, 1), + create_test_index("other_idx", 1, 1, Some(RoaringBitmap::new()), false), + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Fragment reuse index should always be kept + assert!(indices.iter().any(|idx| idx.name == FRAG_REUSE_INDEX_NAME)); + } + + #[test] + fn test_retain_single_empty_scalar_index() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![create_test_index( + "scalar_idx", + 1, + 1, + Some(RoaringBitmap::new()), // Empty bitmap + false, + )]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Single empty scalar index should be kept + assert_eq!(indices.len(), 1); + } + + #[test] + fn test_retain_single_empty_vector_index() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![create_test_index( + "vector_idx", + 1, + 1, + Some(RoaringBitmap::new()), // Empty bitmap + true, + )]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Single empty vector index should be removed + assert_eq!(indices.len(), 0); + } + + #[test] + fn test_retain_single_nonempty_index() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut scalar_indices = vec![create_test_index( + "scalar_idx", + 1, + 1, + Some(RoaringBitmap::from_iter([1])), + false, + )]; + + let mut vector_indices = vec![create_test_index( + "vector_idx", + 1, + 1, + Some(RoaringBitmap::from_iter([1])), + true, + )]; + + Transaction::retain_relevant_indices(&mut scalar_indices, &schema, &fragments); + Transaction::retain_relevant_indices(&mut vector_indices, &schema, &fragments); + + // Both should be kept + assert_eq!(scalar_indices.len(), 1); + assert_eq!(vector_indices.len(), 1); + } + + #[test] + fn test_retain_single_index_with_none_bitmap() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut scalar_indices = vec![create_test_index("scalar_idx", 1, 1, None, false)]; + let mut vector_indices = vec![create_test_index("vector_idx", 1, 1, None, true)]; + + Transaction::retain_relevant_indices(&mut scalar_indices, &schema, &fragments); + Transaction::retain_relevant_indices(&mut vector_indices, &schema, &fragments); + + // Scalar should be kept, vector should be removed + assert_eq!(scalar_indices.len(), 1); + assert_eq!(vector_indices.len(), 0); + } + + #[test] + fn test_retain_multiple_empty_scalar_indices_keeps_oldest() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_test_index("idx", 1, 3, Some(RoaringBitmap::new()), false), + create_test_index("idx", 1, 1, Some(RoaringBitmap::new()), false), // Oldest + create_test_index("idx", 1, 2, Some(RoaringBitmap::new()), false), + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Should keep only the oldest (dataset_version = 1) + assert_eq!(indices.len(), 1); + assert_eq!(indices[0].dataset_version, 1); + } + + #[test] + fn test_retain_multiple_empty_vector_indices_removes_all() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_test_index("vec_idx", 1, 1, Some(RoaringBitmap::new()), true), + create_test_index("vec_idx", 1, 2, Some(RoaringBitmap::new()), true), + create_test_index("vec_idx", 1, 3, Some(RoaringBitmap::new()), true), + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // All empty vector indices should be removed + assert_eq!(indices.len(), 0); + } + + #[test] + fn test_retain_mixed_empty_nonempty_keeps_nonempty() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_test_index("idx", 1, 1, Some(RoaringBitmap::new()), false), // Empty + create_test_index("idx", 1, 2, Some(RoaringBitmap::from_iter([1])), false), // Non-empty + create_test_index("idx", 1, 3, Some(RoaringBitmap::new()), false), // Empty + create_test_index("idx", 1, 4, Some(RoaringBitmap::from_iter([1])), false), // Non-empty + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Should keep only non-empty indices + assert_eq!(indices.len(), 2); + assert!( + indices + .iter() + .all(|idx| idx.dataset_version == 2 || idx.dataset_version == 4) + ); + } + + #[test] + fn test_retain_mixed_empty_nonempty_vector_keeps_nonempty() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_test_index("vec_idx", 1, 1, Some(RoaringBitmap::new()), true), // Empty + create_test_index("vec_idx", 1, 2, Some(RoaringBitmap::from_iter([1])), true), // Non-empty + create_test_index("vec_idx", 1, 3, Some(RoaringBitmap::new()), true), // Empty + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Should keep only non-empty index + assert_eq!(indices.len(), 1); + assert_eq!(indices[0].dataset_version, 2); + } + + #[test] + fn test_retain_fragment_bitmap_with_nonexistent_fragments() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1), Fragment::new(2)]; // Only fragments 1 and 2 exist + + let mut indices = vec![create_test_index( + "idx", + 1, + 1, + Some(RoaringBitmap::from_iter([1, 2, 3, 4])), // References non-existent fragments 3, 4 + false, + )]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Should still keep the index (effective bitmap will be intersection with existing) + assert_eq!(indices.len(), 1); + // Original bitmap should be unchanged + assert_eq!( + indices[0].fragment_bitmap.as_ref().unwrap(), + &RoaringBitmap::from_iter([1, 2, 3, 4]) + ); + } + + #[test] + fn test_retain_effective_empty_bitmap_single_index() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(5), Fragment::new(6)]; + + // Bitmap references fragments that don't exist, so effective bitmap is empty + let mut scalar_indices = vec![create_test_index( + "scalar_idx", + 1, + 1, + Some(RoaringBitmap::from_iter([1, 2, 3])), + false, + )]; + + let mut vector_indices = vec![create_test_index( + "vector_idx", + 1, + 1, + Some(RoaringBitmap::from_iter([1, 2, 3])), + true, + )]; + + Transaction::retain_relevant_indices(&mut scalar_indices, &schema, &fragments); + Transaction::retain_relevant_indices(&mut vector_indices, &schema, &fragments); + + // Scalar should be kept (single index, even if effective bitmap is empty) + // Vector should be removed (empty effective bitmap) + assert_eq!(scalar_indices.len(), 1); + assert_eq!(vector_indices.len(), 0); + } + + #[test] + fn test_retain_different_index_names() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_test_index("idx_a", 1, 1, Some(RoaringBitmap::new()), false), + create_test_index("idx_b", 1, 1, Some(RoaringBitmap::new()), true), + create_test_index("idx_c", 1, 1, Some(RoaringBitmap::from_iter([1])), false), + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // idx_a (empty scalar) should be kept, idx_b (empty vector) removed, idx_c (non-empty) kept + assert_eq!(indices.len(), 2); + assert!(indices.iter().any(|idx| idx.name == "idx_a")); + assert!(indices.iter().any(|idx| idx.name == "idx_c")); + assert!(!indices.iter().any(|idx| idx.name == "idx_b")); + } + + #[test] + fn test_retain_empty_indices_vec() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices: Vec = vec![]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + assert_eq!(indices.len(), 0); + } + + #[test] + fn test_retain_all_indices_removed() { + let schema = create_test_schema(&[1]); + let fragments = vec![Fragment::new(1)]; + + let mut indices = vec![ + create_test_index("vec1", 1, 1, Some(RoaringBitmap::new()), true), + create_test_index("vec2", 1, 1, Some(RoaringBitmap::new()), true), + create_test_index("idx3", 99, 1, Some(RoaringBitmap::from_iter([1])), false), // Bad field + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + assert_eq!(indices.len(), 0); + } + + #[test] + fn test_retain_complex_scenario() { + let schema = create_test_schema(&[1, 2]); + let fragments = vec![Fragment::new(1), Fragment::new(2)]; + + let mut indices = vec![ + // System index - should always be kept + create_system_index(FRAG_REUSE_INDEX_NAME, 1), + // Group "idx_a" - all empty scalars, keep oldest + create_test_index("idx_a", 1, 3, Some(RoaringBitmap::new()), false), + create_test_index("idx_a", 1, 1, Some(RoaringBitmap::new()), false), // Oldest + create_test_index("idx_a", 1, 2, Some(RoaringBitmap::new()), false), + // Group "vec_b" - all empty vectors, remove all + create_test_index("vec_b", 1, 1, Some(RoaringBitmap::new()), true), + create_test_index("vec_b", 1, 2, Some(RoaringBitmap::new()), true), + // Group "idx_c" - mixed empty/non-empty, keep non-empty + create_test_index("idx_c", 2, 1, Some(RoaringBitmap::new()), false), + create_test_index("idx_c", 2, 2, Some(RoaringBitmap::from_iter([1])), false), // Keep + create_test_index("idx_c", 2, 3, Some(RoaringBitmap::from_iter([2])), false), // Keep + // Single non-empty - keep + create_test_index("idx_d", 1, 1, Some(RoaringBitmap::from_iter([1, 2])), false), + // Index with bad field - remove + create_test_index("idx_e", 99, 1, Some(RoaringBitmap::from_iter([1])), false), + ]; + + Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); + + // Expected: frag_reuse, idx_a (oldest), idx_c (2 non-empty), idx_d = 5 total + assert_eq!(indices.len(), 5); + + // Verify system index kept + assert!(indices.iter().any(|idx| idx.name == FRAG_REUSE_INDEX_NAME)); + + // Verify idx_a kept oldest only + let idx_a_indices: Vec<_> = indices.iter().filter(|idx| idx.name == "idx_a").collect(); + assert_eq!(idx_a_indices.len(), 1); + assert_eq!(idx_a_indices[0].dataset_version, 1); + + // Verify vec_b all removed + assert!(!indices.iter().any(|idx| idx.name == "vec_b")); + + // Verify idx_c kept non-empty only + let idx_c_indices: Vec<_> = indices.iter().filter(|idx| idx.name == "idx_c").collect(); + assert_eq!(idx_c_indices.len(), 2); + assert!( + idx_c_indices + .iter() + .all(|idx| idx.dataset_version == 2 || idx.dataset_version == 3) + ); + + // Verify idx_d kept + assert!(indices.iter().any(|idx| idx.name == "idx_d")); + + // Verify idx_e removed (bad field) + assert!(!indices.iter().any(|idx| idx.name == "idx_e")); + } + + #[test] + fn test_handle_rewrite_indices_skips_missing_index() { + // Create an empty indices list + let mut indices = vec![]; + + // Create rewritten_indices referring to a non-existent index + let rewritten_indices = vec![RewrittenIndex { + old_id: Uuid::new_v4(), + new_id: Uuid::new_v4(), + new_index_details: prost_types::Any { + type_url: String::new(), + value: vec![], + }, + new_index_version: 1, + new_index_files: None, + }]; + + // Should succeed (skip missing index) instead of error + let result = Transaction::handle_rewrite_indices(&mut indices, &rewritten_indices, &[]); + assert!(result.is_ok()); + assert!(indices.is_empty()); + } + + #[test] + fn test_prune_overlay_stale_fields_from_indices() { + // Fragment 0 carried an overlay on field 1 committed at v5, and was + // fully compacted into new fragment 7. + let mut old_frag = Fragment::new(0); + old_frag.overlays = vec![overlay_with_field(1, 5)]; + let groups = vec![RewriteGroup { + old_fragments: vec![old_frag], + new_fragments: vec![Fragment::new(7)], + }]; + + // Post-remap state: every index already covers the new fragment (7). + let covering = || Some(RoaringBitmap::from_iter([7u32])); + let mut indices = vec![ + // Stale: covers the overlaid field 1, built (v2) before the overlay. + create_test_index("stale", 1, 2, covering(), false), + // Not stale: covers field 1 but built at the overlay's version (v5); + // `committed_version > dataset_version` is false at equality. + create_test_index("fresh", 1, 5, covering(), false), + // Unrelated: covers field 2, which the overlay never touched. + create_test_index("unrelated", 2, 2, covering(), false), + ]; + + Transaction::prune_overlay_stale_fields_from_indices(&mut indices, &groups); + + assert!( + !indices[0].fragment_bitmap.as_ref().unwrap().contains(7), + "stale index must drop the rewritten fragment from its coverage" + ); + assert!( + indices[1].fragment_bitmap.as_ref().unwrap().contains(7), + "an index built at/after the overlay is not stale" + ); + assert!( + indices[2].fragment_bitmap.as_ref().unwrap().contains(7), + "an index on an un-overlaid field is unaffected" + ); + } + + // Helper functions for retain_relevant_indices tests + fn create_test_index( + name: &str, + field_id: i32, + dataset_version: u64, + fragment_bitmap: Option, + is_vector: bool, + ) -> IndexMetadata { + use prost_types::Any; + use std::sync::Arc; + + let index_details = if is_vector { + Some(Arc::new(Any { + type_url: "type.googleapis.com/lance.index.VectorIndexDetails".to_string(), + value: vec![], + })) + } else { + Some(Arc::new(Any { + type_url: "type.googleapis.com/lance.index.ScalarIndexDetails".to_string(), + value: vec![], + })) + }; + + IndexMetadata { + uuid: Uuid::new_v4(), + fields: vec![field_id], + name: name.to_string(), + dataset_version, + fragment_bitmap, + index_details, + index_version: 1, + created_at: None, + base_id: None, + files: None, + } + } + + fn create_system_index(name: &str, field_id: i32) -> IndexMetadata { + use prost_types::Any; + use std::sync::Arc; + + IndexMetadata { + uuid: Uuid::new_v4(), + fields: vec![field_id], + name: name.to_string(), + dataset_version: 1, + fragment_bitmap: Some(RoaringBitmap::from_iter([1, 2])), + index_details: Some(Arc::new(Any { + type_url: "type.googleapis.com/lance.index.SystemIndexDetails".to_string(), + value: vec![], + })), + index_version: 1, + created_at: None, + base_id: None, + files: None, + } + } + + fn create_test_schema(field_ids: &[i32]) -> Schema { + use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; + use lance_core::datatypes::Schema as LanceSchema; + + let fields: Vec = field_ids + .iter() + .map(|id| ArrowField::new(format!("field_{}", id), DataType::Int32, false)) + .collect(); + + let arrow_schema = ArrowSchema::new(fields); + let mut lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); + + // Assign field IDs + for (i, field_id) in field_ids.iter().enumerate() { + lance_schema.mut_field_by_id(i as i32).unwrap().id = *field_id; + } + + lance_schema + } +} diff --git a/rust/lance-table/src/transaction/manifest_build.rs b/rust/lance-table/src/transaction/manifest_build.rs new file mode 100644 index 00000000000..6e54deb6c33 --- /dev/null +++ b/rust/lance-table/src/transaction/manifest_build.rs @@ -0,0 +1,1697 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Applying an operation to produce the next manifest. +//! +//! [`Transaction::build_manifest`] is the centre of this module and of the +//! transaction machinery generally: given the current manifest and index list, it +//! decides the new fragment list, the surviving indices and the next row id, then +//! assembles the manifest. Everything else in `super` exists to serve it -- the +//! operation vocabulary it matches on, the index rules it applies, the row version +//! metadata it stamps, the validation that runs before it. + +use crate::feature_flags::{FLAG_STABLE_ROW_IDS, apply_feature_flags}; +use crate::format::{ + DataFile, DataStorageFormat, Fragment, IndexMetadata, Manifest, ManifestBuildConfig, + overlay::DataOverlayFile, +}; +use crate::io::{ + commit::CommitHandler, + manifest::{read_manifest, read_manifest_indexes}, +}; +use crate::rowids::version::build_version_meta; +use crate::system_index::mem_wal::update_mem_wal_index_compacted_sstables; +use crate::transaction::UpdateMode::{RewriteColumns, RewriteRows}; +use crate::transaction::row_version::resolve_update_version_metadata; +use crate::transaction::update_map::apply_update_map; +use crate::transaction::validate::merge_fragment_physically_rewritten; +use crate::transaction::{DataReplacementGroup, Operation, Transaction, UpdatedFragmentOffsets}; +use lance_core::datatypes::{ + LANCE_UNENFORCED_CLUSTERING_KEY_POSITION, LANCE_UNENFORCED_PRIMARY_KEY, + LANCE_UNENFORCED_PRIMARY_KEY_POSITION, +}; +use lance_core::{Error, Result}; +use lance_file::version::LanceFileVersion; +use lance_io::object_store::ObjectStore; +use object_store::path::Path; +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; + +impl Transaction { + pub(super) fn fragments_with_ids<'a, T>( + new_fragments: T, + fragment_id: &'a mut u64, + ) -> impl Iterator + 'a + where + T: IntoIterator + 'a, + { + new_fragments.into_iter().map(move |mut f| { + if f.id == 0 { + f.id = *fragment_id; + *fragment_id += 1; + } + f + }) + } + + fn data_storage_format_from_files( + fragments: &[Fragment], + user_requested: Option, + ) -> Result { + if let Some(file_version) = Fragment::try_infer_version(fragments)? { + // Ensure user-requested matches data files + if let Some(user_requested) = user_requested + && user_requested != file_version + { + return Err(Error::invalid_input(format!( + "User requested data storage version ({}) does not match version in data files ({})", + user_requested, file_version + ))); + } + Ok(DataStorageFormat::new(file_version)) + } else { + // If no files use user-requested or default + Ok(user_requested + .map(DataStorageFormat::new) + .unwrap_or_default()) + } + } + + pub async fn restore_old_manifest( + object_store: &ObjectStore, + commit_handler: &dyn CommitHandler, + base_path: &Path, + version: u64, + config: &ManifestBuildConfig, + tx_path: &str, + current_manifest: &Manifest, + ) -> Result<(Manifest, Vec)> { + let location = commit_handler + .resolve_version_location(base_path, version, &object_store.inner) + .await?; + let mut manifest = read_manifest(object_store, &location.path, location.size).await?; + manifest.set_timestamp(config.timestamp_nanos); + manifest.transaction_file = Some(tx_path.to_string()); + let indices = read_manifest_indexes(object_store, &location, &manifest).await?; + manifest.max_fragment_id = manifest + .max_fragment_id + .max(current_manifest.max_fragment_id); + Ok((manifest, indices)) + } + + /// Create a new manifest from the current manifest and the transaction. + /// + /// `current_manifest` should only be None if the dataset does not yet exist. + pub fn build_manifest( + &self, + current_manifest: Option<&Manifest>, + current_indices: Vec, + transaction_file_path: &str, + config: &ManifestBuildConfig, + ) -> Result<(Manifest, Vec)> { + if config.use_stable_row_ids + && current_manifest + .map(|m| !m.uses_stable_row_ids()) + .unwrap_or_default() + { + return Err(Error::not_supported_source( + "Cannot enable stable row ids on existing dataset".into(), + )); + } + // An action-based operation assembles its own manifest. Keeping it out of + // the legacy assembly below is deliberate: that code is a frozen + // compatibility surface, and an action applies to the manifest directly + // rather than being folded into this function's post-image matches. + if let Operation::UserOperation(user_operation) = &self.operation { + return user_operation.build_manifest( + current_manifest, + current_indices, + self.tag.as_deref(), + transaction_file_path, + config, + ); + } + + let mut reference_paths = match current_manifest { + Some(m) => m.base_paths.clone(), + None => HashMap::new(), + }; + + if let Operation::Overwrite { + initial_bases: Some(initial_bases), + .. + } = &self.operation + { + if current_manifest.is_none() { + // CREATE mode: registering base paths + // Base IDs should have been assigned during write operation + // Validate uniqueness and insert them into the manifest + for base_path in initial_bases.iter() { + if reference_paths.contains_key(&base_path.id) { + return Err(Error::invalid_input(format!( + "Duplicate base path ID {} detected. Base path IDs must be unique.", + base_path.id + ))); + } + reference_paths.insert(base_path.id, base_path.clone()); + } + } else { + // OVERWRITE mode with initial_bases should have been rejected by validation + // This branch should never be reached + return Err(Error::invalid_input( + "OVERWRITE mode cannot register new bases. This should have been caught by validation.", + )); + } + } + + // Get the schema and the final fragment list + let schema = match self.operation { + Operation::Overwrite { ref schema, .. } => schema.clone(), + Operation::Merge { ref schema, .. } => schema.clone(), + Operation::Project { ref schema, .. } => schema.clone(), + _ => { + if let Some(current_manifest) = current_manifest { + current_manifest.schema.clone() + } else { + return Err(Error::internal( + "Cannot create a new dataset without a schema".to_string(), + )); + } + } + }; + + let mut fragment_id = if matches!(self.operation, Operation::Overwrite { .. }) { + 0 + } else { + current_manifest + .and_then(|m| m.max_fragment_id()) + .map(|id| id + 1) + .unwrap_or(0) + }; + let mut final_fragments = Vec::new(); + let mut final_indices = current_indices; + + let mut next_row_id = { + // Only use row ids if the feature flag is set already or + match (current_manifest, config.use_stable_row_ids) { + (Some(manifest), _) if manifest.reader_feature_flags & FLAG_STABLE_ROW_IDS != 0 => { + Some(manifest.next_row_id) + } + (None, true) => Some(0), + (_, false) => None, + (Some(_), true) => { + return Err(Error::not_supported_source( + "Cannot enable stable row ids on existing dataset".into(), + )); + } + } + }; + + let maybe_existing_fragments = + current_manifest + .map(|m| m.fragments.as_ref()) + .ok_or_else(|| { + Error::internal(format!( + "No current manifest was provided while building manifest for operation {}", + self.operation.name() + )) + }); + + let new_version = current_manifest.map_or(1, |m| m.version + 1); + + match &self.operation { + Operation::Clone { .. } => { + return Err(Error::internal( + "Clone operation should not enter build_manifest.".to_string(), + )); + } + Operation::Append { fragments } => { + final_fragments.extend(maybe_existing_fragments?.clone()); + let mut new_fragments = + Self::fragments_with_ids(fragments.clone(), &mut fragment_id) + .collect::>(); + if let Some(next_row_id) = &mut next_row_id { + Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; + // Add version metadata for all new fragments + for fragment in new_fragments.iter_mut() { + let version_meta = build_version_meta(fragment, new_version); + fragment.last_updated_at_version_meta = version_meta.clone(); + fragment.created_at_version_meta = version_meta; + } + } + final_fragments.extend(new_fragments); + } + Operation::Delete { + updated_fragments, + deleted_fragment_ids, + .. + } => { + // Remove the deleted fragments + final_fragments.extend(maybe_existing_fragments?.clone()); + final_fragments.retain(|f| !deleted_fragment_ids.contains(&f.id)); + final_fragments.iter_mut().for_each(|f| { + for updated in updated_fragments { + if updated.id == f.id { + *f = updated.clone(); + } + } + }); + Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments) + } + Operation::Update { + removed_fragment_ids, + updated_fragments, + new_fragments, + fields_modified, + compacted_sstables, + fields_for_preserving_frag_bitmap, + update_mode, + updated_fragment_offsets, + .. + } => { + // Extract existing fragments once for reuse + let existing_fragments = maybe_existing_fragments?; + + // Apply updates to existing fragments + let updated_frags: Vec = existing_fragments + .iter() + .filter_map(|f| { + if removed_fragment_ids.contains(&f.id) { + return None; + } + if let Some(updated) = updated_fragments.iter().find(|uf| uf.id == f.id) { + let mut updated = updated.clone(); + // Carry forward the fragment's current overlays (which + // may include ones added by a concurrent commit). An + // in-place column rewrite then tombstones the overlaid + // fields it rewrote, since the fresh base values + // supersede them. + updated.overlays = f.overlays.clone(); + if matches!(update_mode, Some(RewriteColumns)) { + crate::format::overlay::tombstone_overlay_fields( + &mut updated.overlays, + fields_modified, + ); + } + Some(updated) + } else { + Some(f.clone()) + } + }) + .collect(); + + // Update version metadata for updated fragments if stable row IDs are enabled + // Note: We don't update version metadata for fragments with deletion vectors + // because the version sequences are indexed by physical row position, not logical position. + // Version metadata for deleted rows will be filtered out during scan using the deletion vector. + if next_row_id.is_some() { + // Version metadata will be properly set during compaction when deletions are materialized + } + + final_fragments.extend(updated_frags); + + if next_row_id.is_some() + && matches!(update_mode, Some(RewriteColumns)) + && let Some(UpdatedFragmentOffsets(off_map)) = updated_fragment_offsets + && !off_map.is_empty() + { + let prev_version = current_manifest.map(|m| m.version).unwrap_or(0); + for fragment in final_fragments.iter_mut() { + let Some(bitmap) = off_map.get(&fragment.id) else { + continue; + }; + if bitmap.is_empty() { + continue; + } + // Skip fragments with no existing version metadata: the helper + // would fill unmatched rows with prev_version, fabricating a + // last_updated stamp for rows that never had one. + if fragment.last_updated_at_version_meta.is_none() { + continue; + } + let offsets: Vec = bitmap.iter().map(|o| o as usize).collect(); + crate::rowids::version::refresh_row_latest_update_meta_for_partial_frag_rewrite_cols( + fragment, + &offsets, + new_version, + prev_version, + )?; + } + } + + // If we updated any fields, remove those fragments from indices covering those fields + Self::prune_updated_fields_from_indices( + &mut final_indices, + updated_fragments, + fields_modified, + ); + + let mut new_fragments = + Self::fragments_with_ids(new_fragments.clone(), &mut fragment_id) + .collect::>(); + + // Assign row IDs to any fragments that don't have them yet + // (e.g., inserted rows from merge_insert operations) + if let Some(next_row_id) = &mut next_row_id { + Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; + } + + if next_row_id.is_some() { + resolve_update_version_metadata( + existing_fragments, + new_fragments.as_mut_slice(), + new_version, + )?; + } + + if config.use_stable_row_ids + && update_mode.is_some() + && *update_mode == Some(RewriteRows) + { + let pure_updated_frag_ids = + Self::collect_pure_rewrite_row_update_frags_ids(&new_fragments)?; + + // collect all the original frag ids that contains the updated rows + let original_fragment_ids: Vec = removed_fragment_ids + .iter() + .chain(updated_fragments.iter().map(|f| &f.id)) + .copied() + .collect(); + + // The original fragments that carried an overlay: their moved rows may have a + // stale index entry (see `register_pure_rewrite_rows_update_frags_in_indices`). + let original_overlaid_frags: HashMap = existing_fragments + .iter() + .filter(|f| original_fragment_ids.contains(&f.id) && !f.overlays.is_empty()) + .map(|f| (f.id as u32, f)) + .collect(); + + Self::register_pure_rewrite_rows_update_frags_in_indices( + &mut final_indices, + &pure_updated_frag_ids, + &original_fragment_ids, + fields_for_preserving_frag_bitmap, + &original_overlaid_frags, + &schema, + )?; + } + + if let Some(next_row_id) = &mut next_row_id { + Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; + // Note: Version metadata is already set above (lines 1627-1755) + // for Update operations, preserving created_at from original fragments. + // Don't overwrite it here. + } + // Identify fragments that were updated or newly created in this update + let mut target_ids: HashSet = HashSet::new(); + target_ids.extend(new_fragments.iter().map(|f| f.id)); + final_fragments.extend(new_fragments); + Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments); + + if !compacted_sstables.is_empty() { + update_mem_wal_index_compacted_sstables( + &mut final_indices, + new_version, + compacted_sstables.clone(), + )?; + } + } + Operation::Overwrite { fragments, .. } => { + let mut new_fragments = + Self::fragments_with_ids(fragments.clone(), &mut fragment_id) + .collect::>(); + if let Some(next_row_id) = &mut next_row_id { + Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; + // Add version metadata for all new fragments + for fragment in new_fragments.iter_mut() { + let version_meta = build_version_meta(fragment, new_version); + fragment.last_updated_at_version_meta = version_meta.clone(); + fragment.created_at_version_meta = version_meta; + } + } + final_fragments.extend(new_fragments); + final_indices = Vec::new(); + } + Operation::Rewrite { + groups, + rewritten_indices, + frag_reuse_index, + } => { + final_fragments.extend(maybe_existing_fragments?.clone()); + let current_version = current_manifest.map(|m| m.version).unwrap_or_default(); + Self::handle_rewrite_fragments( + &mut final_fragments, + groups, + &mut fragment_id, + current_version, + next_row_id.as_ref(), + )?; + + if next_row_id.is_some() { + // We can re-use indices, but need to rewrite the fragment bitmaps + debug_assert!(rewritten_indices.is_empty()); + for index in final_indices.iter_mut() { + if let Some(fragment_bitmap) = &mut index.fragment_bitmap { + *fragment_bitmap = + Self::recalculate_fragment_bitmap(fragment_bitmap, groups)?; + } + } + } else { + Self::handle_rewrite_indices(&mut final_indices, rewritten_indices, groups)?; + } + + // A full compaction materializes a fragment's overlays into fresh + // base data. Any index older than one of those overlays was built on + // the pre-overlay values, so drop the rewritten fragment from its + // coverage to keep it from serving stale values. + Self::prune_overlay_stale_fields_from_indices(&mut final_indices, groups); + + if let Some(frag_reuse_index) = frag_reuse_index { + final_indices.retain(|idx| idx.name != frag_reuse_index.name); + final_indices.push(frag_reuse_index.clone()); + } + } + Operation::CreateIndex { + new_indices, + removed_indices, + } => { + final_fragments.extend(maybe_existing_fragments?.clone()); + let removed_uuids = removed_indices + .iter() + .map(|old_index| old_index.uuid) + .collect::>(); + let new_uuids = new_indices + .iter() + .map(|new_index| new_index.uuid) + .collect::>(); + final_indices.retain(|existing_index| { + !removed_uuids.contains(&existing_index.uuid) + && !new_uuids.contains(&existing_index.uuid) + }); + final_indices.extend(new_indices.clone()); + } + Operation::ReserveFragments { .. } | Operation::UpdateConfig { .. } => { + final_fragments.extend(maybe_existing_fragments?.clone()); + } + Operation::Merge { fragments, .. } => { + let existing_fragments = maybe_existing_fragments?; + let mut merged_fragments = fragments.clone(); + if next_row_id.is_some() { + let prev_by_id: HashMap = + existing_fragments.iter().map(|f| (f.id, f)).collect(); + for fragment in merged_fragments.iter_mut() { + match prev_by_id.get(&fragment.id) { + Some(prev) => { + if merge_fragment_physically_rewritten(prev, fragment) { + crate::rowids::version::refresh_row_latest_update_meta_for_full_frag_rewrite_cols( + fragment, + new_version, + )?; + } + } + None => { + // Brand-new fragment ID not present in the previous manifest. + // Set both last_updated and created version meta, consistent + // with Append/Overwrite for genuinely new fragments. + crate::rowids::version::refresh_row_latest_update_meta_for_full_frag_rewrite_cols( + fragment, + new_version, + )?; + fragment.created_at_version_meta = + fragment.last_updated_at_version_meta.clone(); + } + } + } + } + final_fragments.extend(merged_fragments); + + // A Merge can rewrite a column's data file in place; the field stays + // in the schema, so the index is retained -- prune its now-stale + // entries for the rewritten fragments. + Self::prune_merge_rewritten_fields_from_indices( + &mut final_indices, + existing_fragments, + fragments, + ); + + // Some fields that have indices may have been removed, so we should + // remove those indices as well. + Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments) + } + Operation::Project { .. } => { + final_fragments.extend(maybe_existing_fragments?.clone()); + + // We might have removed all fields for certain data files, so + // we should remove the data files that are no longer relevant. + let remaining_field_ids = schema + .fields_pre_order() + .map(|f| f.id) + .collect::>(); + for fragment in final_fragments.iter_mut() { + fragment.files.retain(|file| { + file.fields + .iter() + .any(|field_id| remaining_field_ids.contains(field_id)) + }); + } + + // Some fields that have indices may have been removed, so we should + // remove those indices as well. + Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments) + } + Operation::Restore { .. } => { + unreachable!() + } + Operation::DataReplacement { replacements } => { + log::warn!( + "Building manifest with DataReplacement operation. This operation is not stable yet, please use with caution." + ); + + let (old_fragment_ids, new_datafiles): (Vec<&u64>, Vec<&DataFile>) = replacements + .iter() + .map(|DataReplacementGroup(fragment_id, new_file)| (fragment_id, new_file)) + .unzip(); + + // 1. make sure the new files all have the same fields / or empty + // NOTE: arguably this requirement could be relaxed in the future + // for the sake of simplicity, we require the new files to have the same fields + if new_datafiles + .iter() + .map(|f| f.fields.clone()) + .collect::>() + .len() + > 1 + { + let field_info = new_datafiles + .iter() + .enumerate() + .map(|(id, f)| (id, f.fields.clone())) + .fold("".to_string(), |acc, (id, fields)| { + format!("{}File {}: {:?}\n", acc, id, fields) + }); + + return Err(Error::invalid_input(format!( + "All new data files must have the same fields, but found different fields:\n{field_info}" + ))); + } + + let existing_fragments = maybe_existing_fragments?; + + // Collect replaced field IDs before consuming new_datafiles + let replaced_fields: Vec = new_datafiles + .first() + .map(|f| { + f.fields + .iter() + .filter(|&&id| id >= 0) + .map(|&id| id as u32) + .collect() + }) + .unwrap_or_default(); + + // 2. check that the fragments being modified have isomorphic layouts along the columns being replaced + // 3. add modified fragments to final_fragments + for (frag_id, new_file) in old_fragment_ids.iter().zip(new_datafiles) { + let frag = existing_fragments + .iter() + .find(|f| f.id == **frag_id) + .ok_or_else(|| { + Error::invalid_input( + "Fragment being replaced not found in existing fragments", + ) + })?; + let mut new_frag = frag.clone(); + + // TODO(rmeng): check new file and fragment are the same length + + let mut columns_covered = HashSet::new(); + for file in &mut new_frag.files { + if file.fields == new_file.fields + && file.file_major_version == new_file.file_major_version + && file.file_minor_version == new_file.file_minor_version + { + // assign the new file path / size / base to the fragment + file.path = new_file.path.clone(); + file.file_size_bytes = new_file.file_size_bytes.clone(); + file.base_id = new_file.base_id; + } + columns_covered.extend(file.fields.iter()); + } + // SPECIAL CASE: if the column(s) being replaced are not covered by the fragment + // Then it means it's a all-NULL column that is being replaced with real data + // just add it to the final fragments. Push the DataFile as + // given so every field (including base_id) is preserved. + if columns_covered.is_disjoint(&new_file.fields.iter().collect()) { + LanceFileVersion::try_from_major_minor( + new_file.file_major_version, + new_file.file_minor_version, + ) + .expect("Expected valid file version"); + new_frag.files.push(new_file.clone()); + } + + // Nothing changed in the current fragment, which is not expected -- error out + if &new_frag == frag { + return Err(Error::invalid_input( + "Expected to modify the fragment but no changes were made. This means the new data files does not align with any exiting datafiles. Please check if the schema of the new data files matches the schema of the old data files including the file major and minor versions", + )); + } + + // New base values for these fields supersede any overlay + // still shadowing them; tombstone the overlaid fields so the + // replacement is not silently masked. + crate::format::overlay::tombstone_overlay_fields( + &mut new_frag.overlays, + &replaced_fields, + ); + + final_fragments.push(new_frag); + } + + let fragments_changed = old_fragment_ids + .iter() + .cloned() + .cloned() + .collect::>(); + + // 4. push fragments that didn't change back to final_fragments + let unmodified_fragments = existing_fragments + .iter() + .filter(|f| !fragments_changed.contains(&f.id)) + .cloned() + .collect::>(); + + final_fragments.extend(unmodified_fragments); + + // 5. Invalidate index bitmaps for replaced fields + let modified_fragments: Vec = final_fragments + .iter() + .filter(|f| fragments_changed.contains(&f.id)) + .cloned() + .collect(); + + Self::prune_updated_fields_from_indices( + &mut final_indices, + &modified_fragments, + &replaced_fields, + ); + } + Operation::DataOverlay { groups } => { + // Stamp each overlay with the version this commit is producing. + // build_manifest re-runs on every retry with an updated + // current_manifest, so this is naturally re-stamped on retry. + let new_version = current_manifest.map_or(1, |m| m.version + 1); + + let existing_fragments = maybe_existing_fragments?; + // Multiple groups may target the same fragment; merge them in + // order rather than letting a HashMap collapse drop all but the + // last group's overlays. + let mut overlays_by_fragment: HashMap> = HashMap::new(); + for group in groups { + overlays_by_fragment + .entry(group.fragment_id) + .or_default() + .extend(group.overlays.iter()); + } + + // Every group must target an existing fragment. Build a set of + // existing ids once so this is O(groups + fragments) rather than + // O(groups * fragments). + let existing_fragment_ids: HashSet = + existing_fragments.iter().map(|f| f.id).collect(); + for fragment_id in overlays_by_fragment.keys() { + if !existing_fragment_ids.contains(fragment_id) { + return Err(Error::invalid_input(format!( + "DataOverlay targets fragment {fragment_id}, which does not exist" + ))); + } + } + + for fragment in existing_fragments { + let mut fragment = fragment.clone(); + if let Some(new_overlays) = overlays_by_fragment.get(&fragment.id) { + // Appended (not replaced) so concurrently-written overlays + // survive; later entries are newer. + fragment + .overlays + .extend(new_overlays.iter().map(|&overlay| { + let mut overlay = overlay.clone(); + overlay.committed_version = new_version; + overlay + })); + } + final_fragments.push(fragment); + } + } + Operation::UpdateMemWalState { compacted_sstables } => { + update_mem_wal_index_compacted_sstables( + &mut final_indices, + new_version, + compacted_sstables.clone(), + )?; + } + Operation::UpdateBases { .. } => { + // UpdateBases operation doesn't modify fragments or indices + // Base paths are handled in the manifest creation section below + final_fragments.extend(maybe_existing_fragments?.clone()); + } + Operation::UserOperation(_) => { + return Err(Error::internal( + "an action-based operation reached the legacy manifest assembly; \ + it should have been dispatched to UserOperation::build_manifest" + .to_string(), + )); + } + }; + + // If a fragment was reserved then it may not belong at the end of the fragments list. + final_fragments.sort_by_key(|frag| frag.id); + + // Clean up data files that only contain tombstoned fields + Self::remove_tombstoned_data_files(&mut final_fragments); + + // Enforce the newest-last overlay ordering invariant at the write + // boundary. Load normalizes with a sort; this rejects any commit path + // that assembled a fragment's overlays out of order. + for fragment in &final_fragments { + if !fragment.overlays.is_empty() { + crate::format::overlay::verify_overlays_newest_last(&fragment.overlays)?; + } + } + + let user_requested_version = match (&config.storage_format, config.use_legacy_format) { + (Some(storage_format), _) => Some(storage_format.lance_file_version()?), + (None, Some(true)) => Some(LanceFileVersion::Legacy), + (None, Some(false)) => Some(LanceFileVersion::V2_0), + (None, None) => None, + }; + + let mut manifest = if let Some(current_manifest) = current_manifest { + // OVERWRITE with initial_bases on existing dataset is not allowed (caught by validation) + // So we always use new_from_previous which preserves base_paths + let mut prev_manifest = + Manifest::new_from_previous(current_manifest, schema, Arc::new(final_fragments)); + + if let (Some(user_requested_version), Operation::Overwrite { .. }) = + (user_requested_version, &self.operation) + { + // If this is an overwrite operation and the user has requested a specific version + // then overwrite with that version. Otherwise, if the user didn't request a specific + // version, then overwrite with whatever version we had before. + prev_manifest.data_storage_format = DataStorageFormat::new(user_requested_version); + } + + prev_manifest + } else { + let data_storage_format = + Self::data_storage_format_from_files(&final_fragments, user_requested_version)?; + Manifest::new( + schema, + Arc::new(final_fragments), + data_storage_format, + reference_paths, + ) + }; + + manifest.tag.clone_from(&self.tag); + + if config.auto_set_feature_flags { + // Internal operations (e.g. CreateIndex) build with the default config, + // which has use_stable_row_ids = false. Without inheriting from the previous + // manifest, apply_feature_flags would clear FLAG_STABLE_ROW_IDS. + let inherited = current_manifest + .map(|m| m.uses_stable_row_ids()) + .unwrap_or(false); + let use_stable_row_ids = config.use_stable_row_ids || inherited; + apply_feature_flags( + &mut manifest, + use_stable_row_ids, + config.disable_transaction_file, + )?; + } + manifest.set_timestamp(config.timestamp_nanos); + + manifest.update_max_fragment_id(); + + match &self.operation { + Operation::Overwrite { + config_upsert_values: Some(tm), + .. + } => { + manifest.config_mut().extend(tm.clone()); + } + Operation::UpdateConfig { + config_updates, + table_metadata_updates, + schema_metadata_updates, + field_metadata_updates, + } => { + if let Some(config_updates) = config_updates { + let mut config = manifest.config.clone(); + apply_update_map(&mut config, config_updates); + manifest.config = config; + } + if let Some(table_metadata_updates) = table_metadata_updates { + let mut table_metadata = manifest.table_metadata.clone(); + apply_update_map(&mut table_metadata, table_metadata_updates); + manifest.table_metadata = table_metadata; + } + if let Some(schema_metadata_updates) = schema_metadata_updates { + let mut schema_metadata = manifest.schema.metadata.clone(); + apply_update_map(&mut schema_metadata, schema_metadata_updates); + manifest.schema.metadata = schema_metadata; + } + // The unenforced primary and clustering keys are reserved + // schema properties: each is immutable once set, and its + // reserved metadata keys cannot be written with an invalid + // value. Capture the prior keys, and whether this transaction + // writes a reserved key, before applying the updates so + // violations can be rejected below. This runs on every apply, + // including conflict-rebase, so it also rejects the + // concurrent-writer race. + let primary_key_before: Vec = manifest + .schema + .unenforced_primary_key() + .iter() + .map(|field| field.id) + .collect(); + let writes_primary_key = field_metadata_updates.values().any(|update| { + update.update_entries.iter().any(|entry| { + entry.key == LANCE_UNENFORCED_PRIMARY_KEY + || entry.key == LANCE_UNENFORCED_PRIMARY_KEY_POSITION + }) + }); + let clustering_key_before: Vec = manifest + .schema + .unenforced_clustering_key() + .iter() + .map(|field| field.id) + .collect(); + let writes_clustering_key = field_metadata_updates.values().any(|update| { + update + .update_entries + .iter() + .any(|entry| entry.key == LANCE_UNENFORCED_CLUSTERING_KEY_POSITION) + }); + for (field_id, field_metadata_update) in field_metadata_updates { + if let Some(field) = manifest.schema.field_by_id_mut(*field_id) { + apply_update_map(&mut field.metadata, field_metadata_update); + // Also set unenforced primary key based on updated field metadata. + field.unenforced_primary_key_position = field + .metadata + .get(LANCE_UNENFORCED_PRIMARY_KEY_POSITION) + .and_then(|s| s.parse::().ok()) + .or_else(|| { + field + .metadata + .get(LANCE_UNENFORCED_PRIMARY_KEY) + .filter(|s| { + matches!(s.to_lowercase().as_str(), "true" | "1" | "yes") + }) + .map(|_| 0) + }); + // Also set unenforced clustering key based on updated + // field metadata. + field.unenforced_clustering_key_position = field + .metadata + .get(LANCE_UNENFORCED_CLUSTERING_KEY_POSITION) + .and_then(|s| s.parse::().ok()); + } else { + return Err(Error::invalid_input_source( + format!("Field with id {} does not exist", field_id).into(), + )); + } + } + let primary_key_after: Vec = manifest + .schema + .unenforced_primary_key() + .iter() + .map(|field| field.id) + .collect(); + if !primary_key_before.is_empty() { + // The primary key is already set: reject any change to it, + // and any write that touches a reserved primary key. + if writes_primary_key || primary_key_after != primary_key_before { + return Err(Error::invalid_input( + "the unenforced primary key is a reserved key and cannot be changed once set", + )); + } + } else if writes_primary_key && primary_key_after.is_empty() { + // A reserved primary key was written but did not install a + // valid primary key (e.g. a non-marker flag value or a + // non-numeric position). + return Err(Error::invalid_input( + "the unenforced primary key is a reserved key and cannot be set to an invalid value", + )); + } + let clustering_key_after: Vec = manifest + .schema + .unenforced_clustering_key() + .iter() + .map(|field| field.id) + .collect(); + if !clustering_key_before.is_empty() { + // The clustering key is already set: reject any change to + // it, and any write that touches the reserved key. + if writes_clustering_key || clustering_key_after != clustering_key_before { + return Err(Error::invalid_input( + "the unenforced clustering key is a reserved key and cannot be changed once set", + )); + } + } else if writes_clustering_key && clustering_key_after.is_empty() { + // The reserved clustering key was written but did not + // install a valid clustering key (e.g. a non-numeric + // position value). + return Err(Error::invalid_input( + "the unenforced clustering key is a reserved key and cannot be set to an invalid value", + )); + } + } + _ => {} + } + + // Handle UpdateBases operation to update manifest base_paths + if let Operation::UpdateBases { new_bases } = &self.operation { + // Validate and add new base paths to the manifest + for new_base in new_bases { + // Check for conflicts with existing base paths + if let Some(existing_base) = manifest + .base_paths + .values() + .find(|bp| bp.name == new_base.name || bp.path == new_base.path) + { + return Err(Error::invalid_input(format!( + "Conflict detected: Base path with name '{:?}' or path '{}' already exists. Existing: name='{:?}', path='{}'", + new_base.name, new_base.path, existing_base.name, existing_base.path + ))); + } + + // Assign a new ID if not already assigned + let mut base_to_add = new_base.clone(); + if base_to_add.id == 0 { + let next_id = manifest + .base_paths + .keys() + .max() + .map(|&id| id + 1) + .unwrap_or(1); + base_to_add.id = next_id; + } + + manifest.base_paths.insert(base_to_add.id, base_to_add); + } + } + + if let Operation::ReserveFragments { num_fragments } = self.operation { + manifest.max_fragment_id = Some(manifest.max_fragment_id.unwrap_or(0) + num_fragments); + } + + manifest.transaction_file = Some(transaction_file_path.to_string()); + + if let Some(next_row_id) = next_row_id { + manifest.next_row_id = next_row_id; + } + + Ok((manifest, final_indices)) + } + + /// Remove data files that only contain tombstoned fields (-2) + /// These files no longer contain any live data and can be safely dropped + fn remove_tombstoned_data_files(fragments: &mut [Fragment]) { + for fragment in fragments { + fragment.files.retain(|file| { + // Keep file if it has at least one non-tombstoned field + file.fields.iter().any(|&field_id| field_id != -2) + }); + } + } +} + +#[cfg(test)] +#[cfg(test)] +mod tests { + use super::*; + use crate::format::RowIdMeta; + use crate::format::overlay::OverlayCoverage; + use crate::rowids::RowIdSequence; + use crate::rowids::write_row_ids; + use crate::transaction::test_support::{ + default_build_config, make_stable_row_id_manifest, overlay_with_field, + sample_index_metadata, sample_manifest, + }; + use crate::transaction::{DataOverlayGroup, UpdateMode}; + use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; + use lance_core::datatypes::Schema as LanceSchema; + use lance_file::version::LanceFileVersion; + use lance_io::utils::CachedFileSize; + use roaring::RoaringBitmap; + use std::collections::HashMap; + use std::sync::Arc; + + #[test] + fn test_create_index_build_manifest_keeps_unremoved_same_name_indices() { + let manifest = sample_manifest(); + let first_index = sample_index_metadata("vector_idx"); + let second_index = sample_index_metadata("vector_idx"); + let third_index = sample_index_metadata("vector_idx"); + + let transaction = Transaction::new( + manifest.version, + Operation::CreateIndex { + new_indices: vec![third_index.clone()], + removed_indices: vec![second_index.clone()], + }, + None, + ); + + let (_, final_indices) = transaction + .build_manifest( + Some(&manifest), + vec![first_index.clone(), second_index.clone()], + "txn", + &default_build_config(), + ) + .unwrap(); + + assert_eq!(final_indices.len(), 2); + assert!(final_indices.iter().any(|idx| idx.uuid == first_index.uuid)); + assert!(final_indices.iter().any(|idx| idx.uuid == third_index.uuid)); + assert!( + !final_indices + .iter() + .any(|idx| idx.uuid == second_index.uuid) + ); + } + + #[test] + fn test_create_index_build_manifest_deduplicates_relisted_indices_by_uuid() { + let manifest = sample_manifest(); + let first_index = sample_index_metadata("vector_idx"); + let second_index = sample_index_metadata("vector_idx"); + let third_index = sample_index_metadata("vector_idx"); + + let transaction = Transaction::new( + manifest.version, + Operation::CreateIndex { + new_indices: vec![first_index.clone(), third_index.clone()], + removed_indices: vec![second_index.clone()], + }, + None, + ); + + let (_, final_indices) = transaction + .build_manifest( + Some(&manifest), + vec![first_index.clone(), second_index.clone()], + "txn", + &default_build_config(), + ) + .unwrap(); + + assert_eq!(final_indices.len(), 2); + assert_eq!( + final_indices + .iter() + .filter(|idx| idx.uuid == first_index.uuid) + .count(), + 1 + ); + assert!(final_indices.iter().any(|idx| idx.uuid == third_index.uuid)); + assert!( + !final_indices + .iter() + .any(|idx| idx.uuid == second_index.uuid) + ); + } + + #[test] + fn test_remove_tombstoned_data_files() { + // Create a fragment with mixed data files: some normal, some fully tombstoned + let mut fragment = Fragment::new(1); + + // Add a normal data file with valid field IDs + fragment.files.push(DataFile { + path: "normal.lance".to_string(), + fields: Arc::from([1, 2, 3]), + column_indices: Arc::from([]), + file_major_version: 2, + file_minor_version: 0, + file_size_bytes: CachedFileSize::new(1000), + base_id: None, + }); + + // Add a data file with all fields tombstoned + fragment.files.push(DataFile { + path: "all_tombstoned.lance".to_string(), + fields: Arc::from([-2, -2, -2]), + column_indices: Arc::from([]), + file_major_version: 2, + file_minor_version: 0, + file_size_bytes: CachedFileSize::new(500), + base_id: None, + }); + + // Add a data file with mixed tombstoned and valid fields + fragment.files.push(DataFile { + path: "mixed.lance".to_string(), + fields: Arc::from([4, -2, 5]), + column_indices: Arc::from([]), + file_major_version: 2, + file_minor_version: 0, + file_size_bytes: CachedFileSize::new(750), + base_id: None, + }); + + // Add another fully tombstoned file + fragment.files.push(DataFile { + path: "another_tombstoned.lance".to_string(), + fields: Arc::from([-2_i32]), + column_indices: Arc::from([]), + file_major_version: 2, + file_minor_version: 0, + file_size_bytes: CachedFileSize::new(250), + base_id: None, + }); + + let mut fragments = vec![fragment]; + + // Apply the cleanup + Transaction::remove_tombstoned_data_files(&mut fragments); + + // Should have removed the two fully tombstoned files + assert_eq!(fragments[0].files.len(), 2); + assert_eq!(fragments[0].files[0].path, "normal.lance"); + assert_eq!(fragments[0].files[1].path, "mixed.lance"); + } + + /// When a fragment has no existing last_updated_at_version_meta (None), a + /// partial RewriteColumns refresh must leave it as None rather than fabricating + /// prev_version for unmatched rows. + #[test] + fn test_partial_rewrite_skips_fragment_with_no_version_meta() { + let row_ids = RowIdSequence::from([10u64, 11, 12, 13, 14].as_slice()); + let row_id_meta = Some(RowIdMeta::Inline(write_row_ids(&row_ids))); + + let (major, minor) = lance_file::version::LanceFileVersion::Stable.to_numbers(); + let data_file = DataFile::new("data.lance", vec![0], vec![0], major, minor, None, None); + + let fragment = Fragment { + id: 1, + files: vec![data_file], + overlays: vec![], + deletion_file: None, + row_id_meta, + physical_rows: Some(5), + last_updated_at_version_meta: None, + created_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![fragment.clone()]); + + // Simulate a RewriteColumns update that matched offsets 1 and 3 + let off_map = HashMap::from([(1u64, RoaringBitmap::from_iter([1u32, 3]))]); + let tx = Transaction::new( + manifest.version, + Operation::Update { + removed_fragment_ids: vec![], + updated_fragments: vec![fragment], + new_fragments: vec![], + fields_modified: vec![], + compacted_sstables: vec![], + fields_for_preserving_frag_bitmap: vec![], + update_mode: Some(UpdateMode::RewriteColumns), + inserted_rows_filter: None, + updated_fragment_offsets: Some(UpdatedFragmentOffsets(off_map)), + }, + None, + ); + + let (out, _) = tx + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + assert!( + out.fragments[0].last_updated_at_version_meta.is_none(), + "fragment with no prior version metadata must not have fabricated prev_version stamped on unmatched rows" + ); + } + + #[test] + fn merge_build_manifest_refreshes_last_updated_when_data_files_change_stable_row_ids() { + use crate::feature_flags::FLAG_STABLE_ROW_IDS; + use lance_file::version::LanceFileVersion; + + let (major, minor) = LanceFileVersion::Stable.to_numbers(); + let mk_file = |path: &str| DataFile::new(path, vec![0], vec![0], major, minor, None, None); + + let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); + + let row_ids = RowIdSequence::from([100u64, 101, 102, 103, 104].as_slice()); + let row_id_meta = Some(RowIdMeta::Inline(write_row_ids(&row_ids))); + + let prev_fragment = Fragment { + id: 0, + files: vec![mk_file("before.lance")], + overlays: vec![], + deletion_file: None, + row_id_meta, + physical_rows: Some(5), + last_updated_at_version_meta: None, + created_at_version_meta: None, + }; + + let mut manifest = Manifest::new( + lance_schema.clone(), + Arc::new(vec![prev_fragment.clone()]), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + manifest.reader_feature_flags |= FLAG_STABLE_ROW_IDS; + manifest.next_row_id = 100; + + let merged_fragment = Fragment { + files: vec![mk_file("after.lance")], + ..prev_fragment + }; + + let tx = Transaction::new( + manifest.version, + Operation::Merge { + fragments: vec![merged_fragment], + schema: lance_schema, + }, + None, + ); + + let (out, _) = tx + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + assert_eq!(out.version, 2); + let frag = &out.fragments[0]; + let seq = frag + .last_updated_at_version_meta + .as_ref() + .unwrap() + .load_sequence() + .unwrap(); + assert_eq!(seq.version_at(0).unwrap(), 2); + assert_eq!(seq.version_at(4).unwrap(), 2); + } + + #[test] + fn merge_build_manifest_skips_refresh_when_carry_forward_stable_row_ids() { + use crate::feature_flags::FLAG_STABLE_ROW_IDS; + use crate::rowids::version::{RowDatasetVersionMeta, RowDatasetVersionSequence}; + use lance_file::version::LanceFileVersion; + + let (major, minor) = LanceFileVersion::Stable.to_numbers(); + let data_file = DataFile::new("same.lance", vec![0], vec![0], major, minor, None, None); + + let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); + + let row_ids = RowIdSequence::from([200u64, 201, 202, 203, 204].as_slice()); + let row_id_meta = Some(RowIdMeta::Inline(write_row_ids(&row_ids))); + + let uniform_v1 = RowDatasetVersionSequence::from_uniform_row_count(5, 1); + let meta_v1 = RowDatasetVersionMeta::from_sequence(&uniform_v1).unwrap(); + + let prev_fragment = Fragment { + id: 0, + files: vec![data_file.clone()], + overlays: vec![], + deletion_file: None, + row_id_meta: row_id_meta.clone(), + physical_rows: Some(5), + last_updated_at_version_meta: Some(meta_v1.clone()), + created_at_version_meta: None, + }; + + let mut manifest = Manifest::new( + lance_schema.clone(), + Arc::new(vec![prev_fragment]), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + manifest.reader_feature_flags |= FLAG_STABLE_ROW_IDS; + manifest.next_row_id = 100; + + let merged_fragment = Fragment { + id: 0, + files: vec![data_file], + overlays: vec![], + deletion_file: None, + row_id_meta, + physical_rows: Some(5), + last_updated_at_version_meta: Some(meta_v1), + created_at_version_meta: None, + }; + + let tx = Transaction::new( + manifest.version, + Operation::Merge { + fragments: vec![merged_fragment], + schema: lance_schema, + }, + None, + ); + + let (out, _) = tx + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + let seq = out.fragments[0] + .last_updated_at_version_meta + .as_ref() + .unwrap() + .load_sequence() + .unwrap(); + assert_eq!(seq.version_at(0).unwrap(), 1); + assert_eq!(seq.version_at(4).unwrap(), 1); + } + + #[test] + fn merge_build_manifest_no_last_updated_refresh_without_stable_row_ids() { + use crate::feature_flags::FLAG_STABLE_ROW_IDS; + use lance_file::version::LanceFileVersion; + + let (major, minor) = LanceFileVersion::Stable.to_numbers(); + let mk_file = |path: &str| DataFile::new(path, vec![0], vec![0], major, minor, None, None); + + let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); + + let prev_fragment = Fragment { + id: 0, + files: vec![mk_file("before.lance")], + overlays: vec![], + deletion_file: None, + row_id_meta: None, + physical_rows: Some(5), + last_updated_at_version_meta: None, + created_at_version_meta: None, + }; + + let manifest = Manifest::new( + lance_schema.clone(), + Arc::new(vec![prev_fragment.clone()]), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + assert_eq!( + manifest.reader_feature_flags & FLAG_STABLE_ROW_IDS, + 0, + "manifest must not use stable row IDs for this guard test" + ); + + let merged_fragment = Fragment { + files: vec![mk_file("after.lance")], + ..prev_fragment + }; + + let tx = Transaction::new( + manifest.version, + Operation::Merge { + fragments: vec![merged_fragment], + schema: lance_schema, + }, + None, + ); + + let (out, _) = tx + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + assert!( + out.fragments[0].last_updated_at_version_meta.is_none(), + "without stable row IDs, Merge must not populate per-row last_updated metadata" + ); + } + + #[test] + fn merge_build_manifest_sets_both_version_meta_for_new_fragment_id_stable_row_ids() { + use crate::feature_flags::FLAG_STABLE_ROW_IDS; + use lance_file::version::LanceFileVersion; + + let (major, minor) = LanceFileVersion::Stable.to_numbers(); + let mk_file = |path: &str| DataFile::new(path, vec![0], vec![0], major, minor, None, None); + + let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); + + // Existing fragment (id=0) with stable row IDs + let row_ids_0 = RowIdSequence::from([10u64, 11, 12].as_slice()); + let existing_fragment = Fragment { + id: 0, + files: vec![mk_file("existing.lance")], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&row_ids_0))), + physical_rows: Some(3), + last_updated_at_version_meta: None, + created_at_version_meta: None, + }; + + let mut manifest = Manifest::new( + lance_schema.clone(), + Arc::new(vec![existing_fragment.clone()]), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + manifest.reader_feature_flags |= FLAG_STABLE_ROW_IDS; + manifest.next_row_id = 100; + manifest.version = 1; + + // New fragment (id=1) not present in prev manifest — exercises the None branch + let row_ids_1 = RowIdSequence::from([20u64, 21, 22, 23].as_slice()); + let new_fragment = Fragment { + id: 1, + files: vec![mk_file("new.lance")], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&row_ids_1))), + physical_rows: Some(4), + last_updated_at_version_meta: None, + created_at_version_meta: None, + }; + + let tx = Transaction::new( + manifest.version, + Operation::Merge { + fragments: vec![existing_fragment, new_fragment], + schema: lance_schema, + }, + None, + ); + + let (out, _) = tx + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + assert_eq!(out.version, 2); + + let new_frag = out.fragments.iter().find(|f| f.id == 1).unwrap(); + + // last_updated_at_version must be set to the commit version + let last_updated_seq = new_frag + .last_updated_at_version_meta + .as_ref() + .expect("new fragment must have last_updated_at_version_meta") + .load_sequence() + .unwrap(); + assert_eq!(last_updated_seq.version_at(0).unwrap(), 2); + assert_eq!(last_updated_seq.version_at(3).unwrap(), 2); + + // created_at_version must also be set — must not be None + let created_seq = new_frag + .created_at_version_meta + .as_ref() + .expect("new fragment must have created_at_version_meta") + .load_sequence() + .unwrap(); + assert_eq!(created_seq.version_at(0).unwrap(), 2); + assert_eq!(created_seq.version_at(3).unwrap(), 2); + } + + // --- Proposal 1: range pre-filter --- + + // --- Proposal 2: version sequence cache --- + + #[test] + fn test_data_overlay_build_manifest_multi_fragment() { + // Overlays targeting two distinct fragments are each applied and stamped. + // A targeted fragment already carrying an overlay (committed at v3) gets + // the new overlay appended and stamped while its existing overlay is + // preserved, and a fragment the operation does not target is passed + // through with its existing overlays untouched. + let mut frag0 = Fragment::new(0); + frag0.overlays = vec![overlay_with_field(5, 3)]; // targeted, pre-existing at v3 + let frag1 = Fragment::new(1); + let mut frag2 = Fragment::new(2); + frag2.overlays = vec![overlay_with_field(9, 3)]; // untargeted, committed at v3 + let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let mut manifest = Manifest::new( + LanceSchema::try_from(&schema).unwrap(), + Arc::new(vec![frag0, frag1, frag2]), + crate::format::DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + // The pre-existing overlays were committed at v3, so the current + // manifest must be at least that version; the new commit then stamps + // its overlay at v4, keeping the fragment's overlays newest-last. + manifest.version = 3; + + let txn = Transaction::new( + manifest.version, + Operation::DataOverlay { + groups: vec![ + DataOverlayGroup { + fragment_id: 0, + overlays: vec![overlay_with_field(1, 0)], + }, + DataOverlayGroup { + fragment_id: 1, + overlays: vec![overlay_with_field(2, 0)], + }, + ], + }, + None, + ); + + let (result, _) = txn + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + let frag = |id: u64| { + result + .fragments + .iter() + .find(|f| f.id == id) + .unwrap_or_else(|| panic!("fragment {id} missing from result")) + }; + // The already-overlaid target keeps its v3 overlay and appends the new + // one, stamped to the new version. + assert_eq!(frag(0).overlays.len(), 2); + assert_eq!(frag(0).overlays[0].committed_version, 3); + assert_eq!(frag(0).overlays[1].committed_version, result.version); + // The fresh target gets its overlay, stamped to the new version. + assert_eq!(frag(1).overlays.len(), 1); + assert_eq!(frag(1).overlays[0].committed_version, result.version); + // The untargeted fragment is unchanged: same overlay, original version. + assert_eq!(frag(2).overlays.len(), 1); + assert_eq!(frag(2).overlays[0].committed_version, 3); + assert!(result.version > manifest.version); + } + + #[test] + fn test_data_replacement_tombstones_overlaid_fields() { + // A DataReplacement writing new base values for field 5 must stop any + // overlay from shadowing those cells: field 5 is tombstoned in place + // (preserving the overlay's field 3), and an overlay covering only field + // 5 is dropped entirely. + let mut fragment = Fragment::new(0); + fragment.files = vec![ + DataFile::new_legacy_from_fields("f3.lance", vec![3], None), + DataFile::new_legacy_from_fields("f5.lance", vec![5], None), + ]; + fragment.overlays = vec![ + DataOverlayFile { + data_file: DataFile::new_legacy_from_fields("o35.lance", vec![3, 5], None), + coverage: OverlayCoverage::sparse(vec![ + roaring::RoaringBitmap::from_iter([0u32]), + roaring::RoaringBitmap::from_iter([0u32]), + ]), + committed_version: 3, + }, + DataOverlayFile { + data_file: DataFile::new_legacy_from_fields("o5.lance", vec![5], None), + coverage: OverlayCoverage::dense(roaring::RoaringBitmap::from_iter([0u32])), + committed_version: 3, + }, + ]; + + let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let manifest = Manifest::new( + LanceSchema::try_from(&schema).unwrap(), + Arc::new(vec![fragment]), + crate::format::DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + + let txn = Transaction::new( + manifest.version, + Operation::DataReplacement { + replacements: vec![DataReplacementGroup( + 0, + DataFile::new_legacy_from_fields("f5-new.lance", vec![5], None), + )], + }, + None, + ); + + let (result, _) = txn + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + let frag = &result.fragments[0]; + // The base data file for field 5 was swapped in. + assert!(frag.files.iter().any(|f| f.path == "f5-new.lance")); + // The [3, 5] overlay keeps field 3 and tombstones field 5; the [5]-only + // overlay is dropped. + assert_eq!(frag.overlays.len(), 1); + assert_eq!(frag.overlays[0].data_file.fields.as_ref(), &[3, -2]); + } + + #[test] + fn test_data_overlay_build_manifest_merges_duplicate_groups() { + // Two groups targeting the same fragment must both survive (a HashMap + // collapse would have dropped the first). + let manifest = sample_manifest(); + let txn = Transaction::new( + manifest.version, + Operation::DataOverlay { + groups: vec![ + DataOverlayGroup { + fragment_id: 0, + overlays: vec![overlay_with_field(1, 0)], + }, + DataOverlayGroup { + fragment_id: 0, + overlays: vec![overlay_with_field(2, 0)], + }, + ], + }, + None, + ); + + let (result, _) = txn + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + let overlays = &result.fragments[0].overlays; + assert_eq!(overlays.len(), 2); + assert_eq!(overlays[0].data_file.fields.as_ref(), [1i32].as_slice()); + assert_eq!(overlays[1].data_file.fields.as_ref(), [2i32].as_slice()); + } + + #[test] + fn test_data_overlay_build_manifest_rejects_unknown_fragment() { + let manifest = sample_manifest(); + let txn = Transaction::new( + manifest.version, + Operation::DataOverlay { + groups: vec![DataOverlayGroup { + fragment_id: 99, + overlays: vec![overlay_with_field(1, 0)], + }], + }, + None, + ); + let err = txn + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap_err(); + assert!(err.to_string().contains("does not exist"), "{err}"); + } +} diff --git a/rust/lance-table/src/transaction/operation.rs b/rust/lance-table/src/transaction/operation.rs new file mode 100644 index 00000000000..e6bc9428916 --- /dev/null +++ b/rust/lance-table/src/transaction/operation.rs @@ -0,0 +1,314 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! The vocabulary of changes a transaction can describe. +//! +//! Each [`Operation`] variant names one kind of change and carries exactly the +//! inputs needed to apply it: the fragments to add, the fields that were +//! rewritten, the indices that were rebuilt. Applying them is +//! [`super::manifest_build`]; deciding whether two of them collide is +//! [`super::conflicts`]. + +use crate::format::key_existence::KeyExistenceFilter; +use crate::format::overlay::DataOverlayFile; +use crate::format::{BasePath, DataFile, Fragment, IndexFile, IndexMetadata}; +use crate::system_index::mem_wal::CompactedSsTable; +use crate::transaction::UpdateMap; +use crate::transaction::action::UserOperation; +use lance_core::datatypes::Schema; +use lance_core::deepsize::DeepSizeOf; +use roaring::RoaringBitmap; +use std::collections::HashMap; +use uuid::Uuid; + +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub struct DataReplacementGroup(pub u64, pub DataFile); + +/// Overlay files to append to a single fragment, in order (the last entry is +/// newest). The overlays are appended to the fragment's existing `overlays` +/// list rather than replacing it, so overlays written by concurrent commits are +/// preserved. Each overlay's `committed_version` is stamped to the new dataset +/// version at commit time (re-stamped on retry). +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub struct DataOverlayGroup { + pub fragment_id: u64, + pub overlays: Vec, +} + +/// An operation on a dataset. +#[derive(Debug, Clone, DeepSizeOf)] +pub enum Operation { + /// Adding new fragments to the dataset. The fragments contained within + /// haven't yet been assigned a final ID. + Append { fragments: Vec }, + /// Updated fragments contain those that have been modified with new deletion + /// files. The deleted fragment IDs are those that should be removed from + /// the manifest. + Delete { + updated_fragments: Vec, + deleted_fragment_ids: Vec, + predicate: String, + }, + /// Overwrite the entire dataset with the given fragments. This is also + /// used when initially creating a table. + Overwrite { + fragments: Vec, + schema: Schema, + config_upsert_values: Option>, + initial_bases: Option>, + }, + /// A new index has been created. + CreateIndex { + /// The new secondary indices, + /// any existing indices with the same name will be replaced. + new_indices: Vec, + /// The indices that have been modified. + removed_indices: Vec, + }, + /// Data is rewritten but *not* modified. This is used for things like + /// compaction or re-ordering. Contains the old fragments and the new + /// ones that have been replaced. + /// + /// This operation will modify the row addresses of existing rows and + /// so any existing index covering a rewritten fragment will need to be + /// remapped. + Rewrite { + /// Groups of fragments that have been modified + groups: Vec, + /// Indices that have been updated with the new row addresses + rewritten_indices: Vec, + /// The fragment reuse index to be created or updated to + frag_reuse_index: Option, + }, + /// Replace data in a column in the dataset with new data. This is used for + /// null column population where we replace an entirely null column with a + /// new column that has data. + /// + /// This operation will only allow replacing files that contain the same schema + /// e.g. if the original files contain columns A, B, C and the new files contain + /// only columns A, B then the operation is not allowed. As we would need to split + /// the original files into two files, one with column A, B and the other with column C. + /// + /// Corollary to the above: the operation will also not allow replacing files unless the + /// affected columns all have the same datafile layout across the fragments being replaced. + /// + /// e.g. if fragments being replaced contain files with different schema layouts on + /// the column being replaced, the operation is not allowed. + /// say `frag_1: [A] [B, C]` and `frag_2: [A, B] [C]` and we are trying to replace column A + /// with a new column A, the operation is not allowed. + DataReplacement { + replacements: Vec, + }, + /// Attach overlay files to fragments, supplying new values for a subset of + /// `(physical offset, field)` cells without rewriting the fragments' base + /// data files. See [`DataOverlayFile`] and the Data Overlay Files + /// specification for resolution, coverage, and versioning rules. + DataOverlay { groups: Vec }, + /// Merge a new column in + /// 'fragments' is the final fragments include all data files, the new fragments must align with old ones at rows. + /// 'schema' is not forced to include existed columns, which means we could use Merge to drop column data + Merge { + fragments: Vec, + schema: Schema, + }, + /// Restore an old version of the database + Restore { version: u64 }, + /// Reserves fragment ids for future use + /// This can be used when row ids need to be known before a transaction + /// has been committed. It is used during a rewrite operation to allow + /// indices to be remapped to the new row ids as part of the operation. + ReserveFragments { num_fragments: u32 }, + + /// Update values in the dataset. + /// + /// Updates are generally vertical or horizontal. + /// + /// A vertical update adds new rows. In this case, the updated_fragments + /// will only have existing rows deleted and will not have any new fields added. + /// All new data will be contained in new_fragments. + /// This is what is used by a merge_insert that matches the whole schema and what + /// is used by the dataset updater. + /// + /// A horizontal update adds new columns. In this case, the updated fragments + /// may have fields removed or added. It is even possible for a field to be tombstoned + /// and then added back in the same update. (which is a field modification). If any + /// fields are modified in this way then they need to be added to the fields_modified list. + /// This way we can correctly update the indices. + /// This is what is used by a merge insert that does not match the whole schema. + Update { + /// Ids of fragments that have been moved + removed_fragment_ids: Vec, + /// Fragments that have been updated + updated_fragments: Vec, + /// Fragments that have been added + new_fragments: Vec, + /// The fields that have been modified + fields_modified: Vec, + /// MemWAL SSTables to mark as compacted after this transaction. + compacted_sstables: Vec, + /// The fields that used to judge whether to preserve the new frag's id into + /// the frag bitmap of the specified indices. + fields_for_preserving_frag_bitmap: Vec, + /// The mode of update + update_mode: Option, + /// Optional filter for detecting conflicts on inserted row keys. + /// Only tracks keys from INSERT operations during merge insert, not updates. + inserted_rows_filter: Option, + /// Physical row offsets (per fragment) that matched `update_columns` for RewriteColumns. + /// `None` means callers did not supply offsets; `build_manifest` skips partial refresh then. + updated_fragment_offsets: Option, + }, + + /// Project to a new schema. This only changes the schema, not the data. + Project { schema: Schema }, + + /// Update the dataset configuration. + UpdateConfig { + config_updates: Option, + table_metadata_updates: Option, + schema_metadata_updates: Option, + field_metadata_updates: HashMap, + }, + /// Update SSTable compaction progress in the MemWAL index. + /// + /// This is used during merge-insert to atomically record which + /// SSTables have been compacted into the base table. + UpdateMemWalState { + compacted_sstables: Vec, + }, + + /// Clone a dataset. + Clone { + is_shallow: bool, + ref_name: Option, + ref_version: u64, + ref_path: String, + branch_name: Option, + }, + + // Update base paths in the dataset (currently only supports adding new bases). + UpdateBases { + /// The new base paths to add to the manifest. + new_bases: Vec, + }, + + /// A composable operation expressed as a list of manifest deltas. + /// + /// Unlike the variants above, which each describe one kind of change, this + /// carries an ordered list of [`Action`]s that commit atomically. See + /// [`crate::transaction::action`]. + UserOperation(UserOperation), +} + +#[derive(Debug, Clone, PartialEq, DeepSizeOf)] +pub enum UpdateMode { + /// rows are deleted in current fragments and rewritten in new fragments. + /// This is most optimal when the majority of columns are being rewritten + /// or only a few rows are being updated. + RewriteRows, + + /// within each fragment, columns are fully rewritten and inserted as new data files. + /// Old versions of columns are tombstoned. This is most optimal when most rows are affected + /// but a small subset of columns are affected. + RewriteColumns, +} + +/// Matched physical row offsets per fragment for a partial [`UpdateMode::RewriteColumns`] update. +/// +/// Used with stable row IDs so `build_manifest` can refresh row-level version +/// metadata only for rows that were rewritten. +#[derive(Debug, Clone, PartialEq, Eq, Default)] +pub struct UpdatedFragmentOffsets(pub HashMap); + +impl DeepSizeOf for UpdatedFragmentOffsets { + fn deep_size_of_children(&self, context: &mut lance_core::deepsize::Context) -> usize { + self.0.iter().fold(0_usize, |acc, (frag_id, bitmap)| { + acc + frag_id.deep_size_of_children(context) + + (bitmap.len() as usize).saturating_mul(std::mem::size_of::()) + }) + } +} + +impl std::fmt::Display for Operation { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Append { .. } => write!(f, "Append"), + Self::Delete { .. } => write!(f, "Delete"), + Self::Overwrite { .. } => write!(f, "Overwrite"), + Self::CreateIndex { .. } => write!(f, "CreateIndex"), + Self::Rewrite { .. } => write!(f, "Rewrite"), + Self::Merge { .. } => write!(f, "Merge"), + Self::Restore { .. } => write!(f, "Restore"), + Self::ReserveFragments { .. } => write!(f, "ReserveFragments"), + Self::Update { .. } => write!(f, "Update"), + Self::Project { .. } => write!(f, "Project"), + Self::UpdateConfig { .. } => write!(f, "UpdateConfig"), + Self::DataReplacement { .. } => write!(f, "DataReplacement"), + Self::DataOverlay { .. } => write!(f, "DataOverlay"), + Self::Clone { .. } => write!(f, "Clone"), + Self::UpdateMemWalState { .. } => write!(f, "UpdateMemWalState"), + Self::UpdateBases { .. } => write!(f, "UpdateBases"), + Self::UserOperation(op) => write!(f, "UserOperation({})", op.description), + } + } +} + +#[derive(Debug, Clone, PartialEq)] +pub struct RewrittenIndex { + pub old_id: Uuid, + pub new_id: Uuid, + pub new_index_details: prost_types::Any, + pub new_index_version: u32, + /// Files in the new index with their sizes. + /// Empty list from older writers that didn't persist this field. + pub new_index_files: Option>, +} + +impl DeepSizeOf for RewrittenIndex { + fn deep_size_of_children(&self, context: &mut lance_core::deepsize::Context) -> usize { + self.new_index_details + .type_url + .deep_size_of_children(context) + + self.new_index_details.value.deep_size_of_children(context) + } +} + +#[derive(Debug, Clone, DeepSizeOf)] +pub struct RewriteGroup { + pub old_fragments: Vec, + pub new_fragments: Vec, +} + +impl PartialEq for RewriteGroup { + fn eq(&self, other: &Self) -> bool { + fn compare_vec(a: &[T], b: &[T]) -> bool { + a.len() == b.len() && a.iter().all(|f| b.contains(f)) + } + compare_vec(&self.old_fragments, &other.old_fragments) + && compare_vec(&self.new_fragments, &other.new_fragments) + } +} + +impl Operation { + pub fn name(&self) -> &str { + match self { + Self::Append { .. } => "Append", + Self::Delete { .. } => "Delete", + Self::Overwrite { .. } => "Overwrite", + Self::CreateIndex { .. } => "CreateIndex", + Self::Rewrite { .. } => "Rewrite", + Self::Merge { .. } => "Merge", + Self::ReserveFragments { .. } => "ReserveFragments", + Self::Restore { .. } => "Restore", + Self::Update { .. } => "Update", + Self::Project { .. } => "Project", + Self::UpdateConfig { .. } => "UpdateConfig", + Self::DataReplacement { .. } => "DataReplacement", + Self::DataOverlay { .. } => "DataOverlay", + Self::UpdateMemWalState { .. } => "UpdateMemWalState", + Self::Clone { .. } => "Clone", + Self::UpdateBases { .. } => "UpdateBases", + Self::UserOperation { .. } => "UserOperation", + } + } +} diff --git a/rust/lance-table/src/transaction/proto.rs b/rust/lance-table/src/transaction/proto.rs new file mode 100644 index 00000000000..ffc6b6fb112 --- /dev/null +++ b/rust/lance-table/src/transaction/proto.rs @@ -0,0 +1,896 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Conversions between the transaction types and their protobuf encoding. +//! +//! A transaction is persisted as a `pb::Transaction` alongside the manifest it +//! produced, so these conversions are the format contract for everything in this +//! module: a field added to an `Operation` is only durable once it round-trips +//! here. +//! +//! One exception: the action vocabulary's wire code lives with the actions, in +//! [`super::action`] — the envelope in `action.rs` and each action's payload in its +//! own module, next to its logic and tests. + +use crate::format::key_existence::KeyExistenceFilter; +use crate::format::pb; +use crate::format::{BasePath, Fragment, IndexFile, IndexMetadata, overlay::DataOverlayFile}; +use crate::system_index::mem_wal::CompactedSsTable; +use crate::transaction::{ + DataOverlayGroup, DataReplacementGroup, Operation, RewriteGroup, RewrittenIndex, Transaction, + UpdateMap, UpdateMapEntry, UpdateMode, UpdatedFragmentOffsets, translate_config_updates, + translate_schema_metadata_updates, +}; +use lance_core::datatypes::Schema; +use lance_core::{Error, Result}; +use lance_file::datatypes::Fields; +use roaring::RoaringBitmap; +use std::collections::HashMap; +use std::sync::Arc; +use uuid::Uuid; + +impl From<&DataReplacementGroup> for pb::transaction::DataReplacementGroup { + fn from(DataReplacementGroup(fragment_id, new_file): &DataReplacementGroup) -> Self { + Self { + fragment_id: *fragment_id, + new_file: Some(new_file.into()), + } + } +} + +/// Convert a protobug DataReplacementGroup to a rust native DataReplacementGroup +/// this is unfortunately TryFrom instead of From because of the Option in the pb::DataReplacementGroup +impl TryFrom for DataReplacementGroup { + type Error = Error; + + fn try_from(message: pb::transaction::DataReplacementGroup) -> Result { + Ok(Self( + message.fragment_id, + message + .new_file + .ok_or(Error::invalid_input( + "DataReplacementGroup must have a new_file", + ))? + .try_into()?, + )) + } +} + +impl From<&DataOverlayGroup> for pb::transaction::DataOverlayGroup { + fn from(group: &DataOverlayGroup) -> Self { + Self { + fragment_id: group.fragment_id, + overlays: group + .overlays + .iter() + .map(pb::DataOverlayFile::from) + .collect(), + } + } +} + +impl TryFrom for DataOverlayGroup { + type Error = Error; + + fn try_from(message: pb::transaction::DataOverlayGroup) -> Result { + Ok(Self { + fragment_id: message.fragment_id, + overlays: message + .overlays + .into_iter() + .map(DataOverlayFile::try_from) + .collect::>>()?, + }) + } +} + +impl TryFrom for Transaction { + type Error = Error; + + fn try_from(message: pb::Transaction) -> Result { + let operation = match message.operation { + Some(pb::transaction::Operation::Append(pb::transaction::Append { fragments })) => { + Operation::Append { + fragments: fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + } + } + Some(pb::transaction::Operation::Clone(pb::transaction::Clone { + is_shallow, + ref_name, + ref_version, + ref_path, + branch_name, + })) => Operation::Clone { + is_shallow, + ref_name, + ref_version, + ref_path, + branch_name, + }, + Some(pb::transaction::Operation::Delete(pb::transaction::Delete { + updated_fragments, + deleted_fragment_ids, + predicate, + })) => Operation::Delete { + updated_fragments: updated_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + deleted_fragment_ids, + predicate, + }, + Some(pb::transaction::Operation::Overwrite(pb::transaction::Overwrite { + fragments, + schema, + schema_metadata: _schema_metadata, // TODO: handle metadata + config_upsert_values, + initial_bases, + })) => { + let config_upsert_option = if config_upsert_values.is_empty() { + None + } else { + Some(config_upsert_values) + }; + + Operation::Overwrite { + fragments: fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + schema: Schema::try_from(&Fields(schema))?, + config_upsert_values: config_upsert_option, + initial_bases: if initial_bases.is_empty() { + None + } else { + Some(initial_bases.into_iter().map(BasePath::from).collect()) + }, + } + } + Some(pb::transaction::Operation::ReserveFragments( + pb::transaction::ReserveFragments { num_fragments }, + )) => Operation::ReserveFragments { num_fragments }, + Some(pb::transaction::Operation::Rewrite(pb::transaction::Rewrite { + old_fragments, + new_fragments, + groups, + rewritten_indices, + })) => { + let groups = if !groups.is_empty() { + groups + .into_iter() + .map(RewriteGroup::try_from) + .collect::>()? + } else { + vec![RewriteGroup { + old_fragments: old_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + new_fragments: new_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + }] + }; + let rewritten_indices = rewritten_indices + .iter() + .map(RewrittenIndex::try_from) + .collect::>()?; + + Operation::Rewrite { + groups, + rewritten_indices, + frag_reuse_index: None, + } + } + Some(pb::transaction::Operation::CreateIndex(pb::transaction::CreateIndex { + new_indices, + removed_indices, + })) => Operation::CreateIndex { + new_indices: new_indices + .into_iter() + .map(IndexMetadata::try_from) + .collect::>()?, + removed_indices: removed_indices + .into_iter() + .map(IndexMetadata::try_from) + .collect::>()?, + }, + Some(pb::transaction::Operation::Merge(pb::transaction::Merge { + fragments, + schema, + schema_metadata: _schema_metadata, // TODO: handle metadata + })) => Operation::Merge { + fragments: fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + schema: Schema::try_from(&Fields(schema))?, + }, + Some(pb::transaction::Operation::Restore(pb::transaction::Restore { version })) => { + Operation::Restore { version } + } + Some(pb::transaction::Operation::Update(pb::transaction::Update { + removed_fragment_ids, + updated_fragments, + new_fragments, + fields_modified, + compacted_sstables, + fields_for_preserving_frag_bitmap, + update_mode, + inserted_rows, + updated_fragment_offsets, + })) => Operation::Update { + removed_fragment_ids, + updated_fragments: updated_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + new_fragments: new_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + fields_modified, + compacted_sstables: compacted_sstables + .into_iter() + .map(|m| CompactedSsTable::try_from(m).unwrap()) + .collect(), + fields_for_preserving_frag_bitmap, + update_mode: match update_mode { + 0 => Some(UpdateMode::RewriteRows), + 1 => Some(UpdateMode::RewriteColumns), + _ => Some(UpdateMode::RewriteRows), + }, + inserted_rows_filter: inserted_rows + .map(|ik| KeyExistenceFilter::try_from(&ik)) + .transpose()?, + updated_fragment_offsets: { + let m: HashMap = updated_fragment_offsets + .into_iter() + .filter(|(_, list)| !list.values.is_empty()) + .map(|(id, list)| (id, RoaringBitmap::from_iter(list.values))) + .collect(); + if m.is_empty() { + None + } else { + Some(UpdatedFragmentOffsets(m)) + } + }, + }, + Some(pb::transaction::Operation::Project(pb::transaction::Project { schema })) => { + Operation::Project { + schema: Schema::try_from(&Fields(schema))?, + } + } + Some(pb::transaction::Operation::UpdateConfig(update_config)) => { + // Check if new-style fields are present + let has_new_fields = update_config.config_updates.is_some() + || update_config.table_metadata_updates.is_some() + || update_config.schema_metadata_updates.is_some() + || !update_config.field_metadata_updates.is_empty(); + + // Check if old-style fields are present + let has_old_fields = !update_config.upsert_values.is_empty() + || !update_config.delete_keys.is_empty() + || !update_config.schema_metadata.is_empty() + || !update_config.field_metadata.is_empty(); + + // Error if both are present + if has_new_fields && has_old_fields { + return Err(Error::invalid_input_source( + "Cannot mix old and new style UpdateConfig fields".into(), + )); + } + + if has_old_fields { + // Translate old-style to new-style + let config_updates = if !update_config.upsert_values.is_empty() + || !update_config.delete_keys.is_empty() + { + Some(translate_config_updates( + &update_config.upsert_values, + &update_config.delete_keys, + )) + } else { + None + }; + + let schema_metadata_updates = if !update_config.schema_metadata.is_empty() { + Some(translate_schema_metadata_updates( + &update_config.schema_metadata, + )) + } else { + None + }; + + let field_metadata_updates = update_config + .field_metadata + .into_iter() + .map(|(field_id, field_meta_update)| { + ( + field_id as i32, + translate_schema_metadata_updates(&field_meta_update.metadata), + ) + }) + .collect(); + + Operation::UpdateConfig { + config_updates, + table_metadata_updates: None, + schema_metadata_updates, + field_metadata_updates, + } + } else { + // Use new-style fields directly (convert from protobuf) + Operation::UpdateConfig { + config_updates: update_config.config_updates.as_ref().map(UpdateMap::from), + table_metadata_updates: update_config + .table_metadata_updates + .as_ref() + .map(UpdateMap::from), + schema_metadata_updates: update_config + .schema_metadata_updates + .as_ref() + .map(UpdateMap::from), + field_metadata_updates: update_config + .field_metadata_updates + .iter() + .map(|(field_id, pb_update_map)| { + (*field_id, UpdateMap::from(pb_update_map)) + }) + .collect(), + } + } + } + Some(pb::transaction::Operation::DataReplacement( + pb::transaction::DataReplacement { replacements }, + )) => Operation::DataReplacement { + replacements: replacements + .into_iter() + .map(DataReplacementGroup::try_from) + .collect::>>()?, + }, + Some(pb::transaction::Operation::UpdateMemWalState( + pb::transaction::UpdateMemWalState { compacted_sstables }, + )) => Operation::UpdateMemWalState { + compacted_sstables: compacted_sstables + .into_iter() + .map(|m| CompactedSsTable::try_from(m).unwrap()) + .collect(), + }, + Some(pb::transaction::Operation::UpdateBases(pb::transaction::UpdateBases { + new_bases, + })) => Operation::UpdateBases { + new_bases: new_bases.into_iter().map(BasePath::from).collect(), + }, + Some(pb::transaction::Operation::DataOverlay(pb::transaction::DataOverlay { + groups, + })) => Operation::DataOverlay { + groups: groups + .into_iter() + .map(DataOverlayGroup::try_from) + .collect::>>()?, + }, + Some(pb::transaction::Operation::UserOperation(operation)) => { + // The fail-closed contract #7954 held at this level now lives one + // level down, in `Action`'s decode: an action this version does not + // recognize errors rather than being dropped. That is what matters, + // because `load_and_sort_new_transactions` collects concurrent + // transactions with `try_collect` — erroring aborts the in-flight + // commit, whereas a lenient parse would drop a concurrent action out + // of conflict detection and let two colliding commits both succeed. + Operation::UserOperation(operation.try_into()?) + } + None => { + return Err(Error::internal( + "Transaction message did not contain an operation".to_string(), + )); + } + }; + Ok(Self { + read_version: message.read_version, + uuid: message.uuid.clone(), + operation, + tag: if message.tag.is_empty() { + None + } else { + Some(message.tag.clone()) + }, + transaction_properties: if message.transaction_properties.is_empty() { + None + } else { + Some(Arc::new(message.transaction_properties)) + }, + }) + } +} + +impl TryFrom<&pb::transaction::rewrite::RewrittenIndex> for RewrittenIndex { + type Error = Error; + + fn try_from(message: &pb::transaction::rewrite::RewrittenIndex) -> Result { + Ok(Self { + old_id: message + .old_id + .as_ref() + .map(Uuid::try_from) + .ok_or_else(|| { + Error::invalid_input("required field (old_id) missing from message".to_string()) + })??, + new_id: message + .new_id + .as_ref() + .map(Uuid::try_from) + .ok_or_else(|| { + Error::invalid_input("required field (new_id) missing from message".to_string()) + })??, + new_index_details: message + .new_index_details + .as_ref() + .ok_or_else(|| { + Error::invalid_input("new_index_details is a required field".to_string()) + })? + .clone(), + new_index_version: message.new_index_version, + new_index_files: if message.new_index_files.is_empty() { + None + } else { + Some( + message + .new_index_files + .iter() + .map(|f| IndexFile { + path: f.path.clone(), + size_bytes: f.size_bytes, + }) + .collect(), + ) + }, + }) + } +} + +impl TryFrom for RewriteGroup { + type Error = Error; + + fn try_from(message: pb::transaction::rewrite::RewriteGroup) -> Result { + Ok(Self { + old_fragments: message + .old_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + new_fragments: message + .new_fragments + .into_iter() + .map(Fragment::try_from) + .collect::>>()?, + }) + } +} + +impl From<&Transaction> for pb::Transaction { + fn from(value: &Transaction) -> Self { + let operation = match &value.operation { + Operation::Append { fragments } => { + pb::transaction::Operation::Append(pb::transaction::Append { + fragments: fragments.iter().map(pb::DataFragment::from).collect(), + }) + } + Operation::Clone { + is_shallow, + ref_name, + ref_version, + ref_path, + branch_name, + } => pb::transaction::Operation::Clone(pb::transaction::Clone { + is_shallow: *is_shallow, + ref_name: ref_name.clone(), + ref_version: *ref_version, + ref_path: ref_path.clone(), + branch_name: branch_name.clone(), + }), + Operation::Delete { + updated_fragments, + deleted_fragment_ids, + predicate, + } => pb::transaction::Operation::Delete(pb::transaction::Delete { + updated_fragments: updated_fragments + .iter() + .map(pb::DataFragment::from) + .collect(), + deleted_fragment_ids: deleted_fragment_ids.clone(), + predicate: predicate.clone(), + }), + Operation::Overwrite { + fragments, + schema, + config_upsert_values, + initial_bases, + } => { + pb::transaction::Operation::Overwrite(pb::transaction::Overwrite { + fragments: fragments.iter().map(pb::DataFragment::from).collect(), + schema: Fields::from(schema).0, + schema_metadata: Default::default(), // TODO: handle metadata + config_upsert_values: config_upsert_values + .clone() + .unwrap_or(Default::default()), + initial_bases: initial_bases + .as_ref() + .map(|paths| { + paths + .iter() + .cloned() + .map(|bp: BasePath| -> pb::BasePath { bp.into() }) + .collect::>() + }) + .unwrap_or_default(), + }) + } + Operation::ReserveFragments { num_fragments } => { + pb::transaction::Operation::ReserveFragments(pb::transaction::ReserveFragments { + num_fragments: *num_fragments, + }) + } + Operation::Rewrite { + groups, + rewritten_indices, + frag_reuse_index: _, + } => pb::transaction::Operation::Rewrite(pb::transaction::Rewrite { + groups: groups + .iter() + .map(pb::transaction::rewrite::RewriteGroup::from) + .collect(), + rewritten_indices: rewritten_indices + .iter() + .map(|rewritten| rewritten.into()) + .collect(), + ..Default::default() + }), + Operation::CreateIndex { + new_indices, + removed_indices, + } => pb::transaction::Operation::CreateIndex(pb::transaction::CreateIndex { + new_indices: new_indices.iter().map(pb::IndexMetadata::from).collect(), + removed_indices: removed_indices + .iter() + .map(pb::IndexMetadata::from) + .collect(), + }), + Operation::Merge { fragments, schema } => { + pb::transaction::Operation::Merge(pb::transaction::Merge { + fragments: fragments.iter().map(pb::DataFragment::from).collect(), + schema: Fields::from(schema).0, + schema_metadata: Default::default(), // TODO: handle metadata + }) + } + Operation::Restore { version } => { + pb::transaction::Operation::Restore(pb::transaction::Restore { version: *version }) + } + Operation::Update { + removed_fragment_ids, + updated_fragments, + new_fragments, + fields_modified, + compacted_sstables, + fields_for_preserving_frag_bitmap, + update_mode, + inserted_rows_filter, + updated_fragment_offsets, + } => pb::transaction::Operation::Update(pb::transaction::Update { + removed_fragment_ids: removed_fragment_ids.clone(), + updated_fragments: updated_fragments + .iter() + .map(pb::DataFragment::from) + .collect(), + new_fragments: new_fragments.iter().map(pb::DataFragment::from).collect(), + fields_modified: fields_modified.clone(), + compacted_sstables: compacted_sstables + .iter() + .map(pb::CompactedSsTable::from) + .collect(), + fields_for_preserving_frag_bitmap: fields_for_preserving_frag_bitmap.clone(), + update_mode: update_mode + .as_ref() + .map(|mode| match mode { + UpdateMode::RewriteRows => 0, + UpdateMode::RewriteColumns => 1, + }) + .unwrap_or(0), + inserted_rows: inserted_rows_filter.as_ref().map(|ik| ik.into()), + updated_fragment_offsets: updated_fragment_offsets + .as_ref() + .map(|UpdatedFragmentOffsets(m)| { + m.iter() + .filter(|(_, b)| !b.is_empty()) + .map(|(frag_id, b)| { + let values: Vec = b.iter().collect(); + (*frag_id, pb::transaction::UInt32List { values }) + }) + .collect::>() + }) + .unwrap_or_default(), + }), + Operation::Project { schema } => { + pb::transaction::Operation::Project(pb::transaction::Project { + schema: Fields::from(schema).0, + }) + } + Operation::UpdateConfig { + config_updates, + table_metadata_updates, + schema_metadata_updates, + field_metadata_updates, + } => pb::transaction::Operation::UpdateConfig(pb::transaction::UpdateConfig { + config_updates: config_updates.as_ref().map(pb::UpdateMap::from), + table_metadata_updates: table_metadata_updates.as_ref().map(pb::UpdateMap::from), + schema_metadata_updates: schema_metadata_updates.as_ref().map(pb::UpdateMap::from), + field_metadata_updates: field_metadata_updates + .iter() + .map(|(field_id, update_map)| (*field_id, pb::UpdateMap::from(update_map))) + .collect(), + // Leave old fields empty - we only write new-style fields + upsert_values: Default::default(), + delete_keys: Default::default(), + schema_metadata: Default::default(), + field_metadata: Default::default(), + }), + Operation::DataReplacement { replacements } => { + pb::transaction::Operation::DataReplacement(pb::transaction::DataReplacement { + replacements: replacements + .iter() + .map(pb::transaction::DataReplacementGroup::from) + .collect(), + }) + } + Operation::DataOverlay { groups } => { + pb::transaction::Operation::DataOverlay(pb::transaction::DataOverlay { + groups: groups + .iter() + .map(pb::transaction::DataOverlayGroup::from) + .collect(), + }) + } + Operation::UpdateMemWalState { compacted_sstables } => { + pb::transaction::Operation::UpdateMemWalState(pb::transaction::UpdateMemWalState { + compacted_sstables: compacted_sstables + .iter() + .map(pb::CompactedSsTable::from) + .collect::>(), + }) + } + Operation::UpdateBases { new_bases } => { + pb::transaction::Operation::UpdateBases(pb::transaction::UpdateBases { + new_bases: new_bases + .iter() + .cloned() + .map(|bp: BasePath| -> pb::BasePath { bp.into() }) + .collect::>(), + }) + } + Operation::UserOperation(operation) => { + pb::transaction::Operation::UserOperation(operation.into()) + } + }; + + let transaction_properties = value + .transaction_properties + .as_ref() + .map(|arc| arc.as_ref().clone()) + .unwrap_or_default(); + Self { + read_version: value.read_version, + uuid: value.uuid.clone(), + operation: Some(operation), + tag: value.tag.clone().unwrap_or("".to_string()), + transaction_properties, + } + } +} + +impl From<&RewrittenIndex> for pb::transaction::rewrite::RewrittenIndex { + fn from(value: &RewrittenIndex) -> Self { + Self { + old_id: Some((&value.old_id).into()), + new_id: Some((&value.new_id).into()), + new_index_details: Some(value.new_index_details.clone()), + new_index_version: value.new_index_version, + new_index_files: value + .new_index_files + .as_ref() + .map(|files| { + files + .iter() + .map(|f| pb::IndexFile { + path: f.path.clone(), + size_bytes: f.size_bytes, + }) + .collect() + }) + .unwrap_or_default(), + } + } +} + +impl From<&RewriteGroup> for pb::transaction::rewrite::RewriteGroup { + fn from(value: &RewriteGroup) -> Self { + Self { + old_fragments: value + .old_fragments + .iter() + .map(pb::DataFragment::from) + .collect(), + new_fragments: value + .new_fragments + .iter() + .map(pb::DataFragment::from) + .collect(), + } + } +} + +impl From<&UpdateMap> for pb::UpdateMap { + fn from(update_map: &UpdateMap) -> Self { + Self { + update_entries: update_map + .update_entries + .iter() + .map(|entry| pb::UpdateMapEntry { + key: entry.key.clone(), + value: entry.value.clone(), + }) + .collect(), + replace: update_map.replace, + } + } +} + +impl From<&pb::UpdateMap> for UpdateMap { + fn from(pb_update_map: &pb::UpdateMap) -> Self { + Self { + update_entries: pb_update_map + .update_entries + .iter() + .map(|entry| UpdateMapEntry { + key: entry.key.clone(), + value: entry.value.clone(), + }) + .collect(), + replace: pb_update_map.replace, + } + } +} + +impl From<&Transaction> for crate::format::Transaction { + fn from(value: &Transaction) -> Self { + let pb_transaction: pb::Transaction = value.into(); + Self { + inner: pb_transaction, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::DataFile; + use crate::format::overlay::OverlayCoverage; + + #[test] + fn test_data_overlay_operation_roundtrips() { + // A DataOverlay operation survives the protobuf round-trip, preserving + // the target fragment, the overlay's coverage, and its committed_version. + let mut bitmap = roaring::RoaringBitmap::new(); + bitmap.insert(1); + bitmap.insert(4); + let overlay = DataOverlayFile { + data_file: DataFile::new_legacy_from_fields("overlay-0.lance", vec![3], None), + coverage: OverlayCoverage::dense(bitmap.clone()), + committed_version: 6, + }; + let pb_overlay = pb::DataOverlayFile::from(&overlay); + + let message = pb::Transaction { + read_version: 1, + uuid: Uuid::new_v4().to_string(), + operation: Some(pb::transaction::Operation::DataOverlay( + pb::transaction::DataOverlay { + groups: vec![pb::transaction::DataOverlayGroup { + fragment_id: 7, + overlays: vec![pb_overlay], + }], + }, + )), + ..Default::default() + }; + + let txn = Transaction::try_from(message).unwrap(); + match txn.operation { + Operation::DataOverlay { groups } => { + assert_eq!(groups.len(), 1); + assert_eq!(groups[0].fragment_id, 7); + assert_eq!(groups[0].overlays.len(), 1); + assert_eq!(groups[0].overlays[0].committed_version, 6); + assert_eq!( + *groups[0].overlays[0].coverage_for_field(0).unwrap(), + bitmap + ); + } + other => panic!("expected DataOverlay, got {other:?}"), + } + } + + #[test] + fn test_transaction_carrying_unsupported_action_is_rejected_on_load() { + // A `UserOperation` now decodes, but an action this version does not + // implement must still fail closed: never silently skipped, never leniently + // parsed, so that a concurrent V2 commit in the conflict window aborts an + // in-flight commit instead of vanishing from conflict detection. + let message = pb::Transaction { + read_version: 1, + uuid: Uuid::new_v4().to_string(), + operation: Some(pb::transaction::Operation::UserOperation( + pb::UserOperation { + description: "INSERT INTO t VALUES (1)".to_string(), + uuid: Uuid::new_v4().to_string(), + read_version: 1, + actions: vec![pb::UserAction { + description: "append batch".to_string(), + actions: vec![pb::Action { + action: Some(pb::action::Action::AddFragment(pb::AddFragment { + local: 0, + physical_rows: 1, + data_change: Some(true), + ..Default::default() + })), + }], + }], + }, + )), + ..Default::default() + }; + + let err = Transaction::try_from(message).unwrap_err(); + assert!( + matches!(err, Error::NotSupported { .. }), + "expected NotSupported, got: {err:?}" + ); + } + + #[test] + fn test_user_operation_transaction_roundtrips() { + let operation = Operation::UserOperation(crate::transaction::UserOperation { + description: "ALTER TABLE t ADD BASE".to_string(), + uuid: Uuid::new_v4().to_string(), + read_version: 4, + actions: vec![crate::transaction::UserAction { + description: "register bases".to_string(), + actions: vec![ + crate::transaction::Action::AddBase(crate::transaction::AddBase { + local: 0, + name: Some("warm".to_string()), + is_dataset_root: false, + path: "s3://bucket/warm".to_string(), + }), + crate::transaction::Action::AddBase(crate::transaction::AddBase { + local: 1, + name: None, + is_dataset_root: true, + path: "/local/cold".to_string(), + }), + ], + }], + }); + let transaction = Transaction::new(4, operation, None); + + let message = pb::Transaction::from(&transaction); + let decoded = Transaction::try_from(message).unwrap(); + + assert_eq!(decoded.operation, transaction.operation); + assert_eq!(decoded.read_version, transaction.read_version); + assert_eq!(decoded.uuid, transaction.uuid); + } +} diff --git a/rust/lance-table/src/transaction/row_version.rs b/rust/lance-table/src/transaction/row_version.rs new file mode 100644 index 00000000000..9df7b1dfe58 --- /dev/null +++ b/rust/lance-table/src/transaction/row_version.rs @@ -0,0 +1,1139 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Row ids and the per-row version metadata that travels with them. +//! +//! Under stable row ids each fragment carries two run-length encoded sequences: +//! `created_at_version`, stamped once when a row first appears, and +//! `last_updated_at_version`, refreshed whenever a row's values change. Keeping +//! `created_at` correct across an update means tracing each new row back to the +//! fragment and offset it came from, which is what most of this module does. + +use crate::format::{ + Fragment, RowDatasetVersionMeta, RowDatasetVersionRun, RowDatasetVersionSequence, RowIdMeta, +}; +use crate::rowids::segment::U64Segment; +use crate::rowids::version::build_version_meta; +use crate::rowids::{RowIdSequence, read_row_ids, write_row_ids}; +use crate::transaction::Transaction; +use lance_core::{Error, Result}; +use std::cmp::Ordering; +use std::collections::{HashMap, HashSet}; + +/// Fallback version for rows whose original creation version cannot be determined. +/// Version 1 is the initial dataset version in the Lance format. +const UNKNOWN_CREATED_AT_VERSION: u64 = 1; + +/// Look up the `created_at` version for a single UPDATE-branch row ID. +/// +/// Callers must only call this for row IDs that are confirmed to be present in +/// `row_id_to_source` (i.e. UPDATE branch rows whose source exists in an existing +/// fragment). INSERT branch rows (no source) must use `new_version` directly and +/// must not call this function. +/// +/// Uses `row_id_to_source` to find the originating fragment and row offset, then +/// performs a O(K) random-access lookup via [`RowDatasetVersionSequence::version_at`] +/// on the pre-decoded sequence in `version_cache` (keyed by fragment ID). +/// +/// Returns [`UNKNOWN_CREATED_AT_VERSION`] if the source fragment has no +/// `created_at_version_meta` (missing or failed to decode) or the offset is +/// out of range. +fn resolve_created_at_version( + row_id: u64, + row_id_to_source: &HashMap, + version_cache: &HashMap, +) -> u64 { + let Some((orig_frag, row_offset)) = row_id_to_source.get(&row_id) else { + return UNKNOWN_CREATED_AT_VERSION; + }; + let Some(seq) = version_cache.get(&orig_frag.id) else { + return UNKNOWN_CREATED_AT_VERSION; + }; + seq.version_at(*row_offset) + .unwrap_or(UNKNOWN_CREATED_AT_VERSION) +} + +/// For each new fragment produced by an update, set `created_at_version_meta` +/// (preserved from the original rows) and `last_updated_at_version_meta`. +pub(super) fn resolve_update_version_metadata( + existing_fragments: &[Fragment], + new_fragments: &mut [Fragment], + new_version: u64, +) -> Result<()> { + // Collect only the row IDs we actually need to resolve, those appearing in new_fragments + // with inline metadata. This bounds the lookup map to O(updated rows) instead of O(all dataset rows) + let needed_row_ids: HashSet = new_fragments + .iter() + .filter_map(|f| match &f.row_id_meta { + Some(RowIdMeta::Inline(data)) => read_row_ids(data).ok(), + _ => None, + }) + .flat_map(|seq| seq.iter().collect::>()) + .collect(); + + let mut row_id_to_source: HashMap = HashMap::new(); + + if !needed_row_ids.is_empty() { + // Compute the bounding range of the needed set once. Any fragment whose + // entire row-id range lies outside [needed_min, needed_max] cannot contain + // any needed ID and can be skipped before the inner per-row loop. + let needed_min = *needed_row_ids.iter().min().unwrap(); + let needed_max = *needed_row_ids.iter().max().unwrap(); + + // Stable row IDs must be globally unique among *live* rows, but after a rewrite-style + // update the same stable ID can appear twice in `existing_fragments`: once in an older + // fragment's inline `row_id_meta` at the original row offset (rows may be soft-deleted + // via a deletion vector) and again in a newer fragment holding rewritten data. For + // `created_at` we need the mapping from the original fragment/offset; that is always the + // first occurrence when fragments are processed in ascending `id` order. + let mut sorted_frags: Vec<&Fragment> = existing_fragments.iter().collect(); + sorted_frags.sort_by_key(|f| f.id); + for frag in sorted_frags { + if let Some(RowIdMeta::Inline(data)) = &frag.row_id_meta + && let Ok(seq) = read_row_ids(data) + { + // Range pre-filter: skip the per-row inner loop when the fragment's + // bounding row-id range has no overlap with [needed_min, needed_max]. + // row_id_range() returns None for empty sequences, which are also skipped. + // This is a conservative check (may produce false positives for sparse + // segments) but never skips a fragment that actually contains a needed ID. + if seq + .row_id_range() + .is_none_or(|r| *r.end() < needed_min || *r.start() > needed_max) + { + continue; + } + + for (offset, rid) in seq.iter().enumerate() { + if needed_row_ids.contains(&rid) { + row_id_to_source.entry(rid).or_insert((frag, offset)); + } + } + } + } + } + + // Pre-decode the `created_at` version sequence for each source fragment exactly + // once. Without this cache, resolve_created_at_version would call load_sequence() + // (a protobuf decode) for every single updated row, even when many rows originate + // from the same fragment. + let source_frag_ids: HashSet = row_id_to_source.values().map(|(f, _)| f.id).collect(); + let version_cache: HashMap = existing_fragments + .iter() + .filter(|f| source_frag_ids.contains(&f.id)) + .filter_map(|frag| { + let seq = frag + .created_at_version_meta + .as_ref()? + .load_sequence() + .ok()?; + Some((frag.id, seq)) + }) + .collect(); + + for fragment in new_fragments.iter_mut() { + let row_ids = match &fragment.row_id_meta { + Some(RowIdMeta::Inline(data)) => read_row_ids(data).ok(), + Some(RowIdMeta::External(_)) => { + log::warn!( + "Fragment {} has external row ID metadata; \ + version tracking will use defaults", + fragment.id, + ); + None + } + None => None, + }; + + if let Some(row_ids) = row_ids { + let physical_rows = fragment.physical_rows.unwrap_or(0); + let created_at_versions: Vec = row_ids + .iter() + .map(|rid| { + if row_id_to_source.contains_key(&rid) { + // UPDATE branch: stable row ID resolves to a source row in an + // existing fragment. Copy created_at from the original row so + // the row's first-appearance version is preserved across rewrites. + resolve_created_at_version(rid, &row_id_to_source, &version_cache) + } else { + // INSERT branch: stable row ID has no source in existing fragments + // (e.g. NOT MATCHED arm of MERGE INTO). The row first appears in + // this commit, so created_at equals the new commit version. + new_version + } + }) + .collect(); + debug_assert_eq!(created_at_versions.len(), physical_rows); + + let runs = encode_version_runs(&created_at_versions); + let created_at_seq = RowDatasetVersionSequence { runs }; + fragment.created_at_version_meta = Some( + RowDatasetVersionMeta::from_sequence(&created_at_seq).map_err(|e| { + Error::internal(format!( + "Failed to create created_at version metadata: {}", + e + )) + })?, + ); + + fragment.last_updated_at_version_meta = build_version_meta(fragment, new_version); + } else { + let version_meta = build_version_meta(fragment, new_version); + fragment.last_updated_at_version_meta = version_meta.clone(); + fragment.created_at_version_meta = version_meta; + } + } + Ok(()) +} + +/// Run-length encode a sequence of per-row versions into [`RowDatasetVersionRun`]s. +fn encode_version_runs(versions: &[u64]) -> Vec { + if versions.is_empty() { + return Vec::new(); + } + let mut runs = Vec::new(); + let mut current_version = versions[0]; + let mut run_start = 0u64; + for (i, &version) in versions.iter().enumerate().skip(1) { + if version != current_version { + runs.push(RowDatasetVersionRun { + span: U64Segment::Range(run_start..i as u64), + version: current_version, + }); + current_version = version; + run_start = i as u64; + } + } + runs.push(RowDatasetVersionRun { + span: U64Segment::Range(run_start..versions.len() as u64), + version: current_version, + }); + runs +} + +impl Transaction { + /// collect the pure(the num of row IDs are equal to the physical rows) "rewrite rows" updated fragment ids + pub(super) fn collect_pure_rewrite_row_update_frags_ids( + fragments: &[Fragment], + ) -> Result> { + let mut pure_update_frag_ids = Vec::new(); + + for fragment in fragments { + let physical_rows = fragment + .physical_rows + .ok_or_else(|| Error::internal("Fragment does not have physical rows"))? + as u64; + + if let Some(row_id_meta) = &fragment.row_id_meta { + let existing_row_count = match row_id_meta { + RowIdMeta::Inline(data) => { + let sequence = read_row_ids(data)?; + sequence.len() as u64 + } + _ => 0, + }; + + // only filter the fragments that match: all the rows have row id, + // which means it does not contain inserted rows in this fragment + if existing_row_count == physical_rows { + pure_update_frag_ids.push(fragment.id); + } + } + } + + Ok(pure_update_frag_ids) + } + + pub(super) fn assign_row_ids(next_row_id: &mut u64, fragments: &mut [Fragment]) -> Result<()> { + for fragment in fragments { + let physical_rows = fragment + .physical_rows + .ok_or_else(|| Error::internal("Fragment does not have physical rows"))? + as u64; + + if fragment.row_id_meta.is_some() { + // we may meet merge insert case, it only has partial row ids. + // so here, we need to check if the row ids match the physical rows + // if yes, continue + // if not, fill the remaining row ids to the physical rows, then update row_id_meta + + // Check if existing row IDs match the physical rows count + let existing_row_count = match &fragment.row_id_meta { + Some(RowIdMeta::Inline(data)) => { + // Parse the serialized row ID sequence to get the count + let sequence = read_row_ids(data)?; + sequence.len() as u64 + } + _ => 0, + }; + + match existing_row_count.cmp(&physical_rows) { + Ordering::Equal => { + // Row IDs already match physical rows, continue to next fragment + continue; + } + Ordering::Less => { + // Partial row IDs - need to fill the remaining ones + let remaining_rows = physical_rows - existing_row_count; + let new_row_ids = *next_row_id..(*next_row_id + remaining_rows); + + // Merge existing and new row IDs + let combined_sequence = match &fragment.row_id_meta { + Some(RowIdMeta::Inline(data)) => read_row_ids(data)?, + _ => { + return Err(Error::internal( + "Failed to deserialize existing row ID sequence", + )); + } + }; + + let mut row_ids: Vec = combined_sequence.iter().collect(); + for row_id in new_row_ids { + row_ids.push(row_id); + } + let combined_sequence = RowIdSequence::from(row_ids.as_slice()); + + let serialized = write_row_ids(&combined_sequence); + fragment.row_id_meta = Some(RowIdMeta::Inline(serialized)); + *next_row_id += remaining_rows; + } + Ordering::Greater => { + // More row IDs than physical rows - this shouldn't happen + return Err(Error::internal(format!( + "Fragment has more row IDs ({}) than physical rows ({})", + existing_row_count, physical_rows + ))); + } + } + } else { + let row_ids = *next_row_id..(*next_row_id + physical_rows); + let sequence = RowIdSequence::from(row_ids); + // TODO: write to a separate file if large. Possibly share a file with other fragments. + let serialized = write_row_ids(&sequence); + fragment.row_id_meta = Some(RowIdMeta::Inline(serialized)); + *next_row_id += physical_rows; + } + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::transaction::test_support::{ + created_at_versions, default_build_config, last_updated_at_versions, + make_stable_row_id_manifest, update_txn, + }; + use std::sync::Arc; + + #[test] + fn test_assign_row_ids_new_fragment() { + // Test assigning row IDs to a fragment without existing row IDs + let mut fragments = vec![Fragment { + id: 1, + physical_rows: Some(100), + row_id_meta: None, + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }]; + let mut next_row_id = 0; + + Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); + + assert_eq!(next_row_id, 100); + assert!(fragments[0].row_id_meta.is_some()); + + if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { + let sequence = read_row_ids(data).unwrap(); + assert_eq!(sequence.len(), 100); + let row_ids: Vec = sequence.iter().collect(); + assert_eq!(row_ids, (0..100).collect::>()); + } else { + panic!("Expected inline row ID metadata"); + } + } + + #[test] + fn test_assign_row_ids_existing_complete() { + // Test with fragment that already has complete row IDs + let existing_sequence = RowIdSequence::from(0..50); + let serialized = write_row_ids(&existing_sequence); + + let mut fragments = vec![Fragment { + id: 1, + physical_rows: Some(50), + row_id_meta: Some(RowIdMeta::Inline(serialized)), + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }]; + let mut next_row_id = 100; + + Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); + + // next_row_id should not change + assert_eq!(next_row_id, 100); + + if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { + let sequence = read_row_ids(data).unwrap(); + assert_eq!(sequence.len(), 50); + let row_ids: Vec = sequence.iter().collect(); + assert_eq!(row_ids, (0..50).collect::>()); + } else { + panic!("Expected inline row ID metadata"); + } + } + + #[test] + fn test_assign_row_ids_partial_existing() { + // Test with fragment that has partial row IDs (merge insert case) + let existing_sequence = RowIdSequence::from(0..30); + let serialized = write_row_ids(&existing_sequence); + + let mut fragments = vec![Fragment { + id: 1, + physical_rows: Some(50), // More physical rows than existing row IDs + row_id_meta: Some(RowIdMeta::Inline(serialized)), + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }]; + let mut next_row_id = 100; + + Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); + + // next_row_id should advance by 20 (50 - 30) + assert_eq!(next_row_id, 120); + + if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { + let sequence = read_row_ids(data).unwrap(); + assert_eq!(sequence.len(), 50); + let row_ids: Vec = sequence.iter().collect(); + // Should contain original 0-29 plus new 100-119 + let mut expected = (0..30).collect::>(); + expected.extend(100..120); + assert_eq!(row_ids, expected); + } else { + panic!("Expected inline row ID metadata"); + } + } + + #[test] + fn test_assign_row_ids_excess_row_ids() { + // Test error case where fragment has more row IDs than physical rows + let existing_sequence = RowIdSequence::from(0..60); + let serialized = write_row_ids(&existing_sequence); + + let mut fragments = vec![Fragment { + id: 1, + physical_rows: Some(50), // Less physical rows than existing row IDs + row_id_meta: Some(RowIdMeta::Inline(serialized)), + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }]; + let mut next_row_id = 100; + + let result = Transaction::assign_row_ids(&mut next_row_id, &mut fragments); + + assert!(result.is_err()); + if let Err(Error::Internal { message, .. }) = result { + assert!(message.contains("more row IDs (60) than physical rows (50)")); + } else { + panic!("Expected Internal error about excess row IDs"); + } + } + + #[test] + fn test_assign_row_ids_multiple_fragments() { + // Test with multiple fragments, some with existing row IDs, some without + let existing_sequence = RowIdSequence::from(500..520); + let serialized = write_row_ids(&existing_sequence); + + let mut fragments = vec![ + Fragment { + id: 1, + physical_rows: Some(30), // No existing row IDs + row_id_meta: None, + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }, + Fragment { + id: 2, + physical_rows: Some(25), // Partial existing row IDs + row_id_meta: Some(RowIdMeta::Inline(serialized)), + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }, + ]; + let mut next_row_id = 1000; + + Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); + + // Should advance by 30 (first fragment) + 5 (second fragment partial) + assert_eq!(next_row_id, 1035); + + // Check first fragment + if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { + let sequence = read_row_ids(data).unwrap(); + assert_eq!(sequence.len(), 30); + let row_ids: Vec = sequence.iter().collect(); + assert_eq!(row_ids, (1000..1030).collect::>()); + } else { + panic!("Expected inline row ID metadata for first fragment"); + } + + // Check second fragment + if let Some(RowIdMeta::Inline(data)) = &fragments[1].row_id_meta { + let sequence = read_row_ids(data).unwrap(); + assert_eq!(sequence.len(), 25); + let row_ids: Vec = sequence.iter().collect(); + // Should contain original 500-519 plus new 1030-1034 + let mut expected = (500..520).collect::>(); + expected.extend(1030..1035); + assert_eq!(row_ids, expected); + } else { + panic!("Expected inline row ID metadata for second fragment"); + } + } + + #[test] + fn test_assign_row_ids_missing_physical_rows() { + // Test error case where fragment doesn't have physical_rows set + let mut fragments = vec![Fragment { + id: 1, + physical_rows: None, + row_id_meta: None, + files: vec![], + overlays: vec![], + deletion_file: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }]; + let mut next_row_id = 0; + + let result = Transaction::assign_row_ids(&mut next_row_id, &mut fragments); + + assert!(result.is_err()); + if let Err(Error::Internal { message, .. }) = result { + assert!(message.contains("Fragment does not have physical rows")); + } else { + panic!("Expected Internal error about missing physical rows"); + } + } + + #[test] + fn test_update_version_tracking_preserves_created_at() { + let existing_seq = RowIdSequence::from([100u64, 101, 102].as_slice()); + let created_at_seq = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..3), + version: 5, + }], + }; + let existing_fragment = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), + physical_rows: Some(3), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&created_at_seq).unwrap(), + ), + last_updated_at_version_meta: None, + }; + + let new_seq = RowIdSequence::from([100u64, 102].as_slice()); + let new_fragment = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![existing_fragment]); + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + assert_eq!(created_at_versions(&result, 10), vec![5, 5]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); + } + + #[test] + fn test_update_version_tracking_mixed_origins() { + let frag_a_seq = RowIdSequence::from([10u64, 11].as_slice()); + let frag_a_created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..2), + version: 2, + }], + }; + let frag_b_seq = RowIdSequence::from([20u64, 21, 22].as_slice()); + let frag_b_created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..3), + version: 3, + }], + }; + + let manifest = make_stable_row_id_manifest(vec![ + Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&frag_a_seq))), + physical_rows: Some(2), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&frag_a_created).unwrap(), + ), + last_updated_at_version_meta: None, + }, + Fragment { + id: 2, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&frag_b_seq))), + physical_rows: Some(3), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&frag_b_created).unwrap(), + ), + last_updated_at_version_meta: None, + }, + ]); + + // New fragment has rows from both original fragments: row 11 from frag_a, row 20 from frag_b + let new_seq = RowIdSequence::from([11u64, 20].as_slice()); + let new_fragment = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Row 11 came from frag_a (offset 1, version 2), row 20 came from frag_b (offset 0, version 3) + assert_eq!(created_at_versions(&result, 10), vec![2, 3]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); + } + + #[test] + fn test_update_version_tracking_insert_branch_gets_new_version() { + // Simulates the INSERT branch (NOT MATCHED) of a MERGE INTO commit: + // the new fragment contains a mix of rewritten rows (UPDATE branch, row ID + // present in existing fragments) and freshly inserted rows (INSERT branch, + // row ID not present in any existing fragment). + // + // UPDATE branch row (10): created_at must be copied from the source fragment. + // INSERT branch row (999): created_at must equal new_version (the merge commit + // version), because the row first appeared in this commit. + let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); + let existing_created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..2), + version: 5, + }], + }; + let existing_fragment = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), + physical_rows: Some(2), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&existing_created).unwrap(), + ), + last_updated_at_version_meta: None, + }; + + // New fragment has row 10 (UPDATE branch) and row 999 (INSERT branch) + let new_seq = RowIdSequence::from([10u64, 999].as_slice()); + let new_fragment = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + // update_txn uses read_version 4 → new_version is 5 + let manifest = make_stable_row_id_manifest(vec![existing_fragment]); + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Row 10 (UPDATE branch): created_at copied from source (version 5). + // Row 999 (INSERT branch): created_at == new_version (5). + assert_eq!(created_at_versions(&result, 10), vec![5, 5]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); + } + + #[test] + fn test_update_version_tracking_merge_into_distinguishes_insert_and_update_branch() { + // Verifies the MERGE INTO correctness contract when UPDATE branch rows and INSERT + // branch rows have *different* source created_at values, so we can distinguish + // which row got which value. + // + // Existing fragment (id=1): row IDs [10, 11], created_at = version 3. + // New fragment (id=20): row IDs [10, 500, 11, 501]. + // - Rows 10 and 11: UPDATE branch (present in existing fragment) → created_at = 3. + // - Rows 500 and 501: INSERT branch (no source) → created_at = new_version = 5. + let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); + let existing_created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..2), + version: 3, + }], + }; + let existing_fragment = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), + physical_rows: Some(2), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&existing_created).unwrap(), + ), + last_updated_at_version_meta: None, + }; + + let new_seq = RowIdSequence::from([10u64, 500, 11, 501].as_slice()); + let new_fragment = Fragment { + id: 20, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(4), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + // update_txn uses read_version 4 → new_version is 5 + let manifest = make_stable_row_id_manifest(vec![existing_fragment]); + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // UPDATE branch rows (10, 11): created_at preserved from source (version 3). + // INSERT branch rows (500, 501): created_at == new_version (5). + assert_eq!(created_at_versions(&result, 20), vec![3, 5, 3, 5]); + // All rows in the new fragment get last_updated == new_version. + assert_eq!(last_updated_at_versions(&result, 20), vec![5, 5, 5, 5]); + } + + #[test] + fn test_update_version_tracking_source_fragment_no_created_at_defaults_to_1() { + // Source fragment has row_id_meta but no created_at_version_meta. + // The row IS found in the lookup, but the version defaults to 1. + let existing_seq = RowIdSequence::from([50u64, 51].as_slice()); + let existing_fragment = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let new_seq = RowIdSequence::from([50u64].as_slice()); + let new_fragment = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(1), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![existing_fragment]); + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Row 50 is found in source but source has no created_at_version_meta → default 1 + assert_eq!(created_at_versions(&result, 10), vec![1]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5]); + } + + #[test] + fn test_update_version_tracking_no_row_id_meta_fallback() { + let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); + let existing_fragment = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let new_fragment = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: None, + physical_rows: Some(3), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![existing_fragment]); + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Fragment starts with no row_id_meta → assign_row_ids gives it fresh IDs → + // those IDs have no source in existing fragments (INSERT branch) → + // created_at == new_version (5) for each row. + assert_eq!(created_at_versions(&result, 10), vec![5, 5, 5]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5, 5]); + } + + #[test] + fn test_update_version_tracking_corrupt_created_at_defaults_to_1() { + let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); + let existing_fragment = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), + physical_rows: Some(2), + created_at_version_meta: Some(RowDatasetVersionMeta::Inline(Arc::from( + vec![0xFFu8; 8].as_slice(), + ))), + last_updated_at_version_meta: None, + }; + + let new_seq = RowIdSequence::from([10u64].as_slice()); + let new_fragment = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(1), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![existing_fragment]); + let (result, _) = update_txn(vec![new_fragment]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Corrupt metadata causes decode to fail → falls back to UNKNOWN_CREATED_AT_VERSION (1) + assert_eq!(created_at_versions(&result, 10), vec![1]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5]); + } + + /// Fragments whose row-ID range lies entirely outside the needed set must not + /// affect the result. Here fragment 1 has IDs [1000, 1001] which are far above + /// the needed range [10, 11]; it is skipped by the range pre-filter and its + /// created_at version (version 99) must never appear in the output. + #[test] + fn test_update_version_tracking_range_filter_skips_non_overlapping_fragment() { + // Fragment in range – IDs [10, 11], created_at = 5 + let in_range_seq = RowIdSequence::from([10u64, 11].as_slice()); + let in_range_created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..2), + version: 5, + }], + }; + let in_range_frag = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&in_range_seq))), + physical_rows: Some(2), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&in_range_created).unwrap(), + ), + last_updated_at_version_meta: None, + }; + + // Fragment outside range – IDs [1000, 1001], created_at = 99 (must never appear) + let out_of_range_seq = RowIdSequence::from([1000u64, 1001].as_slice()); + let out_of_range_created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..2), + version: 99, + }], + }; + let out_of_range_frag = Fragment { + id: 2, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&out_of_range_seq))), + physical_rows: Some(2), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&out_of_range_created).unwrap(), + ), + last_updated_at_version_meta: None, + }; + + // New fragment rewrites both rows from the in-range fragment + let new_seq = RowIdSequence::from([10u64, 11].as_slice()); + let new_frag = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![in_range_frag, out_of_range_frag]); + let (result, _) = update_txn(vec![new_frag]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Both rows originate from the in-range fragment (version 5). + // The out-of-range fragment's version 99 must not appear. + assert_eq!(created_at_versions(&result, 10), vec![5, 5]); + assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); + } + + /// When the needed row IDs fall exactly at the boundary of a fragment's range, + /// the range pre-filter must NOT skip the fragment (boundary values are inclusive). + #[test] + fn test_update_version_tracking_range_filter_boundary_inclusive() { + // Fragment IDs [10, 11, 12], created_at = 7 + let seq = RowIdSequence::from([10u64, 11, 12].as_slice()); + let created = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..3), + version: 7, + }], + }; + let existing = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&seq))), + physical_rows: Some(3), + created_at_version_meta: Some(RowDatasetVersionMeta::from_sequence(&created).unwrap()), + last_updated_at_version_meta: None, + }; + + // New fragment takes the boundary IDs: 10 (min) and 12 (max) + let new_seq = RowIdSequence::from([10u64, 12].as_slice()); + let new_frag = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![existing]); + let (result, _) = update_txn(vec![new_frag]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Boundary IDs must be found and resolved correctly + assert_eq!(created_at_versions(&result, 10), vec![7, 7]); + } + + /// When multiple updated rows all originate from the same source fragment, + /// the created_at version sequence for that fragment must be decoded exactly + /// once (not once per row). The observable correctness requirement is that + /// all rows get the right version regardless of how many there are. + #[test] + fn test_update_version_tracking_many_rows_same_source_fragment() { + // Source fragment: 100 rows with IDs 0..100, mixed versions (2 runs). + // First 50 rows at version 3, next 50 rows at version 4. + let src_ids: Vec = (0u64..100).collect(); + let src_seq = RowIdSequence::from(src_ids.as_slice()); + let src_created = RowDatasetVersionSequence { + runs: vec![ + RowDatasetVersionRun { + span: U64Segment::Range(0..50), + version: 3, + }, + RowDatasetVersionRun { + span: U64Segment::Range(0..50), + version: 4, + }, + ], + }; + let src_frag = Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&src_seq))), + physical_rows: Some(100), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&src_created).unwrap(), + ), + last_updated_at_version_meta: None, + }; + + // New fragment rewrites all 100 rows preserving their stable IDs. + let new_seq = RowIdSequence::from(src_ids.as_slice()); + let new_frag = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(100), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let manifest = make_stable_row_id_manifest(vec![src_frag]); + let (result, _) = update_txn(vec![new_frag]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + let versions = created_at_versions(&result, 10); + assert_eq!(versions.len(), 100); + // First 50 rows came from version 3, next 50 from version 4 + assert!(versions[..50].iter().all(|&v| v == 3)); + assert!(versions[50..].iter().all(|&v| v == 4)); + } + + /// Rows originating from multiple distinct source fragments must each get + /// the version from their own source, even when all cached together. + #[test] + fn test_update_version_tracking_cache_multiple_source_fragments() { + let seq_a = RowIdSequence::from([10u64, 11, 12].as_slice()); + let created_a = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..3), + version: 2, + }], + }; + let seq_b = RowIdSequence::from([20u64, 21, 22].as_slice()); + let created_b = RowDatasetVersionSequence { + runs: vec![RowDatasetVersionRun { + span: U64Segment::Range(0..3), + version: 8, + }], + }; + + let manifest = make_stable_row_id_manifest(vec![ + Fragment { + id: 1, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&seq_a))), + physical_rows: Some(3), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&created_a).unwrap(), + ), + last_updated_at_version_meta: None, + }, + Fragment { + id: 2, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&seq_b))), + physical_rows: Some(3), + created_at_version_meta: Some( + RowDatasetVersionMeta::from_sequence(&created_b).unwrap(), + ), + last_updated_at_version_meta: None, + }, + ]); + + // New fragment takes rows from both sources: 12 (frag A, offset 2) and 20 (frag B, offset 0) + let new_seq = RowIdSequence::from([12u64, 20].as_slice()); + let new_frag = Fragment { + id: 10, + files: vec![], + overlays: vec![], + deletion_file: None, + row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), + physical_rows: Some(2), + created_at_version_meta: None, + last_updated_at_version_meta: None, + }; + + let (result, _) = update_txn(vec![new_frag]) + .build_manifest(Some(&manifest), vec![], "txn", &default_build_config()) + .unwrap(); + + // Row 12 → frag A offset 2 → version 2; row 20 → frag B offset 0 → version 8 + assert_eq!(created_at_versions(&result, 10), vec![2, 8]); + } + + #[test] + fn test_encode_version_runs_empty() { + let runs = encode_version_runs(&[]); + assert!(runs.is_empty()); + } + + #[test] + fn test_encode_version_runs_single_run() { + let runs = encode_version_runs(&[3, 3, 3]); + assert_eq!(runs.len(), 1); + assert_eq!(runs[0].version, 3); + } + + #[test] + fn test_encode_version_runs_alternating() { + let runs = encode_version_runs(&[1, 2, 1, 2]); + assert_eq!(runs.len(), 4); + assert_eq!(runs[0].version, 1); + assert_eq!(runs[1].version, 2); + assert_eq!(runs[2].version, 1); + assert_eq!(runs[3].version, 2); + } +} diff --git a/rust/lance-table/src/transaction/test_support.rs b/rust/lance-table/src/transaction/test_support.rs new file mode 100644 index 00000000000..11c6fc955a6 --- /dev/null +++ b/rust/lance-table/src/transaction/test_support.rs @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Fixtures shared between the tests of several submodules. + +use crate::feature_flags::FLAG_STABLE_ROW_IDS; +use crate::format::overlay::{DataOverlayFile, OverlayCoverage}; +use crate::format::{ + DataFile, DataStorageFormat, Fragment, IndexMetadata, Manifest, ManifestBuildConfig, +}; +use crate::transaction::{Operation, Transaction}; +use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; +use chrono::Utc; +use lance_core::datatypes::Schema as LanceSchema; +use lance_file::version::LanceFileVersion; +use std::collections::HashMap; +use std::sync::Arc; +use uuid::Uuid; + +/// The build config that `lance`'s `ManifestWriteConfig::default()` resolves to. +pub fn default_build_config() -> ManifestBuildConfig { + ManifestBuildConfig { + auto_set_feature_flags: true, + timestamp_nanos: std::time::SystemTime::now() + .duration_since(std::time::SystemTime::UNIX_EPOCH) + .unwrap() + .as_nanos(), + use_stable_row_ids: false, + use_legacy_format: None, + storage_format: None, + disable_transaction_file: false, + } +} + +pub fn sample_manifest() -> Manifest { + let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + Manifest::new( + LanceSchema::try_from(&schema).unwrap(), + Arc::new(vec![Fragment::new(0)]), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ) +} + +pub fn sample_index_metadata(name: &str) -> IndexMetadata { + IndexMetadata { + uuid: Uuid::new_v4(), + fields: vec![0], + name: name.to_string(), + dataset_version: 0, + fragment_bitmap: Some([0].into_iter().collect()), + index_details: None, + index_version: 1, + created_at: Some(Utc::now()), + base_id: None, + files: None, + } +} + +pub fn overlay_with_field(field: i32, committed_version: u64) -> DataOverlayFile { + DataOverlayFile { + data_file: DataFile::new_legacy_from_fields("o.lance", vec![field], None), + coverage: OverlayCoverage::dense(roaring::RoaringBitmap::from_iter([0u32])), + committed_version, + } +} + +/// Existing fragments use id >= 1 to avoid collision with `Fragment::new(0)` +/// used by `sample_manifest`. New (updated) fragments use id = 10. +pub fn make_stable_row_id_manifest(fragments: Vec) -> Manifest { + let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); + let mut manifest = Manifest::new( + LanceSchema::try_from(&schema).unwrap(), + Arc::new(fragments), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + manifest.reader_feature_flags = FLAG_STABLE_ROW_IDS; + manifest.next_row_id = 1000; + manifest.version = 4; + manifest +} + +pub fn update_txn(new_fragments: Vec) -> Transaction { + Transaction::new( + 4, + Operation::Update { + removed_fragment_ids: vec![], + updated_fragments: vec![], + new_fragments, + fields_modified: vec![], + compacted_sstables: vec![], + fields_for_preserving_frag_bitmap: vec![], + update_mode: None, + inserted_rows_filter: None, + updated_fragment_offsets: None, + }, + None, + ) +} + +pub fn created_at_versions(manifest: &Manifest, frag_id: u64) -> Vec { + let frag = manifest.fragments.iter().find(|f| f.id == frag_id).unwrap(); + let seq = frag + .created_at_version_meta + .as_ref() + .unwrap() + .load_sequence() + .unwrap(); + seq.versions().collect() +} + +pub fn last_updated_at_versions(manifest: &Manifest, frag_id: u64) -> Vec { + let frag = manifest.fragments.iter().find(|f| f.id == frag_id).unwrap(); + let seq = frag + .last_updated_at_version_meta + .as_ref() + .unwrap() + .load_sequence() + .unwrap(); + seq.versions().collect() +} diff --git a/rust/lance-table/src/transaction/update_map.rs b/rust/lance-table/src/transaction/update_map.rs new file mode 100644 index 00000000000..b7d09609bc4 --- /dev/null +++ b/rust/lance-table/src/transaction/update_map.rs @@ -0,0 +1,128 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Incremental edits to the string maps a manifest carries. +//! +//! Dataset config, table metadata, schema metadata and per-field metadata are all +//! `HashMap`, and all four are updated the same way: a list of +//! entries where a `None` value means delete the key, plus a flag choosing between +//! merging into the existing map and replacing it outright. + +use lance_core::deepsize::DeepSizeOf; + +/// An entry for a map update. If value is None, the key will be removed from the map. +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub struct UpdateMapEntry { + /// The key of the map entry to update. + pub key: String, + /// The value to set for the key. + pub value: Option, +} + +impl From<(String, Option)> for UpdateMapEntry { + fn from((key, value): (String, Option)) -> Self { + Self { key, value } + } +} + +impl From<(String, String)> for UpdateMapEntry { + fn from((key, value): (String, String)) -> Self { + Self::from((key, Some(value))) + } +} + +impl From<(&str, Option<&str>)> for UpdateMapEntry { + fn from((key, value): (&str, Option<&str>)) -> Self { + Self { + key: key.to_string(), + value: value.map(str::to_owned), + } + } +} + +impl From<(&str, &str)> for UpdateMapEntry { + fn from((key, value): (&str, &str)) -> Self { + Self::from((key, Some(value))) + } +} + +/// Represents updates to a map (either incremental or replacement) +#[derive(Debug, Clone, DeepSizeOf, PartialEq)] +pub struct UpdateMap { + pub update_entries: Vec, + /// If true, the map will be replaced entirely with the new entries. + /// If false, the new entries will be merged with the existing map. + pub replace: bool, +} + +/// Helper function to apply UpdateMap changes to a HashMap +pub(super) fn apply_update_map( + target: &mut std::collections::HashMap, + update_map: &UpdateMap, +) { + if update_map.replace { + // Full replacement - clear existing and replace with new entries that have values + target.clear(); + for entry in &update_map.update_entries { + if let Some(value) = &entry.value { + target.insert(entry.key.clone(), value.clone()); + } + } + } else { + // Incremental update - merge entries + for entry in &update_map.update_entries { + if let Some(value) = &entry.value { + target.insert(entry.key.clone(), value.clone()); + } else { + target.remove(&entry.key); + } + } + } +} + +/// Helper function to translate old-style config updates to new UpdateMap format +pub fn translate_config_updates( + upsert_values: &std::collections::HashMap, + delete_keys: &[String], +) -> UpdateMap { + let mut update_entries = Vec::new(); + + // Add upsert entries (with values) + for (key, value) in upsert_values { + update_entries.push(UpdateMapEntry { + key: key.clone(), + value: Some(value.clone()), + }); + } + + // Add delete entries (without values) + for key in delete_keys { + update_entries.push(UpdateMapEntry { + key: key.clone(), + value: None, + }); + } + + UpdateMap { + update_entries, + replace: false, // Old style was always incremental + } +} + +/// Helper function to translate old-style schema metadata to new UpdateMap format +pub fn translate_schema_metadata_updates( + schema_metadata: &std::collections::HashMap, +) -> UpdateMap { + let update_entries = schema_metadata + .iter() + .map(|(key, value)| UpdateMapEntry { + key: key.clone(), + value: Some(value.clone()), + }) + .collect(); + + UpdateMap { + update_entries, + replace: true, // Old style schema metadata was full replacement + } +} diff --git a/rust/lance-table/src/transaction/validate.rs b/rust/lance-table/src/transaction/validate.rs new file mode 100644 index 00000000000..2228287c89d --- /dev/null +++ b/rust/lance-table/src/transaction/validate.rs @@ -0,0 +1,457 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + +//! Pre-commit validation of an operation against the manifest it applies to. +//! +//! These checks reject transactions that could not produce a coherent manifest — +//! a fragment list that disagrees with the schema, a merge that silently dropped +//! or rewrote data files — before any manifest is written. + +use crate::format::{Fragment, Manifest}; +use crate::transaction::Operation; +use crate::transaction::action::{Action, UserOperation}; +use lance_core::datatypes::Schema; +use lance_core::{Error, Result}; +use lance_file::version::LanceFileVersion; +use std::collections::{HashMap, HashSet}; + +/// Validate the operation is valid for the given manifest. +pub fn validate_operation(manifest: Option<&Manifest>, operation: &Operation) -> Result<()> { + let manifest = match (manifest, operation) { + ( + None, + Operation::Overwrite { + fragments, schema, .. + }, + ) => { + // Validate here because we are going to return early. + schema_fragments_valid(None, schema, fragments)?; + + return Ok(()); + } + (None, Operation::Clone { .. }) => return Ok(()), + (Some(manifest), _) => manifest, + (None, _) => { + return Err(Error::invalid_input(format!( + "Cannot apply operation {} to non-existent dataset", + operation.name() + ))); + } + }; + + match operation { + Operation::Append { fragments } => { + // Fragments must contain all fields in the schema + schema_fragments_valid(Some(manifest), &manifest.schema, fragments) + } + Operation::Project { schema } => { + schema_fragments_valid(Some(manifest), schema, manifest.fragments.as_ref()) + } + Operation::Merge { fragments, schema } => { + merge_fragments_valid(manifest, fragments)?; + schema_fragments_valid(Some(manifest), schema, fragments) + } + Operation::Overwrite { + fragments, + schema, + config_upsert_values: None, + initial_bases: _, + } => { + // Pass None for manifest because Overwrite replaces all fragments. + // The old manifest's storage format is irrelevant for validating + // the new fragments (e.g., LEGACY→STABLE transitions). + schema_fragments_valid(None, schema, fragments) + } + Operation::Update { + updated_fragments, + new_fragments, + .. + } => { + schema_fragments_valid(Some(manifest), &manifest.schema, updated_fragments)?; + schema_fragments_valid(Some(manifest), &manifest.schema, new_fragments) + } + Operation::UserOperation(operation) => validate_user_operation(operation), + _ => Ok(()), + } +} + +/// Check an action-based operation's own structure, independent of the manifest. +/// +/// Local tokens are scoped to the whole operation, not to a single `UserAction`, +/// so distinctness is checked across the flattened action list. Apply would also +/// catch a duplicate, but rejecting here keeps the diagnostic at the API boundary +/// where the caller built the operation. +fn validate_user_operation(operation: &UserOperation) -> Result<()> { + if operation.actions.iter().all(|step| step.actions.is_empty()) { + return Err(Error::invalid_input( + "a UserOperation must contain at least one action", + )); + } + + let mut seen = HashSet::new(); + for action in operation.actions() { + let Action::AddBase(add_base) = action; + if !seen.insert(add_base.local) { + return Err(Error::invalid_input(format!( + "local token {} appears more than once in this UserOperation; \ + local tokens must be distinct within an operation", + add_base.local + ))); + } + } + Ok(()) +} + +fn schema_fragments_valid( + manifest: Option<&Manifest>, + schema: &Schema, + fragments: &[Fragment], +) -> Result<()> { + if let Some(manifest) = manifest + && manifest.data_storage_format.lance_file_version()? == LanceFileVersion::Legacy + { + return schema_fragments_legacy_valid(schema, fragments); + } + // validate that each data file at least contains one field. + for fragment in fragments { + for data_file in &fragment.files { + if data_file.fields.iter().len() == 0 { + return Err(Error::invalid_input(format!( + "Datafile {} does not contain any fields", + data_file.path + ))); + } + } + } + Ok(()) +} + +/// Check that each fragment contains all fields in the schema. +/// It is not required that the schema contains all fields in the fragment. +/// There may be masked fields. +fn schema_fragments_legacy_valid(schema: &Schema, fragments: &[Fragment]) -> Result<()> { + // TODO: add additional validation. Consider consolidating with various + // validate() methods in the codebase. + for fragment in fragments { + for field in schema.fields_pre_order() { + if !fragment + .files + .iter() + .flat_map(|f| f.fields.iter()) + .any(|f_id| f_id == &field.id) + { + return Err(Error::invalid_input(format!( + "Fragment {} does not contain field {:?}", + fragment.id, field + ))); + } + } + } + Ok(()) +} + +/// Returns true if Operation::Merge rewrote this fragment's column data files (Fragment::files +/// changed versus the previous manifest). Used to bump last_updated_at_version_meta only when +/// new column values were materialized to disk. +/// +/// Deletion file changes alone are not treated as rewrites: tombstones remove rows but +/// survivors did not receive new column bytes; stamping last_updated for those rows would be +/// incorrect for CDF. +#[inline] +pub(super) fn merge_fragment_physically_rewritten(prev: &Fragment, merged: &Fragment) -> bool { + debug_assert_eq!(prev.id, merged.id); + if prev.files.len() != merged.files.len() { + return true; + } + // Compare identity fields only. file_size_bytes is an AtomicU64 cache that + // concurrent scans can populate in place on the manifest's DataFile, so it + // must not be part of the rewrite check. + prev.files.iter().zip(merged.files.iter()).any(|(p, m)| { + p.path != m.path + || p.fields != m.fields + || p.column_indices != m.column_indices + || p.file_major_version != m.file_major_version + || p.file_minor_version != m.file_minor_version + || p.base_id != m.base_id + }) +} + +/// Validate that Merge operations preserve all original fragments. +/// Merge operations should only add columns or rows, not reduce fragments. +/// This ensures fragments correspond at one-to-one with the original fragment list. +fn merge_fragments_valid(manifest: &Manifest, new_fragments: &[Fragment]) -> Result<()> { + let original_fragments = manifest.fragments.as_ref(); + + // Additional validation: ensure we're not accidentally reducing the fragment count + if new_fragments.len() < original_fragments.len() { + return Err(Error::invalid_input(format!( + "Merge operation reduced fragment count from {} to {}. \ + Merge operations should only add columns, not reduce fragments.", + original_fragments.len(), + new_fragments.len() + ))); + } + + // Collect new fragment IDs + let new_fragment_map: HashMap = + new_fragments.iter().map(|f| (f.id, f)).collect(); + + // Check that all original fragments are preserved in the new fragments list + // Validate that each original fragment's metadata is preserved + let mut missing_fragments: Vec = Vec::new(); + for original_fragment in original_fragments { + if let Some(new_fragment) = new_fragment_map.get(&original_fragment.id) { + // Validate physical_rows (row count) hasn't changed + if original_fragment.physical_rows != new_fragment.physical_rows { + return Err(Error::invalid_input(format!( + "Merge operation changed row count for fragment {}. \ + Original: {:?}, New: {:?}. \ + Merge operations should preserve fragment row counts and only add new columns.", + original_fragment.id, + original_fragment.physical_rows, + new_fragment.physical_rows + ))); + } + } else { + missing_fragments.push(original_fragment.id); + } + } + + if !missing_fragments.is_empty() { + return Err(Error::invalid_input(format!( + "Merge operation is missing original fragments: {:?}. \ + Merge operations should preserve all original fragments and only add new columns. \ + Expected fragments: {:?}, but got: {:?}", + missing_fragments, + original_fragments.iter().map(|f| f.id).collect::>(), + new_fragment_map.keys().copied().collect::>() + ))); + } + + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::format::{DataFile, DataStorageFormat}; + use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; + use lance_core::datatypes::Schema as LanceSchema; + use std::collections::HashMap; + use std::sync::Arc; + + #[test] + fn test_merge_fragments_valid() { + // Create a simple schema for testing + let schema = ArrowSchema::new(vec![ + ArrowField::new("id", DataType::Int32, false), + ArrowField::new("name", DataType::Utf8, false), + ]); + + // Create original fragments + let original_fragments = vec![Fragment::new(1), Fragment::new(2), Fragment::new(3)]; + + // Create a manifest with original fragments + let manifest = Manifest::new( + LanceSchema::try_from(&schema).unwrap(), + Arc::new(original_fragments), + DataStorageFormat::new(LanceFileVersion::V2_0), + HashMap::new(), + ); + + // Test 1: Empty fragments should fail + let empty_fragments = vec![]; + let result = merge_fragments_valid(&manifest, &empty_fragments); + assert!(result.is_err()); + assert!( + result + .unwrap_err() + .to_string() + .contains("reduced fragment count") + ); + + // Test 2: Missing original fragments should fail + let missing_fragments = vec![ + Fragment::new(1), + Fragment::new(2), + // Fragment 3 is missing + Fragment::new(4), // New fragment + ]; + let result = merge_fragments_valid(&manifest, &missing_fragments); + assert!(result.is_err()); + assert!( + result + .unwrap_err() + .to_string() + .contains("missing original fragments") + ); + + // Test 3: Reduced fragment count should fail + let reduced_fragments = vec![ + Fragment::new(1), + Fragment::new(2), + // Fragment 3 is missing, no new fragments added + ]; + let result = merge_fragments_valid(&manifest, &reduced_fragments); + assert!(result.is_err()); + assert!( + result + .unwrap_err() + .to_string() + .contains("reduced fragment count") + ); + + // Test 4: Valid merge with all original fragments plus new ones should succeed + let valid_fragments = vec![ + Fragment::new(1), + Fragment::new(2), + Fragment::new(3), + Fragment::new(4), // New fragment + Fragment::new(5), // Another new fragment + ]; + let result = merge_fragments_valid(&manifest, &valid_fragments); + assert!(result.is_ok()); + + // Test 5: Same fragments (no new ones) should succeed + let same_fragments = vec![Fragment::new(1), Fragment::new(2), Fragment::new(3)]; + let result = merge_fragments_valid(&manifest, &same_fragments); + assert!(result.is_ok()); + } + + /// Regression test for https://github.com/lance-format/lance/issues/6417 + /// + /// When overwriting a LEGACY dataset with STABLE-format fragments, the + /// validation should not use the old manifest's format. STABLE fragments + /// omit struct parent fields, which the strict legacy check rejects. + #[test] + fn test_overwrite_legacy_to_stable_with_struct_fields() { + use arrow_schema::Fields; + + // Schema: id (field 0), name (field 1), address (field 2, struct parent), + // city (field 3), country (field 4) + let arrow_schema = ArrowSchema::new(vec![ + ArrowField::new("id", DataType::Int32, false), + ArrowField::new("name", DataType::Utf8, false), + ArrowField::new( + "address", + DataType::Struct(Fields::from(vec![ + ArrowField::new("city", DataType::Utf8, false), + ArrowField::new("country", DataType::Utf8, false), + ])), + false, + ), + ]); + let schema = LanceSchema::try_from(&arrow_schema).unwrap(); + + // Old manifest is LEGACY format + let legacy_manifest = Manifest::new( + schema.clone(), + Arc::new(vec![Fragment::new(0)]), + DataStorageFormat::new(LanceFileVersion::Legacy), + HashMap::new(), + ); + + // New fragments in STABLE format omit struct parent field (id=2), + // only including leaf fields: id=0, name=1, city=3, country=4 + let stable_fragment = Fragment { + id: 0, + files: vec![DataFile::new( + "data.lance", + vec![0, 1, 3, 4], // no field 2 (struct parent) + vec![0, 1, 2, 3], + lance_file::format::MAJOR_VERSION as u32, + lance_file::format::MINOR_VERSION as u32, + None, + None, + )], + physical_rows: Some(10), + overlays: vec![], + deletion_file: None, + row_id_meta: None, + last_updated_at_version_meta: None, + created_at_version_meta: None, + }; + + let operation = Operation::Overwrite { + fragments: vec![stable_fragment], + schema, + config_upsert_values: None, + initial_bases: None, + }; + + // This should succeed — the old manifest's LEGACY format should not + // cause strict validation of the new STABLE fragments. + validate_operation(Some(&legacy_manifest), &operation).unwrap(); + } + + fn user_operation(steps: Vec>) -> Operation { + Operation::UserOperation(UserOperation { + description: "test".to_string(), + uuid: "u".to_string(), + read_version: 1, + actions: steps + .into_iter() + .map(|locals| crate::transaction::UserAction { + description: "step".to_string(), + actions: locals + .into_iter() + .map(|local| { + Action::AddBase(crate::transaction::AddBase { + local, + name: Some(format!("base-{local}")), + is_dataset_root: false, + path: format!("s3://bucket/{local}"), + }) + }) + .collect(), + }) + .collect(), + }) + } + + fn manifest() -> Manifest { + Manifest::new( + LanceSchema::default(), + Arc::new(Vec::new()), + DataStorageFormat::default(), + HashMap::new(), + ) + } + + #[test] + fn test_user_operation_with_distinct_local_tokens_is_valid() { + validate_operation( + Some(&manifest()), + &user_operation(vec![vec![0, 1], vec![2]]), + ) + .unwrap(); + } + + #[test] + fn test_user_operation_with_duplicate_local_token_is_rejected() { + // Tokens are scoped to the whole operation, not to a single UserAction, so a + // token reused across two steps is still a duplicate. + let err = validate_operation(Some(&manifest()), &user_operation(vec![vec![0], vec![0]])) + .unwrap_err(); + + assert!( + matches!(err, Error::InvalidInput { .. }), + "expected InvalidInput, got: {err:?}" + ); + assert!( + err.to_string() + .contains("local token 0 appears more than once") + ); + } + + #[test] + fn test_user_operation_with_no_actions_is_rejected() { + for steps in [vec![], vec![vec![]]] { + let err = validate_operation(Some(&manifest()), &user_operation(steps)).unwrap_err(); + assert!( + err.to_string().contains("at least one action"), + "got: {err}" + ); + } + } +} diff --git a/rust/lance/src/dataset.rs b/rust/lance/src/dataset.rs index 9f90a4ba1d5..fe3af8d4106 100644 --- a/rust/lance/src/dataset.rs +++ b/rust/lance/src/dataset.rs @@ -43,8 +43,8 @@ use lance_io::utils::{ }; use lance_namespace::LanceNamespace; use lance_table::format::{ - DataFile, DataStorageFormat, DeletionFile, Fragment, IndexMetadata, MAGIC, Manifest, RowIdMeta, - pb, + DataFile, DataStorageFormat, DeletionFile, Fragment, IndexMetadata, MAGIC, Manifest, + ManifestBuildConfig, RowIdMeta, pb, }; use lance_table::io::commit::{ CommitConfig, CommitError, CommitHandler, CommitLock, ManifestLocation, ManifestNamingScheme, @@ -90,7 +90,28 @@ mod schema_evolution; pub mod sql; pub mod statistics; mod take; -pub mod transaction; +/// Transaction definitions for updating datasets +/// +/// Prior to creating a new manifest, a transaction must be created representing +/// the changes being made to the dataset. By representing them as incremental +/// changes, we can detect whether concurrent operations are compatible with +/// one another. We can also rebuild manifests when retrying committing a +/// manifest. +/// +/// The definitions live in [`lance_table::transaction`]: building a manifest from +/// a transaction reads and writes only table metadata, so it belongs at the table +/// layer. This module re-exports them at the path callers have always used. +/// +/// For more details please refer to the +/// [Transaction Specification](https://lance.org/format/table/transaction/#transaction-types). +pub mod transaction { + pub use lance_table::transaction::{ + DataOverlayGroup, DataReplacementGroup, Operation, RewriteGroup, RewrittenIndex, + Transaction, TransactionBuilder, UpdateMap, UpdateMapEntry, UpdateMode, + UpdatedFragmentOffsets, translate_config_updates, translate_schema_metadata_updates, + validate_operation, + }; +} pub mod udtf; pub mod updater; mod utils; @@ -3977,6 +3998,21 @@ impl ManifestWriteConfig { pub fn disable_transaction_file(&self) -> bool { self.disable_transaction_file } + + /// Resolve into the config `Transaction::build_manifest` consumes. + /// + /// The timestamp is resolved here rather than during the build so it goes + /// through this crate's mockable `SystemTime`. + pub(crate) fn to_build_config(&self) -> ManifestBuildConfig { + ManifestBuildConfig { + auto_set_feature_flags: self.auto_set_feature_flags, + timestamp_nanos: timestamp_to_nanos(self.timestamp), + use_stable_row_ids: self.use_stable_row_ids, + use_legacy_format: self.use_legacy_format, + storage_format: self.storage_format.clone(), + disable_transaction_file: self.disable_transaction_file, + } + } } /// Commit a manifest file and create a copy at the latest manifest path. diff --git a/rust/lance/src/dataset/overlay.rs b/rust/lance/src/dataset/overlay.rs index 83d4bc17938..1f55c545559 100644 --- a/rust/lance/src/dataset/overlay.rs +++ b/rust/lance/src/dataset/overlay.rs @@ -44,146 +44,16 @@ use lance_core::datatypes::{Field, Schema}; use lance_core::{Error, Result}; use roaring::RoaringBitmap; -use lance_table::format::overlay::DataOverlayFile; -use lance_table::format::{DataFile, Fragment, IndexMetadata}; +use lance_table::format::DataFile; use lance_table::utils::stream::ReadBatchFut; use crate::dataset::fragment::{FileFragment, FragReadConfig, GenericFileReader}; -/// The physical offsets within a fragment whose value for an indexed field may be -/// stale relative to an index built at `index_version`, and so must be excluded -/// from that index's results and re-evaluated against current values on the flat -/// path. -/// -/// The set is the union, over every overlay whose `committed_version` is newer -/// than `index_version`, of that overlay's coverage **restricted to the indexed -/// fields**. The restriction makes exclusion field-aware: an overlay that touches -/// only non-indexed fields contributes nothing. An overlay whose -/// `committed_version <= index_version` is already incorporated by the index and -/// is ignored. -pub fn overlay_exclusion_offsets( - overlays: &[DataOverlayFile], - indexed_field_ids: &[i32], - index_version: u64, - schema: &Schema, -) -> Result { - let mut excluded = RoaringBitmap::new(); - for overlay in overlays { - if overlay.committed_version <= index_version { - continue; - } - for (field_pos, field_id) in overlay.data_file.fields.iter().enumerate() { - let overlay_ancestry = schema.field_ancestry_by_id(*field_id); - let affects_index = indexed_field_ids.iter().any(|indexed_field_id| { - indexed_field_id == field_id - || overlay_ancestry.as_ref().is_some_and(|ancestry| { - ancestry - .iter() - .any(|ancestor| ancestor.id == *indexed_field_id) - }) - || schema - .field_ancestry_by_id(*indexed_field_id) - .is_some_and(|ancestry| { - ancestry.iter().any(|ancestor| ancestor.id == *field_id) - }) - }); - if affects_index { - excluded |= &*overlay.coverage_for_field(field_pos)?; - } - } - } - Ok(excluded) -} - -// Stale row offsets contributed by one fragment's overlays for a given index version. -// Applies a cheap version gate first: if every overlay predates the segment it is already -// incorporated by the index, so there is nothing stale and the field/bitmap work is skipped. -fn stale_offsets_for_fragment( - fragment: &Fragment, - fields: &[i32], - index_version: u64, - schema: &Schema, -) -> Result { - if fragment - .overlays - .iter() - .all(|o| o.committed_version <= index_version) - { - return Ok(RoaringBitmap::new()); - } - overlay_exclusion_offsets(&fragment.overlays, fields, index_version, schema) -} - -// A missing `fragment_bitmap` means the index predates fragment-bitmap tracking; treat it as -// covering every fragment (matching `DatasetPreFilter::new`) so overlay-stale rows can't slip -// through unmasked. Only skip fragments explicitly absent from a present bitmap. -fn covers_fragment(coverage: Option<&RoaringBitmap>, frag_id: u32) -> bool { - coverage.is_none_or(|c| c.contains(frag_id)) -} - -/// Index by fragment id the fragments that carry at least one overlay. Overlays are rare, so -/// this is empty on the common path, letting callers skip index loading entirely; when non-empty -/// it bounds the stale-collection loops to `O(overlaid fragments)`. -pub fn overlaid_fragments(fragments: &[Fragment]) -> HashMap { - fragments - .iter() - .filter(|f| !f.overlays.is_empty()) - .map(|f| (f.id as u32, f)) - .collect() -} - -/// Insert into `stale` the ids of fragments covered by `segment` whose index entries may be -/// stale because an overlay committed after the segment was built touches a field the segment -/// indexes. Field-aware and version-gated via [`overlay_exclusion_offsets`]. -/// -/// `overlaid_frags` holds only the fragments that actually carry overlays (rare), so the loop is -/// `O(overlaid_frags)` rather than `O(fragments the segment covers)`. -pub fn collect_overlay_stale_frags( - segment: &IndexMetadata, - overlaid_frags: &HashMap, - stale: &mut RoaringBitmap, - schema: &Schema, -) -> Result<()> { - let coverage = segment.fragment_bitmap.as_ref(); - for (&frag_id, fragment) in overlaid_frags { - if stale.contains(frag_id) || !covers_fragment(coverage, frag_id) { - continue; - } - if !stale_offsets_for_fragment(fragment, &segment.fields, segment.dataset_version, schema)? - .is_empty() - { - stale.insert(frag_id); - } - } - Ok(()) -} - -/// Like [`collect_overlay_stale_frags`] but with row-level granularity: instead of marking the -/// whole fragment stale, it computes exactly which row offsets within each covered fragment are -/// stale and accumulates them into `stale` (fragment_id → stale row offsets). -/// -/// Used by the scalar and vector paths to block only the affected rows from index results and -/// re-evaluate only those rows on the flat path, keeping overhead proportional to the number of -/// overlaid rows rather than the whole fragment size. -pub fn collect_overlay_stale_rows_for_segment( - segment: &IndexMetadata, - overlaid_frags: &HashMap, - stale: &mut HashMap, - schema: &Schema, -) -> Result<()> { - let coverage = segment.fragment_bitmap.as_ref(); - for (&frag_id, fragment) in overlaid_frags { - if !covers_fragment(coverage, frag_id) { - continue; - } - let excluded = - stale_offsets_for_fragment(fragment, &segment.fields, segment.dataset_version, schema)?; - if !excluded.is_empty() { - *stale.entry(frag_id).or_default() |= &excluded; - } - } - Ok(()) -} +// Deciding which rows an overlay makes stale needs only fragment and index metadata, +// so it lives at the table layer; this module resolves the reads that consume it. +pub use lance_table::format::overlay::staleness::{ + collect_overlay_stale_frags, collect_overlay_stale_rows_for_segment, overlaid_fragments, +}; /// The plan for merging one field's overlays into one batch: which source (base or /// a particular overlay) supplies each output row, and which overlay values must be @@ -895,7 +765,6 @@ async fn fetch_overlay_values( mod tests { use super::*; use arrow_array::{Int32Array, StringArray, UInt32Array}; - use lance_table::format::overlay::OverlayCoverage; use std::sync::Arc; fn i32_array(values: impl IntoIterator>) -> ArrayRef { @@ -906,19 +775,6 @@ mod tests { RoaringBitmap::from_iter(offsets) } - fn flat_test_schema() -> Schema { - use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; - - let mut schema = Schema::try_from(&ArrowSchema::new( - (0..5) - .map(|id| ArrowField::new(format!("field_{id}"), DataType::Int32, true)) - .collect::>(), - )) - .unwrap(); - schema.set_field_id(None); - schema - } - /// Physical offsets for a contiguous range `[start, start + len)`. fn offsets(start: u32, len: usize) -> Vec { (start..start + len as u32).collect() @@ -1212,216 +1068,4 @@ mod tests { assert!(spliced.is_null(1)); assert!(!spliced.is_null(2)); } - - /// A dense overlay covering `offsets` for `field_ids`, committed at `version`. - fn dense_overlay( - field_ids: Vec, - offsets: impl IntoIterator, - version: u64, - ) -> lance_table::format::overlay::DataOverlayFile { - DataOverlayFile { - data_file: DataFile::new_legacy_from_fields("o.lance", field_ids, None), - coverage: OverlayCoverage::dense(bitmap(offsets)), - committed_version: version, - } - } - - #[test] - fn test_exclusion_offsets_version_gate() { - let schema = flat_test_schema(); - // index built at version 5; only overlays committed > 5 are excluded. - let overlays = vec![ - dense_overlay(vec![3], [0, 1], 4), - dense_overlay(vec![3], [2, 7], 6), - ]; - let excluded = overlay_exclusion_offsets(&overlays, &[3], 5, &schema).unwrap(); - assert_eq!(excluded, bitmap([2, 7])); - // An overlay exactly at the index version is already incorporated. - let overlays = vec![dense_overlay(vec![3], [9], 5)]; - assert!( - overlay_exclusion_offsets(&overlays, &[3], 5, &schema) - .unwrap() - .is_empty() - ); - } - - #[test] - fn test_exclusion_offsets_is_field_aware() { - let schema = flat_test_schema(); - // An overlay touching only an unrelated field excludes nothing. - let overlays = vec![dense_overlay(vec![2], [0, 1, 2], 9)]; - assert!( - overlay_exclusion_offsets(&overlays, &[3], 1, &schema) - .unwrap() - .is_empty() - ); - // The union spans only the indexed fields the overlay actually carries. - let overlays = vec![dense_overlay(vec![2, 3], [4], 9)]; - assert_eq!( - overlay_exclusion_offsets(&overlays, &[3], 1, &schema).unwrap(), - bitmap([4]) - ); - } - - #[test] - fn test_exclusion_offsets_matches_nested_fields() { - let (schema, _) = nested_struct(); - let outer = &schema.fields[0]; - let middle = &outer.children[0]; - let a = &middle.children[0]; - let b = &middle.children[1]; - - let overlays = vec![dense_overlay(vec![a.id], [1], 9)]; - assert_eq!( - overlay_exclusion_offsets(&overlays, &[outer.id], 1, &schema).unwrap(), - bitmap([1]) - ); - assert!( - overlay_exclusion_offsets(&overlays, &[b.id], 1, &schema) - .unwrap() - .is_empty() - ); - - let overlays = vec![dense_overlay(vec![middle.id], [2], 9)]; - assert_eq!( - overlay_exclusion_offsets(&overlays, &[a.id], 1, &schema).unwrap(), - bitmap([2]) - ); - } - - #[test] - fn test_exclusion_offsets_sparse_per_field() { - let schema = flat_test_schema(); - // Sparse overlay: field 2 covers {2,3}, field 4 covers {1}. - let overlay = DataOverlayFile { - data_file: DataFile::new_legacy_from_fields("o.lance", vec![2, 4], None), - coverage: OverlayCoverage::sparse(vec![bitmap([2, 3]), bitmap([1])]), - committed_version: 9, - }; - let overlays = vec![overlay]; - // Only the bitmap for the indexed field (4) contributes. - assert_eq!( - overlay_exclusion_offsets(&overlays, &[4], 1, &schema).unwrap(), - bitmap([1]) - ); - assert_eq!( - overlay_exclusion_offsets(&overlays, &[2], 1, &schema).unwrap(), - bitmap([2, 3]) - ); - } - - #[test] - fn test_exclusion_offsets_unions_multiple_overlays() { - let schema = flat_test_schema(); - let overlays = vec![ - dense_overlay(vec![3], [1], 6), - dense_overlay(vec![3], [4, 5], 7), - ]; - assert_eq!( - overlay_exclusion_offsets(&overlays, &[3], 1, &schema).unwrap(), - bitmap([1, 4, 5]) - ); - } - - /// An index segment covering `fields`, built at `dataset_version`, with the given - /// fragment coverage (`None` = legacy index predating fragment-bitmap tracking). - fn segment( - fields: Vec, - dataset_version: u64, - fragment_bitmap: Option, - ) -> IndexMetadata { - IndexMetadata { - uuid: uuid::Uuid::new_v4(), - name: "idx".into(), - fields, - dataset_version, - fragment_bitmap, - index_details: None, - index_version: 0, - created_at: None, - base_id: None, - files: None, - } - } - - fn fragment_with_overlay(id: u64, overlay: DataOverlayFile) -> Fragment { - let mut fragment = Fragment::new(id); - fragment.overlays.push(overlay); - fragment - } - - #[test] - fn test_collect_frags_missing_bitmap_covers_all() { - let schema = flat_test_schema(); - // A segment with no fragment_bitmap (legacy index predating bitmap tracking) must treat - // every overlaid fragment as covered so stale rows can't leak past the index unmasked. - let fragment = fragment_with_overlay(3, dense_overlay(vec![3], [1, 2], 9)); - let overlaid: HashMap = HashMap::from([(3u32, &fragment)]); - - let mut stale = RoaringBitmap::new(); - collect_overlay_stale_frags(&segment(vec![3], 1, None), &overlaid, &mut stale, &schema) - .unwrap(); - assert_eq!(stale, bitmap([3]), "missing bitmap must cover fragment 3"); - - // A present bitmap that excludes fragment 3 leaves it untouched. - let mut stale = RoaringBitmap::new(); - collect_overlay_stale_frags( - &segment(vec![3], 1, Some(bitmap([0]))), - &overlaid, - &mut stale, - &schema, - ) - .unwrap(); - assert!( - stale.is_empty(), - "fragment absent from bitmap is not covered" - ); - - // A present bitmap that includes fragment 3 marks it stale. - let mut stale = RoaringBitmap::new(); - collect_overlay_stale_frags( - &segment(vec![3], 1, Some(bitmap([3]))), - &overlaid, - &mut stale, - &schema, - ) - .unwrap(); - assert_eq!(stale, bitmap([3])); - } - - #[test] - fn test_collect_rows_missing_bitmap_covers_all() { - let schema = flat_test_schema(); - // Same covers-all guarantee at row-level granularity. - let fragment = fragment_with_overlay(3, dense_overlay(vec![3], [1, 2], 9)); - let overlaid: HashMap = HashMap::from([(3u32, &fragment)]); - - let mut stale = HashMap::new(); - collect_overlay_stale_rows_for_segment( - &segment(vec![3], 1, None), - &overlaid, - &mut stale, - &schema, - ) - .unwrap(); - assert_eq!( - stale.get(&3), - Some(&bitmap([1, 2])), - "missing bitmap must cover fragment 3" - ); - - // A present bitmap that excludes fragment 3 yields no stale rows. - let mut stale = HashMap::new(); - collect_overlay_stale_rows_for_segment( - &segment(vec![3], 1, Some(bitmap([0]))), - &overlaid, - &mut stale, - &schema, - ) - .unwrap(); - assert!( - stale.is_empty(), - "fragment absent from bitmap contributes no rows" - ); - } } diff --git a/rust/lance/src/dataset/scanner.rs b/rust/lance/src/dataset/scanner.rs index a4b0b6b864b..0a2f80a8186 100644 --- a/rust/lance/src/dataset/scanner.rs +++ b/rust/lance/src/dataset/scanner.rs @@ -4308,7 +4308,8 @@ impl Scanner { /// /// The check is field-aware (an overlay touching only unindexed fields excludes nothing) and /// version-gated (an overlay with `committed_version <= index.dataset_version` is already - /// incorporated by the index), via [`overlay_exclusion_offsets`]. + /// incorporated by the index), via + /// [`lance_table::format::overlay::staleness::overlay_exclusion_offsets`]. async fn overlay_stale_index_rows( &self, index_expr: &ScalarIndexExpr, diff --git a/rust/lance/src/dataset/tests/dataset_transactions.rs b/rust/lance/src/dataset/tests/dataset_transactions.rs index 74790d301c1..b9cdec04c68 100644 --- a/rust/lance/src/dataset/tests/dataset_transactions.rs +++ b/rust/lance/src/dataset/tests/dataset_transactions.rs @@ -6,20 +6,25 @@ use std::sync::Arc; use std::vec; use crate::dataset::builder::DatasetBuilder; -use crate::dataset::transaction::{Operation, Transaction}; +use crate::dataset::transaction::{Operation, Transaction, UpdateMode, UpdatedFragmentOffsets}; use crate::dataset::{ManifestWriteConfig, TRANSACTIONS_DIR, write_manifest_file}; use crate::io::ObjectStoreParams; use crate::session::Session; use crate::{Dataset, Result}; +use lance_file::version::LanceFileVersion; use lance_table::io::commit::ManifestNamingScheme; use crate::dataset::write::{CommitBuilder, InsertBuilder, WriteMode, WriteParams}; use crate::index::DatasetIndexExt; use arrow_array::Array; use arrow_array::RecordBatch; +use arrow_array::cast::AsArray; +use arrow_array::types::UInt64Type; use arrow_array::{Int32Array, RecordBatchIterator, StringArray, types::Int32Type}; use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; +use lance_core::utils::address::RowAddress; use lance_core::utils::tempfile::{TempDir, TempStrDir}; +use lance_core::{ROW_ADDR, ROW_CREATED_AT_VERSION, ROW_LAST_UPDATED_AT_VERSION}; use lance_datagen::{BatchCount, RowCount, array}; use crate::datafusion::LanceTableProvider; @@ -391,7 +396,7 @@ async fn test_inline_transaction() { Some(ds.manifest.as_ref()), ds.load_indices().await.unwrap().as_ref().clone(), &tx_file, - &ManifestWriteConfig::default(), + &ManifestWriteConfig::default().to_build_config(), ) .unwrap(); let location = write_manifest_file( @@ -726,3 +731,184 @@ async fn test_list_detached_manifests() { assert_eq!(versions.len(), 1); assert_eq!(versions[0].version, 1); } + +/// Partial RewriteColumns refresh in `build_manifest`: only matched physical +/// rows get `last_updated_at_version` bumped; same-fragment unmatched rows and +/// untouched fragments keep both version sequences. +#[tokio::test] +async fn test_build_manifest_partial_last_updated_rewrite_columns_stable_row_ids() { + let dir = TempStrDir::default(); + let uri = dir.as_str(); + + let schema = Arc::new(ArrowSchema::new(vec![ + ArrowField::new("i", DataType::Int32, false), + ArrowField::new("x", DataType::Int32, false), + ])); + let batch0 = RecordBatch::try_new( + schema.clone(), + vec![ + Arc::new(Int32Array::from_iter_values(0..8)), + Arc::new(Int32Array::from(vec![0_i32; 8])), + ], + ) + .unwrap(); + let reader0 = RecordBatchIterator::new(vec![Ok(batch0)], schema.clone()); + let write_params = WriteParams { + enable_stable_row_ids: true, + data_storage_version: Some(LanceFileVersion::Stable), + ..Default::default() + }; + let mut dataset = Dataset::write(reader0, uri, Some(write_params)) + .await + .unwrap(); + + let batch1 = RecordBatch::try_new( + schema.clone(), + vec![ + Arc::new(Int32Array::from_iter_values(100..108)), + Arc::new(Int32Array::from(vec![0_i32; 8])), + ], + ) + .unwrap(); + let reader1 = RecordBatchIterator::new(vec![Ok(batch1)], schema.clone()); + dataset.append(reader1, None).await.unwrap(); + + let frags = dataset.get_fragments(); + assert_eq!( + frags.len(), + 2, + "expected two fragments (append creates a new fragment)" + ); + + async fn scan_row_versions(ds: &Dataset) -> HashMap<(u32, u32), (u64, u64)> { + let mut scanner = ds.scan(); + scanner + .project(&[ + ROW_ADDR, + ROW_LAST_UPDATED_AT_VERSION, + ROW_CREATED_AT_VERSION, + ]) + .unwrap(); + let batches = scanner + .try_into_stream() + .await + .unwrap() + .try_collect::>() + .await + .unwrap(); + let mut out = HashMap::new(); + for batch in batches { + let addrs = batch + .column_by_name(ROW_ADDR) + .unwrap() + .as_primitive::(); + let last = batch + .column_by_name(ROW_LAST_UPDATED_AT_VERSION) + .unwrap() + .as_primitive::(); + let created = batch + .column_by_name(ROW_CREATED_AT_VERSION) + .unwrap() + .as_primitive::(); + for row in 0..batch.num_rows() { + let addr = RowAddress::from(addrs.value(row)); + out.insert( + (addr.fragment_id(), addr.row_offset()), + (last.value(row), created.value(row)), + ); + } + } + out + } + + let before = scan_row_versions(&dataset).await; + assert_eq!(before.len(), 16); + + // Update only rows i in {2, 4, 6} within fragment 0 (physical offsets 2, 4, 6). + let update_schema = Arc::new(ArrowSchema::new(vec![ + ArrowField::new("i", DataType::Int32, false), + ArrowField::new("x", DataType::Int32, false), + ])); + let update_batch = RecordBatch::try_new( + update_schema.clone(), + vec![ + Arc::new(Int32Array::from(vec![2, 4, 6])), + Arc::new(Int32Array::from(vec![99, 99, 99])), + ], + ) + .unwrap(); + let right: Box = Box::new(RecordBatchIterator::new( + vec![Ok(update_batch)].into_iter(), + update_schema, + )); + + let mut frag0 = dataset.get_fragment(0).unwrap(); + let u = frag0 + .update_columns_with_offsets(right, "i", "i") + .await + .unwrap(); + assert_eq!(u.matched_offsets.iter().count(), 3); + for off in [2_u32, 4, 6] { + assert!(u.matched_offsets.contains(off)); + } + + let updated_fragment_offsets = Some(UpdatedFragmentOffsets(HashMap::from([( + u.fragment.id, + u.matched_offsets, + )]))); + + let op = Operation::Update { + removed_fragment_ids: vec![], + updated_fragments: vec![u.fragment], + new_fragments: vec![], + fields_modified: u.fields_modified, + compacted_sstables: Vec::new(), + fields_for_preserving_frag_bitmap: vec![], + update_mode: Some(UpdateMode::RewriteColumns), + inserted_rows_filter: None, + updated_fragment_offsets, + }; + + let read_v = dataset.version().version; + let dataset = Dataset::commit( + uri, + op, + Some(read_v), + None, + None, + Arc::new(Session::default()), + true, + ) + .await + .unwrap(); + + let new_v = dataset.version().version; + assert_eq!(new_v, read_v + 1); + + let after = scan_row_versions(&dataset).await; + for off in 0..8_u32 { + let key = (0, off); + let (last_before, created_before) = before[&key]; + let (last_after, created_after) = after[&key]; + assert_eq!(created_after, created_before); + if off == 2 || off == 4 || off == 6 { + assert_eq!( + last_after, new_v, + "matched row offset {off} should advance last_updated to new version" + ); + } else { + assert_eq!( + last_after, last_before, + "unmatched row offset {off} in fragment 0 should keep last_updated" + ); + } + } + + for off in 0..8_u32 { + let key = (1, off); + assert_eq!( + after[&key], before[&key], + "fragment 1 row offset {off}: both version columns unchanged" + ); + } +} diff --git a/rust/lance/src/dataset/transaction.rs b/rust/lance/src/dataset/transaction.rs deleted file mode 100644 index 35a85100bf3..00000000000 --- a/rust/lance/src/dataset/transaction.rs +++ /dev/null @@ -1,6767 +0,0 @@ -// SPDX-License-Identifier: Apache-2.0 -// SPDX-FileCopyrightText: Copyright The Lance Authors - -//! Transaction definitions for updating datasets -//! -//! Prior to creating a new manifest, a transaction must be created representing -//! the changes being made to the dataset. By representing them as incremental -//! changes, we can detect whether concurrent operations are compatible with -//! one another. We can also rebuild manifests when retrying committing a -//! manifest. -//! -//! For more details please refer to the -//! [Transaction Specification](https://lance.org/format/table/transaction/#transaction-types). - -use super::ManifestWriteConfig; -use super::write::merge_insert::inserted_rows::KeyExistenceFilter; -use crate::dataset::overlay::collect_overlay_stale_frags; -use crate::dataset::transaction::UpdateMode::{RewriteColumns, RewriteRows}; -use crate::index::mem_wal::update_mem_wal_index_compacted_sstables; -use crate::utils::temporal::timestamp_to_nanos; -use lance_core::datatypes::{ - LANCE_UNENFORCED_CLUSTERING_KEY_POSITION, LANCE_UNENFORCED_PRIMARY_KEY, - LANCE_UNENFORCED_PRIMARY_KEY_POSITION, -}; -use lance_core::deepsize::DeepSizeOf; -use lance_core::{Error, Result, datatypes::Schema}; -use lance_file::{datatypes::Fields, version::LanceFileVersion}; -use lance_index::mem_wal::CompactedSsTable; -use lance_index::{frag_reuse::FRAG_REUSE_INDEX_NAME, is_system_index}; -use lance_io::object_store::ObjectStore; -use lance_table::feature_flags::{FLAG_STABLE_ROW_IDS, apply_feature_flags}; -use lance_table::rowids::read_row_ids; -use lance_table::{ - format::{ - BasePath, DataFile, DataStorageFormat, Fragment, IndexFile, IndexMetadata, Manifest, - RowDatasetVersionMeta, RowDatasetVersionRun, RowDatasetVersionSequence, RowIdMeta, - overlay::DataOverlayFile, pb, - }, - io::{ - commit::CommitHandler, - manifest::{read_manifest, read_manifest_indexes}, - }, - rowids::{RowIdSequence, segment::U64Segment, version::build_version_meta, write_row_ids}, -}; -use object_store::path::Path; -use roaring::RoaringBitmap; -use std::cmp::Ordering; -use std::{ - collections::{HashMap, HashSet}, - sync::Arc, -}; -use uuid::Uuid; - -/// Fallback version for rows whose original creation version cannot be determined. -/// Version 1 is the initial dataset version in the Lance format. -const UNKNOWN_CREATED_AT_VERSION: u64 = 1; - -/// Look up the `created_at` version for a single UPDATE-branch row ID. -/// -/// Callers must only call this for row IDs that are confirmed to be present in -/// `row_id_to_source` (i.e. UPDATE branch rows whose source exists in an existing -/// fragment). INSERT branch rows (no source) must use `new_version` directly and -/// must not call this function. -/// -/// Uses `row_id_to_source` to find the originating fragment and row offset, then -/// performs a O(K) random-access lookup via [`RowDatasetVersionSequence::version_at`] -/// on the pre-decoded sequence in `version_cache` (keyed by fragment ID). -/// -/// Returns [`UNKNOWN_CREATED_AT_VERSION`] if the source fragment has no -/// `created_at_version_meta` (missing or failed to decode) or the offset is -/// out of range. -fn resolve_created_at_version( - row_id: u64, - row_id_to_source: &HashMap, - version_cache: &HashMap, -) -> u64 { - let Some((orig_frag, row_offset)) = row_id_to_source.get(&row_id) else { - return UNKNOWN_CREATED_AT_VERSION; - }; - let Some(seq) = version_cache.get(&orig_frag.id) else { - return UNKNOWN_CREATED_AT_VERSION; - }; - seq.version_at(*row_offset) - .unwrap_or(UNKNOWN_CREATED_AT_VERSION) -} - -/// For each new fragment produced by an update, set `created_at_version_meta` -/// (preserved from the original rows) and `last_updated_at_version_meta`. -fn resolve_update_version_metadata( - existing_fragments: &[Fragment], - new_fragments: &mut [Fragment], - new_version: u64, -) -> Result<()> { - // Collect only the row IDs we actually need to resolve, those appearing in new_fragments - // with inline metadata. This bounds the lookup map to O(updated rows) instead of O(all dataset rows) - let needed_row_ids: HashSet = new_fragments - .iter() - .filter_map(|f| match &f.row_id_meta { - Some(RowIdMeta::Inline(data)) => read_row_ids(data).ok(), - _ => None, - }) - .flat_map(|seq| seq.iter().collect::>()) - .collect(); - - let mut row_id_to_source: HashMap = HashMap::new(); - - if !needed_row_ids.is_empty() { - // Compute the bounding range of the needed set once. Any fragment whose - // entire row-id range lies outside [needed_min, needed_max] cannot contain - // any needed ID and can be skipped before the inner per-row loop. - let needed_min = *needed_row_ids.iter().min().unwrap(); - let needed_max = *needed_row_ids.iter().max().unwrap(); - - // Stable row IDs must be globally unique among *live* rows, but after a rewrite-style - // update the same stable ID can appear twice in `existing_fragments`: once in an older - // fragment's inline `row_id_meta` at the original row offset (rows may be soft-deleted - // via a deletion vector) and again in a newer fragment holding rewritten data. For - // `created_at` we need the mapping from the original fragment/offset; that is always the - // first occurrence when fragments are processed in ascending `id` order. - let mut sorted_frags: Vec<&Fragment> = existing_fragments.iter().collect(); - sorted_frags.sort_by_key(|f| f.id); - for frag in sorted_frags { - if let Some(RowIdMeta::Inline(data)) = &frag.row_id_meta - && let Ok(seq) = read_row_ids(data) - { - // Range pre-filter: skip the per-row inner loop when the fragment's - // bounding row-id range has no overlap with [needed_min, needed_max]. - // row_id_range() returns None for empty sequences, which are also skipped. - // This is a conservative check (may produce false positives for sparse - // segments) but never skips a fragment that actually contains a needed ID. - if seq - .row_id_range() - .is_none_or(|r| *r.end() < needed_min || *r.start() > needed_max) - { - continue; - } - - for (offset, rid) in seq.iter().enumerate() { - if needed_row_ids.contains(&rid) { - row_id_to_source.entry(rid).or_insert((frag, offset)); - } - } - } - } - } - - // Pre-decode the `created_at` version sequence for each source fragment exactly - // once. Without this cache, resolve_created_at_version would call load_sequence() - // (a protobuf decode) for every single updated row, even when many rows originate - // from the same fragment. - let source_frag_ids: HashSet = row_id_to_source.values().map(|(f, _)| f.id).collect(); - let version_cache: HashMap = existing_fragments - .iter() - .filter(|f| source_frag_ids.contains(&f.id)) - .filter_map(|frag| { - let seq = frag - .created_at_version_meta - .as_ref()? - .load_sequence() - .ok()?; - Some((frag.id, seq)) - }) - .collect(); - - for fragment in new_fragments.iter_mut() { - let row_ids = match &fragment.row_id_meta { - Some(RowIdMeta::Inline(data)) => read_row_ids(data).ok(), - Some(RowIdMeta::External(_)) => { - log::warn!( - "Fragment {} has external row ID metadata; \ - version tracking will use defaults", - fragment.id, - ); - None - } - None => None, - }; - - if let Some(row_ids) = row_ids { - let physical_rows = fragment.physical_rows.unwrap_or(0); - let created_at_versions: Vec = row_ids - .iter() - .map(|rid| { - if row_id_to_source.contains_key(&rid) { - // UPDATE branch: stable row ID resolves to a source row in an - // existing fragment. Copy created_at from the original row so - // the row's first-appearance version is preserved across rewrites. - resolve_created_at_version(rid, &row_id_to_source, &version_cache) - } else { - // INSERT branch: stable row ID has no source in existing fragments - // (e.g. NOT MATCHED arm of MERGE INTO). The row first appears in - // this commit, so created_at equals the new commit version. - new_version - } - }) - .collect(); - debug_assert_eq!(created_at_versions.len(), physical_rows); - - let runs = encode_version_runs(&created_at_versions); - let created_at_seq = RowDatasetVersionSequence { runs }; - fragment.created_at_version_meta = Some( - RowDatasetVersionMeta::from_sequence(&created_at_seq).map_err(|e| { - Error::internal(format!( - "Failed to create created_at version metadata: {}", - e - )) - })?, - ); - - fragment.last_updated_at_version_meta = build_version_meta(fragment, new_version); - } else { - let version_meta = build_version_meta(fragment, new_version); - fragment.last_updated_at_version_meta = version_meta.clone(); - fragment.created_at_version_meta = version_meta; - } - } - Ok(()) -} - -/// Run-length encode a sequence of per-row versions into [`RowDatasetVersionRun`]s. -fn encode_version_runs(versions: &[u64]) -> Vec { - if versions.is_empty() { - return Vec::new(); - } - let mut runs = Vec::new(); - let mut current_version = versions[0]; - let mut run_start = 0u64; - for (i, &version) in versions.iter().enumerate().skip(1) { - if version != current_version { - runs.push(RowDatasetVersionRun { - span: U64Segment::Range(run_start..i as u64), - version: current_version, - }); - current_version = version; - run_start = i as u64; - } - } - runs.push(RowDatasetVersionRun { - span: U64Segment::Range(run_start..versions.len() as u64), - version: current_version, - }); - runs -} - -/// A change to a dataset that can be retried -/// -/// This contains enough information to be able to build the next manifest, -/// given the current manifest. -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub struct Transaction { - /// The version of the table this transaction is based off of. If this is - /// the first transaction, this should be 0. - pub read_version: u64, - pub uuid: String, - pub operation: Operation, - pub tag: Option, - pub transaction_properties: Option>>, -} - -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub struct DataReplacementGroup(pub u64, pub DataFile); - -/// Overlay files to append to a single fragment, in order (the last entry is -/// newest). The overlays are appended to the fragment's existing `overlays` -/// list rather than replacing it, so overlays written by concurrent commits are -/// preserved. Each overlay's `committed_version` is stamped to the new dataset -/// version at commit time (re-stamped on retry). -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub struct DataOverlayGroup { - pub fragment_id: u64, - pub overlays: Vec, -} - -/// An entry for a map update. If value is None, the key will be removed from the map. -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub struct UpdateMapEntry { - /// The key of the map entry to update. - pub key: String, - /// The value to set for the key. - pub value: Option, -} - -impl From<(String, Option)> for UpdateMapEntry { - fn from((key, value): (String, Option)) -> Self { - Self { key, value } - } -} - -impl From<(String, String)> for UpdateMapEntry { - fn from((key, value): (String, String)) -> Self { - Self::from((key, Some(value))) - } -} - -impl From<(&str, Option<&str>)> for UpdateMapEntry { - fn from((key, value): (&str, Option<&str>)) -> Self { - Self { - key: key.to_string(), - value: value.map(str::to_owned), - } - } -} - -impl From<(&str, &str)> for UpdateMapEntry { - fn from((key, value): (&str, &str)) -> Self { - Self::from((key, Some(value))) - } -} - -/// Represents updates to a map (either incremental or replacement) -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub struct UpdateMap { - pub update_entries: Vec, - /// If true, the map will be replaced entirely with the new entries. - /// If false, the new entries will be merged with the existing map. - pub replace: bool, -} - -/// An operation on a dataset. -#[derive(Debug, Clone, DeepSizeOf)] -pub enum Operation { - /// Adding new fragments to the dataset. The fragments contained within - /// haven't yet been assigned a final ID. - Append { fragments: Vec }, - /// Updated fragments contain those that have been modified with new deletion - /// files. The deleted fragment IDs are those that should be removed from - /// the manifest. - Delete { - updated_fragments: Vec, - deleted_fragment_ids: Vec, - predicate: String, - }, - /// Overwrite the entire dataset with the given fragments. This is also - /// used when initially creating a table. - Overwrite { - fragments: Vec, - schema: Schema, - config_upsert_values: Option>, - initial_bases: Option>, - }, - /// A new index has been created. - CreateIndex { - /// The new secondary indices, - /// any existing indices with the same name will be replaced. - new_indices: Vec, - /// The indices that have been modified. - removed_indices: Vec, - }, - /// Data is rewritten but *not* modified. This is used for things like - /// compaction or re-ordering. Contains the old fragments and the new - /// ones that have been replaced. - /// - /// This operation will modify the row addresses of existing rows and - /// so any existing index covering a rewritten fragment will need to be - /// remapped. - Rewrite { - /// Groups of fragments that have been modified - groups: Vec, - /// Indices that have been updated with the new row addresses - rewritten_indices: Vec, - /// The fragment reuse index to be created or updated to - frag_reuse_index: Option, - }, - /// Replace data in a column in the dataset with new data. This is used for - /// null column population where we replace an entirely null column with a - /// new column that has data. - /// - /// This operation will only allow replacing files that contain the same schema - /// e.g. if the original files contain columns A, B, C and the new files contain - /// only columns A, B then the operation is not allowed. As we would need to split - /// the original files into two files, one with column A, B and the other with column C. - /// - /// Corollary to the above: the operation will also not allow replacing files unless the - /// affected columns all have the same datafile layout across the fragments being replaced. - /// - /// e.g. if fragments being replaced contain files with different schema layouts on - /// the column being replaced, the operation is not allowed. - /// say `frag_1: [A] [B, C]` and `frag_2: [A, B] [C]` and we are trying to replace column A - /// with a new column A, the operation is not allowed. - DataReplacement { - replacements: Vec, - }, - /// Attach overlay files to fragments, supplying new values for a subset of - /// `(physical offset, field)` cells without rewriting the fragments' base - /// data files. See [`DataOverlayFile`] and the Data Overlay Files - /// specification for resolution, coverage, and versioning rules. - DataOverlay { groups: Vec }, - /// Merge a new column in - /// 'fragments' is the final fragments include all data files, the new fragments must align with old ones at rows. - /// 'schema' is not forced to include existed columns, which means we could use Merge to drop column data - Merge { - fragments: Vec, - schema: Schema, - }, - /// Restore an old version of the database - Restore { version: u64 }, - /// Reserves fragment ids for future use - /// This can be used when row ids need to be known before a transaction - /// has been committed. It is used during a rewrite operation to allow - /// indices to be remapped to the new row ids as part of the operation. - ReserveFragments { num_fragments: u32 }, - - /// Update values in the dataset. - /// - /// Updates are generally vertical or horizontal. - /// - /// A vertical update adds new rows. In this case, the updated_fragments - /// will only have existing rows deleted and will not have any new fields added. - /// All new data will be contained in new_fragments. - /// This is what is used by a merge_insert that matches the whole schema and what - /// is used by the dataset updater. - /// - /// A horizontal update adds new columns. In this case, the updated fragments - /// may have fields removed or added. It is even possible for a field to be tombstoned - /// and then added back in the same update. (which is a field modification). If any - /// fields are modified in this way then they need to be added to the fields_modified list. - /// This way we can correctly update the indices. - /// This is what is used by a merge insert that does not match the whole schema. - Update { - /// Ids of fragments that have been moved - removed_fragment_ids: Vec, - /// Fragments that have been updated - updated_fragments: Vec, - /// Fragments that have been added - new_fragments: Vec, - /// The fields that have been modified - fields_modified: Vec, - /// MemWAL SSTables to mark as compacted after this transaction. - compacted_sstables: Vec, - /// The fields that used to judge whether to preserve the new frag's id into - /// the frag bitmap of the specified indices. - fields_for_preserving_frag_bitmap: Vec, - /// The mode of update - update_mode: Option, - /// Optional filter for detecting conflicts on inserted row keys. - /// Only tracks keys from INSERT operations during merge insert, not updates. - inserted_rows_filter: Option, - /// Physical row offsets (per fragment) that matched `update_columns` for RewriteColumns. - /// `None` means callers did not supply offsets; `build_manifest` skips partial refresh then. - updated_fragment_offsets: Option, - }, - - /// Project to a new schema. This only changes the schema, not the data. - Project { schema: Schema }, - - /// Update the dataset configuration. - UpdateConfig { - config_updates: Option, - table_metadata_updates: Option, - schema_metadata_updates: Option, - field_metadata_updates: HashMap, - }, - /// Update SSTable compaction progress in the MemWAL index. - /// - /// This is used during merge-insert to atomically record which - /// SSTables have been compacted into the base table. - UpdateMemWalState { - compacted_sstables: Vec, - }, - - /// Clone a dataset. - Clone { - is_shallow: bool, - ref_name: Option, - ref_version: u64, - ref_path: String, - branch_name: Option, - }, - - // Update base paths in the dataset (currently only supports adding new bases). - UpdateBases { - /// The new base paths to add to the manifest. - new_bases: Vec, - }, -} - -#[derive(Debug, Clone, PartialEq, DeepSizeOf)] -pub enum UpdateMode { - /// rows are deleted in current fragments and rewritten in new fragments. - /// This is most optimal when the majority of columns are being rewritten - /// or only a few rows are being updated. - RewriteRows, - - /// within each fragment, columns are fully rewritten and inserted as new data files. - /// Old versions of columns are tombstoned. This is most optimal when most rows are affected - /// but a small subset of columns are affected. - RewriteColumns, -} - -/// Matched physical row offsets per fragment for a partial [`UpdateMode::RewriteColumns`] update. -/// -/// Used with stable row IDs so `build_manifest` can refresh row-level version -/// metadata only for rows that were rewritten. -#[derive(Debug, Clone, PartialEq, Eq, Default)] -pub struct UpdatedFragmentOffsets(pub HashMap); - -impl DeepSizeOf for UpdatedFragmentOffsets { - fn deep_size_of_children(&self, context: &mut lance_core::deepsize::Context) -> usize { - self.0.iter().fold(0_usize, |acc, (frag_id, bitmap)| { - acc + frag_id.deep_size_of_children(context) - + (bitmap.len() as usize).saturating_mul(std::mem::size_of::()) - }) - } -} - -impl std::fmt::Display for Operation { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::Append { .. } => write!(f, "Append"), - Self::Delete { .. } => write!(f, "Delete"), - Self::Overwrite { .. } => write!(f, "Overwrite"), - Self::CreateIndex { .. } => write!(f, "CreateIndex"), - Self::Rewrite { .. } => write!(f, "Rewrite"), - Self::Merge { .. } => write!(f, "Merge"), - Self::Restore { .. } => write!(f, "Restore"), - Self::ReserveFragments { .. } => write!(f, "ReserveFragments"), - Self::Update { .. } => write!(f, "Update"), - Self::Project { .. } => write!(f, "Project"), - Self::UpdateConfig { .. } => write!(f, "UpdateConfig"), - Self::DataReplacement { .. } => write!(f, "DataReplacement"), - Self::DataOverlay { .. } => write!(f, "DataOverlay"), - Self::Clone { .. } => write!(f, "Clone"), - Self::UpdateMemWalState { .. } => write!(f, "UpdateMemWalState"), - Self::UpdateBases { .. } => write!(f, "UpdateBases"), - } - } -} - -impl From<&Transaction> for lance_table::format::Transaction { - fn from(value: &Transaction) -> Self { - let pb_transaction: pb::Transaction = value.into(); - Self { - inner: pb_transaction, - } - } -} - -impl PartialEq for Operation { - fn eq(&self, other: &Self) -> bool { - // Many of the operations contain `Vec` where the order of the - // elements don't matter. So we need to compare them in a way that - // ignores the order of the elements. - // TODO: we can make it so the vecs are always constructed in order. - // Then we can use `==` instead of `compare_vec`. - fn compare_vec(a: &[T], b: &[T]) -> bool { - a.len() == b.len() && a.iter().all(|f| b.contains(f)) - } - match (self, other) { - (Self::Append { fragments: a }, Self::Append { fragments: b }) => compare_vec(a, b), - ( - Self::Clone { - is_shallow: a_is_shallow, - ref_name: a_ref_name, - ref_version: a_ref_version, - ref_path: a_source_path, - branch_name: a_branch_name, - }, - Self::Clone { - is_shallow: b_is_shallow, - ref_name: b_ref_name, - ref_version: b_ref_version, - ref_path: b_source_path, - branch_name: b_branch_name, - }, - ) => { - a_is_shallow == b_is_shallow - && a_ref_name == b_ref_name - && a_ref_version == b_ref_version - && a_source_path == b_source_path - && a_branch_name == b_branch_name - } - ( - Self::Delete { - updated_fragments: a_updated, - deleted_fragment_ids: a_deleted, - predicate: a_predicate, - }, - Self::Delete { - updated_fragments: b_updated, - deleted_fragment_ids: b_deleted, - predicate: b_predicate, - }, - ) => { - compare_vec(a_updated, b_updated) - && compare_vec(a_deleted, b_deleted) - && a_predicate == b_predicate - } - ( - Self::Overwrite { - fragments: a_fragments, - schema: a_schema, - config_upsert_values: a_config, - initial_bases: a_initial, - }, - Self::Overwrite { - fragments: b_fragments, - schema: b_schema, - config_upsert_values: b_config, - initial_bases: b_initial, - }, - ) => { - compare_vec(a_fragments, b_fragments) - && a_schema == b_schema - && a_config == b_config - && a_initial == b_initial - } - ( - Self::CreateIndex { - new_indices: a_new, - removed_indices: a_removed, - }, - Self::CreateIndex { - new_indices: b_new, - removed_indices: b_removed, - }, - ) => compare_vec(a_new, b_new) && compare_vec(a_removed, b_removed), - ( - Self::Rewrite { - groups: a_groups, - rewritten_indices: a_indices, - frag_reuse_index: a_frag_reuse_index, - }, - Self::Rewrite { - groups: b_groups, - rewritten_indices: b_indices, - frag_reuse_index: b_frag_reuse_index, - }, - ) => { - compare_vec(a_groups, b_groups) - && compare_vec(a_indices, b_indices) - && a_frag_reuse_index == b_frag_reuse_index - } - ( - Self::Merge { - fragments: a_fragments, - schema: a_schema, - }, - Self::Merge { - fragments: b_fragments, - schema: b_schema, - }, - ) => compare_vec(a_fragments, b_fragments) && a_schema == b_schema, - (Self::Restore { version: a }, Self::Restore { version: b }) => a == b, - ( - Self::ReserveFragments { num_fragments: a }, - Self::ReserveFragments { num_fragments: b }, - ) => a == b, - ( - Self::Update { - removed_fragment_ids: a_removed, - updated_fragments: a_updated, - new_fragments: a_new, - fields_modified: a_fields, - compacted_sstables: a_compacted_sstables, - fields_for_preserving_frag_bitmap: a_fields_for_preserving_frag_bitmap, - update_mode: a_update_mode, - inserted_rows_filter: a_inserted_rows_filter, - updated_fragment_offsets: a_updated_fragment_offsets, - }, - Self::Update { - removed_fragment_ids: b_removed, - updated_fragments: b_updated, - new_fragments: b_new, - fields_modified: b_fields, - compacted_sstables: b_compacted_sstables, - fields_for_preserving_frag_bitmap: b_fields_for_preserving_frag_bitmap, - update_mode: b_update_mode, - inserted_rows_filter: b_inserted_rows_filter, - updated_fragment_offsets: b_updated_fragment_offsets, - }, - ) => { - compare_vec(a_removed, b_removed) - && compare_vec(a_updated, b_updated) - && compare_vec(a_new, b_new) - && compare_vec(a_fields, b_fields) - && compare_vec(a_compacted_sstables, b_compacted_sstables) - && compare_vec( - a_fields_for_preserving_frag_bitmap, - b_fields_for_preserving_frag_bitmap, - ) - && a_update_mode == b_update_mode - && a_inserted_rows_filter == b_inserted_rows_filter - && a_updated_fragment_offsets == b_updated_fragment_offsets - } - (Self::Project { schema: a }, Self::Project { schema: b }) => a == b, - ( - Self::UpdateConfig { - config_updates: a_config, - table_metadata_updates: a_table_metadata, - schema_metadata_updates: a_schema, - field_metadata_updates: a_field, - }, - Self::UpdateConfig { - config_updates: b_config, - table_metadata_updates: b_table_metadata, - schema_metadata_updates: b_schema, - field_metadata_updates: b_field, - }, - ) => { - a_config == b_config - && a_table_metadata == b_table_metadata - && a_schema == b_schema - && a_field == b_field - } - ( - Self::DataReplacement { replacements: a }, - Self::DataReplacement { replacements: b }, - ) => a.len() == b.len() && a.iter().all(|r| b.contains(r)), - // Handle all remaining combinations. - // We spell out all combinations explicitly to prevent - // us accidentally handling a new case in the wrong way. - (Self::Append { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Append { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Delete { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Overwrite { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::CreateIndex { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Rewrite { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Merge { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Restore { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::ReserveFragments { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Update { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Project { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::UpdateConfig { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::DataReplacement { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::UpdateMemWalState { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - ( - Self::UpdateMemWalState { - compacted_sstables: a_compacted, - }, - Self::UpdateMemWalState { - compacted_sstables: b_compacted, - }, - ) => compare_vec(a_compacted, b_compacted), - (Self::Clone { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::UpdateBases { new_bases: a }, Self::UpdateBases { new_bases: b }) => { - compare_vec(a, b) - } - - (Self::UpdateBases { .. }, Self::Append { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Delete { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Overwrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::CreateIndex { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Rewrite { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Merge { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Restore { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::ReserveFragments { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Update { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Project { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::UpdateConfig { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::DataReplacement { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::UpdateMemWalState { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateBases { .. }, Self::Clone { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - - (Self::Append { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Delete { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Overwrite { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::CreateIndex { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Rewrite { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Merge { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Restore { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::ReserveFragments { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Update { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Project { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateConfig { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataReplacement { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::UpdateMemWalState { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::Clone { .. }, Self::UpdateBases { .. }) => { - std::mem::discriminant(self) == std::mem::discriminant(other) - } - (Self::DataOverlay { groups: a }, Self::DataOverlay { groups: b }) => compare_vec(a, b), - (Self::DataOverlay { .. }, _) | (_, Self::DataOverlay { .. }) => false, - } - } -} - -#[derive(Debug, Clone, PartialEq)] -pub struct RewrittenIndex { - pub old_id: Uuid, - pub new_id: Uuid, - pub new_index_details: prost_types::Any, - pub new_index_version: u32, - /// Files in the new index with their sizes. - /// Empty list from older writers that didn't persist this field. - pub new_index_files: Option>, -} - -impl DeepSizeOf for RewrittenIndex { - fn deep_size_of_children(&self, context: &mut lance_core::deepsize::Context) -> usize { - self.new_index_details - .type_url - .deep_size_of_children(context) - + self.new_index_details.value.deep_size_of_children(context) - } -} - -#[derive(Debug, Clone, DeepSizeOf)] -pub struct RewriteGroup { - pub old_fragments: Vec, - pub new_fragments: Vec, -} - -impl PartialEq for RewriteGroup { - fn eq(&self, other: &Self) -> bool { - fn compare_vec(a: &[T], b: &[T]) -> bool { - a.len() == b.len() && a.iter().all(|f| b.contains(f)) - } - compare_vec(&self.old_fragments, &other.old_fragments) - && compare_vec(&self.new_fragments, &other.new_fragments) - } -} - -impl Operation { - /// Returns the config keys that have been upserted by this operation. - fn get_upsert_config_keys(&self) -> Vec { - match self { - Self::Overwrite { - config_upsert_values: Some(upsert_values), - .. - } => { - let vec: Vec = upsert_values.keys().cloned().collect(); - vec - } - Self::UpdateConfig { - config_updates: Some(config_updates), - .. - } => config_updates - .update_entries - .iter() - .filter_map(|entry| { - if entry.value.is_some() { - Some(entry.key.clone()) - } else { - None - } - }) - .collect(), - _ => Vec::::new(), - } - } - - /// Returns the config keys that have been deleted by this operation. - fn get_delete_config_keys(&self) -> Vec { - match self { - Self::UpdateConfig { - config_updates: Some(config_updates), - .. - } => config_updates - .update_entries - .iter() - .filter_map(|entry| { - if entry.value.is_none() { - Some(entry.key.clone()) - } else { - None - } - }) - .collect(), - _ => Vec::::new(), - } - } - - pub(crate) fn modifies_same_metadata(&self, other: &Self) -> bool { - match (self, other) { - ( - Self::UpdateConfig { - table_metadata_updates, - schema_metadata_updates, - field_metadata_updates, - .. - }, - Self::UpdateConfig { - table_metadata_updates: other_table_metadata, - schema_metadata_updates: other_schema_metadata, - field_metadata_updates: other_field_metadata, - .. - }, - ) => { - if Self::update_maps_conflict( - table_metadata_updates.as_ref(), - other_table_metadata.as_ref(), - ) { - return true; - } - if schema_metadata_updates.is_some() && other_schema_metadata.is_some() { - return true; - } - if !field_metadata_updates.is_empty() && !other_field_metadata.is_empty() { - for field in field_metadata_updates.keys() { - if other_field_metadata.contains_key(field) { - return true; - } - } - } - false - } - _ => false, - } - } - - fn update_maps_conflict(left: Option<&UpdateMap>, right: Option<&UpdateMap>) -> bool { - let (Some(left), Some(right)) = (left, right) else { - return false; - }; - if left.replace || right.replace { - return true; - } - let left_keys = left - .update_entries - .iter() - .map(|entry| entry.key.as_str()) - .collect::>(); - right - .update_entries - .iter() - .any(|entry| left_keys.contains(entry.key.as_str())) - } - - /// Check whether another operation upserts a key that is referenced by another operation - pub(crate) fn upsert_key_conflict(&self, other: &Self) -> bool { - let self_upsert_keys = self.get_upsert_config_keys(); - let other_upsert_keys = other.get_upsert_config_keys(); - - let self_delete_keys = self.get_delete_config_keys(); - let other_delete_keys = other.get_delete_config_keys(); - - self_upsert_keys - .iter() - .any(|x| other_upsert_keys.contains(x) || other_delete_keys.contains(x)) - || other_upsert_keys - .iter() - .any(|x| self_upsert_keys.contains(x) || self_delete_keys.contains(x)) - } - - pub fn name(&self) -> &str { - match self { - Self::Append { .. } => "Append", - Self::Delete { .. } => "Delete", - Self::Overwrite { .. } => "Overwrite", - Self::CreateIndex { .. } => "CreateIndex", - Self::Rewrite { .. } => "Rewrite", - Self::Merge { .. } => "Merge", - Self::ReserveFragments { .. } => "ReserveFragments", - Self::Restore { .. } => "Restore", - Self::Update { .. } => "Update", - Self::Project { .. } => "Project", - Self::UpdateConfig { .. } => "UpdateConfig", - Self::DataReplacement { .. } => "DataReplacement", - Self::DataOverlay { .. } => "DataOverlay", - Self::UpdateMemWalState { .. } => "UpdateMemWalState", - Self::Clone { .. } => "Clone", - Self::UpdateBases { .. } => "UpdateBases", - } - } -} - -/// Helper function to apply UpdateMap changes to a HashMap -fn apply_update_map( - target: &mut std::collections::HashMap, - update_map: &UpdateMap, -) { - if update_map.replace { - // Full replacement - clear existing and replace with new entries that have values - target.clear(); - for entry in &update_map.update_entries { - if let Some(value) = &entry.value { - target.insert(entry.key.clone(), value.clone()); - } - } - } else { - // Incremental update - merge entries - for entry in &update_map.update_entries { - if let Some(value) = &entry.value { - target.insert(entry.key.clone(), value.clone()); - } else { - target.remove(&entry.key); - } - } - } -} - -/// Helper function to translate old-style config updates to new UpdateMap format -pub fn translate_config_updates( - upsert_values: &std::collections::HashMap, - delete_keys: &[String], -) -> UpdateMap { - let mut update_entries = Vec::new(); - - // Add upsert entries (with values) - for (key, value) in upsert_values { - update_entries.push(UpdateMapEntry { - key: key.clone(), - value: Some(value.clone()), - }); - } - - // Add delete entries (without values) - for key in delete_keys { - update_entries.push(UpdateMapEntry { - key: key.clone(), - value: None, - }); - } - - UpdateMap { - update_entries, - replace: false, // Old style was always incremental - } -} - -/// Helper function to translate old-style schema metadata to new UpdateMap format -pub fn translate_schema_metadata_updates( - schema_metadata: &std::collections::HashMap, -) -> UpdateMap { - let update_entries = schema_metadata - .iter() - .map(|(key, value)| UpdateMapEntry { - key: key.clone(), - value: Some(value.clone()), - }) - .collect(); - - UpdateMap { - update_entries, - replace: true, // Old style schema metadata was full replacement - } -} - -impl From<&UpdateMap> for pb::transaction::UpdateMap { - fn from(update_map: &UpdateMap) -> Self { - Self { - update_entries: update_map - .update_entries - .iter() - .map(|entry| pb::transaction::UpdateMapEntry { - key: entry.key.clone(), - value: entry.value.clone(), - }) - .collect(), - replace: update_map.replace, - } - } -} - -impl From<&pb::transaction::UpdateMap> for UpdateMap { - fn from(pb_update_map: &pb::transaction::UpdateMap) -> Self { - Self { - update_entries: pb_update_map - .update_entries - .iter() - .map(|entry| UpdateMapEntry { - key: entry.key.clone(), - value: entry.value.clone(), - }) - .collect(), - replace: pb_update_map.replace, - } - } -} - -/// Add TransactionBuilder for flexibly setting option without using `mut` -pub struct TransactionBuilder { - read_version: u64, - // uuid is optional for builder since it can autogenerate - uuid: Option, - operation: Operation, - tag: Option, - transaction_properties: Option>>, -} - -impl TransactionBuilder { - pub fn new(read_version: u64, operation: Operation) -> Self { - Self { - read_version, - uuid: None, - operation, - tag: None, - transaction_properties: None, - } - } - - pub fn uuid(mut self, uuid: String) -> Self { - self.uuid = Some(uuid); - self - } - - pub fn tag(mut self, tag: Option) -> Self { - self.tag = tag; - self - } - - pub fn transaction_properties( - mut self, - transaction_properties: Option>>, - ) -> Self { - self.transaction_properties = transaction_properties; - self - } - - pub fn build(self) -> Transaction { - let uuid = self - .uuid - .unwrap_or_else(|| Uuid::new_v4().hyphenated().to_string()); - Transaction { - read_version: self.read_version, - uuid, - operation: self.operation, - tag: self.tag, - transaction_properties: self.transaction_properties, - } - } -} - -impl Transaction { - pub fn new_from_version(read_version: u64, operation: Operation) -> Self { - TransactionBuilder::new(read_version, operation).build() - } - - pub fn new(read_version: u64, operation: Operation, tag: Option) -> Self { - TransactionBuilder::new(read_version, operation) - .tag(tag) - .build() - } - - fn fragments_with_ids<'a, T>( - new_fragments: T, - fragment_id: &'a mut u64, - ) -> impl Iterator + 'a - where - T: IntoIterator + 'a, - { - new_fragments.into_iter().map(move |mut f| { - if f.id == 0 { - f.id = *fragment_id; - *fragment_id += 1; - } - f - }) - } - - fn data_storage_format_from_files( - fragments: &[Fragment], - user_requested: Option, - ) -> Result { - if let Some(file_version) = Fragment::try_infer_version(fragments)? { - // Ensure user-requested matches data files - if let Some(user_requested) = user_requested - && user_requested != file_version - { - return Err(Error::invalid_input(format!( - "User requested data storage version ({}) does not match version in data files ({})", - user_requested, file_version - ))); - } - Ok(DataStorageFormat::new(file_version)) - } else { - // If no files use user-requested or default - Ok(user_requested - .map(DataStorageFormat::new) - .unwrap_or_default()) - } - } - - pub(crate) async fn restore_old_manifest( - object_store: &ObjectStore, - commit_handler: &dyn CommitHandler, - base_path: &Path, - version: u64, - config: &ManifestWriteConfig, - tx_path: &str, - current_manifest: &Manifest, - ) -> Result<(Manifest, Vec)> { - let location = commit_handler - .resolve_version_location(base_path, version, &object_store.inner) - .await?; - let mut manifest = read_manifest(object_store, &location.path, location.size).await?; - manifest.set_timestamp(timestamp_to_nanos(config.timestamp)); - manifest.transaction_file = Some(tx_path.to_string()); - let indices = read_manifest_indexes(object_store, &location, &manifest).await?; - manifest.max_fragment_id = manifest - .max_fragment_id - .max(current_manifest.max_fragment_id); - Ok((manifest, indices)) - } - - /// Create a new manifest from the current manifest and the transaction. - /// - /// `current_manifest` should only be None if the dataset does not yet exist. - pub(crate) fn build_manifest( - &self, - current_manifest: Option<&Manifest>, - current_indices: Vec, - transaction_file_path: &str, - config: &ManifestWriteConfig, - ) -> Result<(Manifest, Vec)> { - if config.use_stable_row_ids - && current_manifest - .map(|m| !m.uses_stable_row_ids()) - .unwrap_or_default() - { - return Err(Error::not_supported_source( - "Cannot enable stable row ids on existing dataset".into(), - )); - } - let mut reference_paths = match current_manifest { - Some(m) => m.base_paths.clone(), - None => HashMap::new(), - }; - - if let Operation::Overwrite { - initial_bases: Some(initial_bases), - .. - } = &self.operation - { - if current_manifest.is_none() { - // CREATE mode: registering base paths - // Base IDs should have been assigned during write operation - // Validate uniqueness and insert them into the manifest - for base_path in initial_bases.iter() { - if reference_paths.contains_key(&base_path.id) { - return Err(Error::invalid_input(format!( - "Duplicate base path ID {} detected. Base path IDs must be unique.", - base_path.id - ))); - } - reference_paths.insert(base_path.id, base_path.clone()); - } - } else { - // OVERWRITE mode with initial_bases should have been rejected by validation - // This branch should never be reached - return Err(Error::invalid_input( - "OVERWRITE mode cannot register new bases. This should have been caught by validation.", - )); - } - } - - // Get the schema and the final fragment list - let schema = match self.operation { - Operation::Overwrite { ref schema, .. } => schema.clone(), - Operation::Merge { ref schema, .. } => schema.clone(), - Operation::Project { ref schema, .. } => schema.clone(), - _ => { - if let Some(current_manifest) = current_manifest { - current_manifest.schema.clone() - } else { - return Err(Error::internal( - "Cannot create a new dataset without a schema".to_string(), - )); - } - } - }; - - let mut fragment_id = if matches!(self.operation, Operation::Overwrite { .. }) { - 0 - } else { - current_manifest - .and_then(|m| m.max_fragment_id()) - .map(|id| id + 1) - .unwrap_or(0) - }; - let mut final_fragments = Vec::new(); - let mut final_indices = current_indices; - - let mut next_row_id = { - // Only use row ids if the feature flag is set already or - match (current_manifest, config.use_stable_row_ids) { - (Some(manifest), _) if manifest.reader_feature_flags & FLAG_STABLE_ROW_IDS != 0 => { - Some(manifest.next_row_id) - } - (None, true) => Some(0), - (_, false) => None, - (Some(_), true) => { - return Err(Error::not_supported_source( - "Cannot enable stable row ids on existing dataset".into(), - )); - } - } - }; - - let maybe_existing_fragments = - current_manifest - .map(|m| m.fragments.as_ref()) - .ok_or_else(|| { - Error::internal(format!( - "No current manifest was provided while building manifest for operation {}", - self.operation.name() - )) - }); - - let new_version = current_manifest.map_or(1, |m| m.version + 1); - - match &self.operation { - Operation::Clone { .. } => { - return Err(Error::internal( - "Clone operation should not enter build_manifest.".to_string(), - )); - } - Operation::Append { fragments } => { - final_fragments.extend(maybe_existing_fragments?.clone()); - let mut new_fragments = - Self::fragments_with_ids(fragments.clone(), &mut fragment_id) - .collect::>(); - if let Some(next_row_id) = &mut next_row_id { - Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; - // Add version metadata for all new fragments - for fragment in new_fragments.iter_mut() { - let version_meta = build_version_meta(fragment, new_version); - fragment.last_updated_at_version_meta = version_meta.clone(); - fragment.created_at_version_meta = version_meta; - } - } - final_fragments.extend(new_fragments); - } - Operation::Delete { - updated_fragments, - deleted_fragment_ids, - .. - } => { - // Remove the deleted fragments - final_fragments.extend(maybe_existing_fragments?.clone()); - final_fragments.retain(|f| !deleted_fragment_ids.contains(&f.id)); - final_fragments.iter_mut().for_each(|f| { - for updated in updated_fragments { - if updated.id == f.id { - *f = updated.clone(); - } - } - }); - Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments) - } - Operation::Update { - removed_fragment_ids, - updated_fragments, - new_fragments, - fields_modified, - compacted_sstables, - fields_for_preserving_frag_bitmap, - update_mode, - updated_fragment_offsets, - .. - } => { - // Extract existing fragments once for reuse - let existing_fragments = maybe_existing_fragments?; - - // Apply updates to existing fragments - let updated_frags: Vec = existing_fragments - .iter() - .filter_map(|f| { - if removed_fragment_ids.contains(&f.id) { - return None; - } - if let Some(updated) = updated_fragments.iter().find(|uf| uf.id == f.id) { - let mut updated = updated.clone(); - // Carry forward the fragment's current overlays (which - // may include ones added by a concurrent commit). An - // in-place column rewrite then tombstones the overlaid - // fields it rewrote, since the fresh base values - // supersede them. - updated.overlays = f.overlays.clone(); - if matches!(update_mode, Some(RewriteColumns)) { - lance_table::format::overlay::tombstone_overlay_fields( - &mut updated.overlays, - fields_modified, - ); - } - Some(updated) - } else { - Some(f.clone()) - } - }) - .collect(); - - // Update version metadata for updated fragments if stable row IDs are enabled - // Note: We don't update version metadata for fragments with deletion vectors - // because the version sequences are indexed by physical row position, not logical position. - // Version metadata for deleted rows will be filtered out during scan using the deletion vector. - if next_row_id.is_some() { - // Version metadata will be properly set during compaction when deletions are materialized - } - - final_fragments.extend(updated_frags); - - if next_row_id.is_some() - && matches!(update_mode, Some(RewriteColumns)) - && let Some(UpdatedFragmentOffsets(off_map)) = updated_fragment_offsets - && !off_map.is_empty() - { - let prev_version = current_manifest.map(|m| m.version).unwrap_or(0); - for fragment in final_fragments.iter_mut() { - let Some(bitmap) = off_map.get(&fragment.id) else { - continue; - }; - if bitmap.is_empty() { - continue; - } - // Skip fragments with no existing version metadata: the helper - // would fill unmatched rows with prev_version, fabricating a - // last_updated stamp for rows that never had one. - if fragment.last_updated_at_version_meta.is_none() { - continue; - } - let offsets: Vec = bitmap.iter().map(|o| o as usize).collect(); - lance_table::rowids::version::refresh_row_latest_update_meta_for_partial_frag_rewrite_cols( - fragment, - &offsets, - new_version, - prev_version, - )?; - } - } - - // If we updated any fields, remove those fragments from indices covering those fields - Self::prune_updated_fields_from_indices( - &mut final_indices, - updated_fragments, - fields_modified, - ); - - let mut new_fragments = - Self::fragments_with_ids(new_fragments.clone(), &mut fragment_id) - .collect::>(); - - // Assign row IDs to any fragments that don't have them yet - // (e.g., inserted rows from merge_insert operations) - if let Some(next_row_id) = &mut next_row_id { - Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; - } - - if next_row_id.is_some() { - resolve_update_version_metadata( - existing_fragments, - new_fragments.as_mut_slice(), - new_version, - )?; - } - - if config.use_stable_row_ids - && update_mode.is_some() - && *update_mode == Some(RewriteRows) - { - let pure_updated_frag_ids = - Self::collect_pure_rewrite_row_update_frags_ids(&new_fragments)?; - - // collect all the original frag ids that contains the updated rows - let original_fragment_ids: Vec = removed_fragment_ids - .iter() - .chain(updated_fragments.iter().map(|f| &f.id)) - .copied() - .collect(); - - // The original fragments that carried an overlay: their moved rows may have a - // stale index entry (see `register_pure_rewrite_rows_update_frags_in_indices`). - let original_overlaid_frags: HashMap = existing_fragments - .iter() - .filter(|f| original_fragment_ids.contains(&f.id) && !f.overlays.is_empty()) - .map(|f| (f.id as u32, f)) - .collect(); - - Self::register_pure_rewrite_rows_update_frags_in_indices( - &mut final_indices, - &pure_updated_frag_ids, - &original_fragment_ids, - fields_for_preserving_frag_bitmap, - &original_overlaid_frags, - &schema, - )?; - } - - if let Some(next_row_id) = &mut next_row_id { - Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; - // Note: Version metadata is already set above (lines 1627-1755) - // for Update operations, preserving created_at from original fragments. - // Don't overwrite it here. - } - // Identify fragments that were updated or newly created in this update - let mut target_ids: HashSet = HashSet::new(); - target_ids.extend(new_fragments.iter().map(|f| f.id)); - final_fragments.extend(new_fragments); - Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments); - - if !compacted_sstables.is_empty() { - update_mem_wal_index_compacted_sstables( - &mut final_indices, - new_version, - compacted_sstables.clone(), - )?; - } - } - Operation::Overwrite { fragments, .. } => { - let mut new_fragments = - Self::fragments_with_ids(fragments.clone(), &mut fragment_id) - .collect::>(); - if let Some(next_row_id) = &mut next_row_id { - Self::assign_row_ids(next_row_id, new_fragments.as_mut_slice())?; - // Add version metadata for all new fragments - for fragment in new_fragments.iter_mut() { - let version_meta = build_version_meta(fragment, new_version); - fragment.last_updated_at_version_meta = version_meta.clone(); - fragment.created_at_version_meta = version_meta; - } - } - final_fragments.extend(new_fragments); - final_indices = Vec::new(); - } - Operation::Rewrite { - groups, - rewritten_indices, - frag_reuse_index, - } => { - final_fragments.extend(maybe_existing_fragments?.clone()); - let current_version = current_manifest.map(|m| m.version).unwrap_or_default(); - Self::handle_rewrite_fragments( - &mut final_fragments, - groups, - &mut fragment_id, - current_version, - next_row_id.as_ref(), - )?; - - if next_row_id.is_some() { - // We can re-use indices, but need to rewrite the fragment bitmaps - debug_assert!(rewritten_indices.is_empty()); - for index in final_indices.iter_mut() { - if let Some(fragment_bitmap) = &mut index.fragment_bitmap { - *fragment_bitmap = - Self::recalculate_fragment_bitmap(fragment_bitmap, groups)?; - } - } - } else { - Self::handle_rewrite_indices(&mut final_indices, rewritten_indices, groups)?; - } - - // A full compaction materializes a fragment's overlays into fresh - // base data. Any index older than one of those overlays was built on - // the pre-overlay values, so drop the rewritten fragment from its - // coverage to keep it from serving stale values. - Self::prune_overlay_stale_fields_from_indices(&mut final_indices, groups); - - if let Some(frag_reuse_index) = frag_reuse_index { - final_indices.retain(|idx| idx.name != frag_reuse_index.name); - final_indices.push(frag_reuse_index.clone()); - } - } - Operation::CreateIndex { - new_indices, - removed_indices, - } => { - final_fragments.extend(maybe_existing_fragments?.clone()); - let removed_uuids = removed_indices - .iter() - .map(|old_index| old_index.uuid) - .collect::>(); - let new_uuids = new_indices - .iter() - .map(|new_index| new_index.uuid) - .collect::>(); - final_indices.retain(|existing_index| { - !removed_uuids.contains(&existing_index.uuid) - && !new_uuids.contains(&existing_index.uuid) - }); - final_indices.extend(new_indices.clone()); - } - Operation::ReserveFragments { .. } | Operation::UpdateConfig { .. } => { - final_fragments.extend(maybe_existing_fragments?.clone()); - } - Operation::Merge { fragments, .. } => { - let existing_fragments = maybe_existing_fragments?; - let mut merged_fragments = fragments.clone(); - if next_row_id.is_some() { - let prev_by_id: HashMap = - existing_fragments.iter().map(|f| (f.id, f)).collect(); - for fragment in merged_fragments.iter_mut() { - match prev_by_id.get(&fragment.id) { - Some(prev) => { - if merge_fragment_physically_rewritten(prev, fragment) { - lance_table::rowids::version::refresh_row_latest_update_meta_for_full_frag_rewrite_cols( - fragment, - new_version, - )?; - } - } - None => { - // Brand-new fragment ID not present in the previous manifest. - // Set both last_updated and created version meta, consistent - // with Append/Overwrite for genuinely new fragments. - lance_table::rowids::version::refresh_row_latest_update_meta_for_full_frag_rewrite_cols( - fragment, - new_version, - )?; - fragment.created_at_version_meta = - fragment.last_updated_at_version_meta.clone(); - } - } - } - } - final_fragments.extend(merged_fragments); - - // A Merge can rewrite a column's data file in place; the field stays - // in the schema, so the index is retained -- prune its now-stale - // entries for the rewritten fragments. - Self::prune_merge_rewritten_fields_from_indices( - &mut final_indices, - existing_fragments, - fragments, - ); - - // Some fields that have indices may have been removed, so we should - // remove those indices as well. - Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments) - } - Operation::Project { .. } => { - final_fragments.extend(maybe_existing_fragments?.clone()); - - // We might have removed all fields for certain data files, so - // we should remove the data files that are no longer relevant. - let remaining_field_ids = schema - .fields_pre_order() - .map(|f| f.id) - .collect::>(); - for fragment in final_fragments.iter_mut() { - fragment.files.retain(|file| { - file.fields - .iter() - .any(|field_id| remaining_field_ids.contains(field_id)) - }); - } - - // Some fields that have indices may have been removed, so we should - // remove those indices as well. - Self::retain_relevant_indices(&mut final_indices, &schema, &final_fragments) - } - Operation::Restore { .. } => { - unreachable!() - } - Operation::DataReplacement { replacements } => { - log::warn!( - "Building manifest with DataReplacement operation. This operation is not stable yet, please use with caution." - ); - - let (old_fragment_ids, new_datafiles): (Vec<&u64>, Vec<&DataFile>) = replacements - .iter() - .map(|DataReplacementGroup(fragment_id, new_file)| (fragment_id, new_file)) - .unzip(); - - // 1. make sure the new files all have the same fields / or empty - // NOTE: arguably this requirement could be relaxed in the future - // for the sake of simplicity, we require the new files to have the same fields - if new_datafiles - .iter() - .map(|f| f.fields.clone()) - .collect::>() - .len() - > 1 - { - let field_info = new_datafiles - .iter() - .enumerate() - .map(|(id, f)| (id, f.fields.clone())) - .fold("".to_string(), |acc, (id, fields)| { - format!("{}File {}: {:?}\n", acc, id, fields) - }); - - return Err(Error::invalid_input(format!( - "All new data files must have the same fields, but found different fields:\n{field_info}" - ))); - } - - let existing_fragments = maybe_existing_fragments?; - - // Collect replaced field IDs before consuming new_datafiles - let replaced_fields: Vec = new_datafiles - .first() - .map(|f| { - f.fields - .iter() - .filter(|&&id| id >= 0) - .map(|&id| id as u32) - .collect() - }) - .unwrap_or_default(); - - // 2. check that the fragments being modified have isomorphic layouts along the columns being replaced - // 3. add modified fragments to final_fragments - for (frag_id, new_file) in old_fragment_ids.iter().zip(new_datafiles) { - let frag = existing_fragments - .iter() - .find(|f| f.id == **frag_id) - .ok_or_else(|| { - Error::invalid_input( - "Fragment being replaced not found in existing fragments", - ) - })?; - let mut new_frag = frag.clone(); - - // TODO(rmeng): check new file and fragment are the same length - - let mut columns_covered = HashSet::new(); - for file in &mut new_frag.files { - if file.fields == new_file.fields - && file.file_major_version == new_file.file_major_version - && file.file_minor_version == new_file.file_minor_version - { - // assign the new file path / size / base to the fragment - file.path = new_file.path.clone(); - file.file_size_bytes = new_file.file_size_bytes.clone(); - file.base_id = new_file.base_id; - } - columns_covered.extend(file.fields.iter()); - } - // SPECIAL CASE: if the column(s) being replaced are not covered by the fragment - // Then it means it's a all-NULL column that is being replaced with real data - // just add it to the final fragments. Push the DataFile as - // given so every field (including base_id) is preserved. - if columns_covered.is_disjoint(&new_file.fields.iter().collect()) { - LanceFileVersion::try_from_major_minor( - new_file.file_major_version, - new_file.file_minor_version, - ) - .expect("Expected valid file version"); - new_frag.files.push(new_file.clone()); - } - - // Nothing changed in the current fragment, which is not expected -- error out - if &new_frag == frag { - return Err(Error::invalid_input( - "Expected to modify the fragment but no changes were made. This means the new data files does not align with any exiting datafiles. Please check if the schema of the new data files matches the schema of the old data files including the file major and minor versions", - )); - } - - // New base values for these fields supersede any overlay - // still shadowing them; tombstone the overlaid fields so the - // replacement is not silently masked. - lance_table::format::overlay::tombstone_overlay_fields( - &mut new_frag.overlays, - &replaced_fields, - ); - - final_fragments.push(new_frag); - } - - let fragments_changed = old_fragment_ids - .iter() - .cloned() - .cloned() - .collect::>(); - - // 4. push fragments that didn't change back to final_fragments - let unmodified_fragments = existing_fragments - .iter() - .filter(|f| !fragments_changed.contains(&f.id)) - .cloned() - .collect::>(); - - final_fragments.extend(unmodified_fragments); - - // 5. Invalidate index bitmaps for replaced fields - let modified_fragments: Vec = final_fragments - .iter() - .filter(|f| fragments_changed.contains(&f.id)) - .cloned() - .collect(); - - Self::prune_updated_fields_from_indices( - &mut final_indices, - &modified_fragments, - &replaced_fields, - ); - } - Operation::DataOverlay { groups } => { - // Stamp each overlay with the version this commit is producing. - // build_manifest re-runs on every retry with an updated - // current_manifest, so this is naturally re-stamped on retry. - let new_version = current_manifest.map_or(1, |m| m.version + 1); - - let existing_fragments = maybe_existing_fragments?; - // Multiple groups may target the same fragment; merge them in - // order rather than letting a HashMap collapse drop all but the - // last group's overlays. - let mut overlays_by_fragment: HashMap> = HashMap::new(); - for group in groups { - overlays_by_fragment - .entry(group.fragment_id) - .or_default() - .extend(group.overlays.iter()); - } - - // Every group must target an existing fragment. Build a set of - // existing ids once so this is O(groups + fragments) rather than - // O(groups * fragments). - let existing_fragment_ids: HashSet = - existing_fragments.iter().map(|f| f.id).collect(); - for fragment_id in overlays_by_fragment.keys() { - if !existing_fragment_ids.contains(fragment_id) { - return Err(Error::invalid_input(format!( - "DataOverlay targets fragment {fragment_id}, which does not exist" - ))); - } - } - - for fragment in existing_fragments { - let mut fragment = fragment.clone(); - if let Some(new_overlays) = overlays_by_fragment.get(&fragment.id) { - // Appended (not replaced) so concurrently-written overlays - // survive; later entries are newer. - fragment - .overlays - .extend(new_overlays.iter().map(|&overlay| { - let mut overlay = overlay.clone(); - overlay.committed_version = new_version; - overlay - })); - } - final_fragments.push(fragment); - } - } - Operation::UpdateMemWalState { compacted_sstables } => { - update_mem_wal_index_compacted_sstables( - &mut final_indices, - new_version, - compacted_sstables.clone(), - )?; - } - Operation::UpdateBases { .. } => { - // UpdateBases operation doesn't modify fragments or indices - // Base paths are handled in the manifest creation section below - final_fragments.extend(maybe_existing_fragments?.clone()); - } - }; - - // If a fragment was reserved then it may not belong at the end of the fragments list. - final_fragments.sort_by_key(|frag| frag.id); - - // Clean up data files that only contain tombstoned fields - Self::remove_tombstoned_data_files(&mut final_fragments); - - // Enforce the newest-last overlay ordering invariant at the write - // boundary. Load normalizes with a sort; this rejects any commit path - // that assembled a fragment's overlays out of order. - for fragment in &final_fragments { - if !fragment.overlays.is_empty() { - lance_table::format::overlay::verify_overlays_newest_last(&fragment.overlays)?; - } - } - - let user_requested_version = match (&config.storage_format, config.use_legacy_format) { - (Some(storage_format), _) => Some(storage_format.lance_file_version()?), - (None, Some(true)) => Some(LanceFileVersion::Legacy), - (None, Some(false)) => Some(LanceFileVersion::V2_0), - (None, None) => None, - }; - - let mut manifest = if let Some(current_manifest) = current_manifest { - // OVERWRITE with initial_bases on existing dataset is not allowed (caught by validation) - // So we always use new_from_previous which preserves base_paths - let mut prev_manifest = - Manifest::new_from_previous(current_manifest, schema, Arc::new(final_fragments)); - - if let (Some(user_requested_version), Operation::Overwrite { .. }) = - (user_requested_version, &self.operation) - { - // If this is an overwrite operation and the user has requested a specific version - // then overwrite with that version. Otherwise, if the user didn't request a specific - // version, then overwrite with whatever version we had before. - prev_manifest.data_storage_format = DataStorageFormat::new(user_requested_version); - } - - prev_manifest - } else { - let data_storage_format = - Self::data_storage_format_from_files(&final_fragments, user_requested_version)?; - Manifest::new( - schema, - Arc::new(final_fragments), - data_storage_format, - reference_paths, - ) - }; - - manifest.tag.clone_from(&self.tag); - - if config.auto_set_feature_flags { - // Internal operations (e.g. CreateIndex) use ManifestWriteConfig::default() - // which has use_stable_row_ids = false. Without inheriting from the previous - // manifest, apply_feature_flags would clear FLAG_STABLE_ROW_IDS. - let inherited = current_manifest - .map(|m| m.uses_stable_row_ids()) - .unwrap_or(false); - let use_stable_row_ids = config.use_stable_row_ids || inherited; - apply_feature_flags( - &mut manifest, - use_stable_row_ids, - config.disable_transaction_file, - )?; - } - manifest.set_timestamp(timestamp_to_nanos(config.timestamp)); - - manifest.update_max_fragment_id(); - - match &self.operation { - Operation::Overwrite { - config_upsert_values: Some(tm), - .. - } => { - manifest.config_mut().extend(tm.clone()); - } - Operation::UpdateConfig { - config_updates, - table_metadata_updates, - schema_metadata_updates, - field_metadata_updates, - } => { - if let Some(config_updates) = config_updates { - let mut config = manifest.config.clone(); - apply_update_map(&mut config, config_updates); - manifest.config = config; - } - if let Some(table_metadata_updates) = table_metadata_updates { - let mut table_metadata = manifest.table_metadata.clone(); - apply_update_map(&mut table_metadata, table_metadata_updates); - manifest.table_metadata = table_metadata; - } - if let Some(schema_metadata_updates) = schema_metadata_updates { - let mut schema_metadata = manifest.schema.metadata.clone(); - apply_update_map(&mut schema_metadata, schema_metadata_updates); - manifest.schema.metadata = schema_metadata; - } - // The unenforced primary and clustering keys are reserved - // schema properties: each is immutable once set, and its - // reserved metadata keys cannot be written with an invalid - // value. Capture the prior keys, and whether this transaction - // writes a reserved key, before applying the updates so - // violations can be rejected below. This runs on every apply, - // including conflict-rebase, so it also rejects the - // concurrent-writer race. - let primary_key_before: Vec = manifest - .schema - .unenforced_primary_key() - .iter() - .map(|field| field.id) - .collect(); - let writes_primary_key = field_metadata_updates.values().any(|update| { - update.update_entries.iter().any(|entry| { - entry.key == LANCE_UNENFORCED_PRIMARY_KEY - || entry.key == LANCE_UNENFORCED_PRIMARY_KEY_POSITION - }) - }); - let clustering_key_before: Vec = manifest - .schema - .unenforced_clustering_key() - .iter() - .map(|field| field.id) - .collect(); - let writes_clustering_key = field_metadata_updates.values().any(|update| { - update - .update_entries - .iter() - .any(|entry| entry.key == LANCE_UNENFORCED_CLUSTERING_KEY_POSITION) - }); - for (field_id, field_metadata_update) in field_metadata_updates { - if let Some(field) = manifest.schema.field_by_id_mut(*field_id) { - apply_update_map(&mut field.metadata, field_metadata_update); - // Also set unenforced primary key based on updated field metadata. - field.unenforced_primary_key_position = field - .metadata - .get(LANCE_UNENFORCED_PRIMARY_KEY_POSITION) - .and_then(|s| s.parse::().ok()) - .or_else(|| { - field - .metadata - .get(LANCE_UNENFORCED_PRIMARY_KEY) - .filter(|s| { - matches!(s.to_lowercase().as_str(), "true" | "1" | "yes") - }) - .map(|_| 0) - }); - // Also set unenforced clustering key based on updated - // field metadata. - field.unenforced_clustering_key_position = field - .metadata - .get(LANCE_UNENFORCED_CLUSTERING_KEY_POSITION) - .and_then(|s| s.parse::().ok()); - } else { - return Err(Error::invalid_input_source( - format!("Field with id {} does not exist", field_id).into(), - )); - } - } - let primary_key_after: Vec = manifest - .schema - .unenforced_primary_key() - .iter() - .map(|field| field.id) - .collect(); - if !primary_key_before.is_empty() { - // The primary key is already set: reject any change to it, - // and any write that touches a reserved primary key. - if writes_primary_key || primary_key_after != primary_key_before { - return Err(Error::invalid_input( - "the unenforced primary key is a reserved key and cannot be changed once set", - )); - } - } else if writes_primary_key && primary_key_after.is_empty() { - // A reserved primary key was written but did not install a - // valid primary key (e.g. a non-marker flag value or a - // non-numeric position). - return Err(Error::invalid_input( - "the unenforced primary key is a reserved key and cannot be set to an invalid value", - )); - } - let clustering_key_after: Vec = manifest - .schema - .unenforced_clustering_key() - .iter() - .map(|field| field.id) - .collect(); - if !clustering_key_before.is_empty() { - // The clustering key is already set: reject any change to - // it, and any write that touches the reserved key. - if writes_clustering_key || clustering_key_after != clustering_key_before { - return Err(Error::invalid_input( - "the unenforced clustering key is a reserved key and cannot be changed once set", - )); - } - } else if writes_clustering_key && clustering_key_after.is_empty() { - // The reserved clustering key was written but did not - // install a valid clustering key (e.g. a non-numeric - // position value). - return Err(Error::invalid_input( - "the unenforced clustering key is a reserved key and cannot be set to an invalid value", - )); - } - } - _ => {} - } - - // Handle UpdateBases operation to update manifest base_paths - if let Operation::UpdateBases { new_bases } = &self.operation { - // Validate and add new base paths to the manifest - for new_base in new_bases { - // Check for conflicts with existing base paths - if let Some(existing_base) = manifest - .base_paths - .values() - .find(|bp| bp.name == new_base.name || bp.path == new_base.path) - { - return Err(Error::invalid_input(format!( - "Conflict detected: Base path with name '{:?}' or path '{}' already exists. Existing: name='{:?}', path='{}'", - new_base.name, new_base.path, existing_base.name, existing_base.path - ))); - } - - // Assign a new ID if not already assigned - let mut base_to_add = new_base.clone(); - if base_to_add.id == 0 { - let next_id = manifest - .base_paths - .keys() - .max() - .map(|&id| id + 1) - .unwrap_or(1); - base_to_add.id = next_id; - } - - manifest.base_paths.insert(base_to_add.id, base_to_add); - } - } - - if let Operation::ReserveFragments { num_fragments } = self.operation { - manifest.max_fragment_id = Some(manifest.max_fragment_id.unwrap_or(0) + num_fragments); - } - - manifest.transaction_file = Some(transaction_file_path.to_string()); - - if let Some(next_row_id) = next_row_id { - manifest.next_row_id = next_row_id; - } - - Ok((manifest, final_indices)) - } - - fn register_pure_rewrite_rows_update_frags_in_indices( - indices: &mut [IndexMetadata], - pure_update_frag_ids: &[u64], - original_fragment_ids: &[u64], - fields_for_preserving_frag_bitmap: &[u32], - original_overlaid_frags: &HashMap, - schema: &Schema, - ) -> Result<()> { - if pure_update_frag_ids.is_empty() { - return Ok(()); - } - - let value_updated_field_set = fields_for_preserving_frag_bitmap - .iter() - .collect::>(); - - for index in indices.iter_mut() { - let index_covers_modified_field = index.fields.iter().any(|field_id| { - value_updated_field_set.contains(&u32::try_from(*field_id).unwrap()) - }); - if index_covers_modified_field { - continue; - } - let Some(fragment_bitmap) = index.fragment_bitmap.as_ref() else { - continue; - }; - - // Check that all the original fragments containing the updated rows are covered by - // the index. If not, some updated rows were not indexed, so we cannot index them. - let index_covers_all_original_fragments = original_fragment_ids - .iter() - .all(|&fragment_id| fragment_bitmap.contains(fragment_id as u32)); - if !index_covers_all_original_fragments { - continue; - } - - // A rewrite materializes overlays. If any of those overlays touched the - // column being indexed then the rewrite will modify that column. As a - // result, that index will no longer cover the fragment and it does not - // count as a pure rewrite and we must exclude it from the index's fragment - // bitmap. - let mut overlay_stale = RoaringBitmap::new(); - collect_overlay_stale_frags( - index, - original_overlaid_frags, - &mut overlay_stale, - schema, - )?; - if !overlay_stale.is_empty() { - continue; - } - - if let Some(fragment_bitmap) = index.fragment_bitmap.as_mut() { - for fragment_id in pure_update_frag_ids.iter().map(|f| *f as u32) { - fragment_bitmap.insert(fragment_id); - } - } - } - Ok(()) - } - - /// If an operation modifies one or more fields in a fragment then we need to remove - /// that fragment from any indices that cover one of the modified fields. - fn prune_updated_fields_from_indices( - indices: &mut [IndexMetadata], - updated_fragments: &[Fragment], - fields_modified: &[u32], - ) { - if fields_modified.is_empty() { - return; - } - - // If we modified any fields in the fragments then we need to remove those fragments - // from the index if the index covers one of those modified fields. - let fields_modified_set = fields_modified.iter().collect::>(); - for index in indices.iter_mut() { - if index - .fields - .iter() - .any(|field_id| fields_modified_set.contains(&u32::try_from(*field_id).unwrap())) - && let Some(fragment_bitmap) = &mut index.fragment_bitmap - { - for fragment_id in updated_fragments.iter().map(|f| f.id as u32) { - fragment_bitmap.remove(fragment_id); - } - } - } - } - - /// Map each (non-tombstoned) field id in a fragment to the path of the data - /// file that backs it. - fn fragment_field_paths(frag: &Fragment) -> HashMap { - let mut map = HashMap::new(); - for file in &frag.files { - for &field_id in file.fields.iter() { - if field_id >= 0 { - map.insert(field_id, file.path.as_str()); - } - } - } - map - } - - /// A `Merge` can rewrite a column's data *in place* -- the field stays in the - /// schema but its backing data file changes (the overlay fragment carries a new - /// file for the field and tombstones its old field id). `retain_relevant_indices` - /// only drops indices for *removed* fields, so without this the index keeps - /// covering the rewritten fragments with stale entries. Remove each such fragment - /// from any index covering a field whose backing data file changed. - fn prune_merge_rewritten_fields_from_indices( - indices: &mut [IndexMetadata], - prev_fragments: &[Fragment], - new_fragments: &[Fragment], - ) { - let prev_by_id: HashMap = - prev_fragments.iter().map(|f| (f.id, f)).collect(); - for new_frag in new_fragments { - let Some(prev) = prev_by_id.get(&new_frag.id) else { - continue; // brand-new fragment: nothing stale to prune - }; - let prev_paths = Self::fragment_field_paths(prev); - let new_paths = Self::fragment_field_paths(new_frag); - // Fields still present whose backing file path changed == rewritten data. - let changed: Vec = prev_paths - .iter() - .filter(|(field_id, prev_path)| { - new_paths - .get(*field_id) - .is_some_and(|new_path| new_path != *prev_path) - }) - .map(|(field_id, _)| *field_id as u32) - .collect(); - if changed.is_empty() { - continue; - } - Self::prune_updated_fields_from_indices( - indices, - std::slice::from_ref(new_frag), - &changed, - ); - } - } - - /// After a `Rewrite` fully compacts a fragment, its data overlays are baked - /// into the new fragment's base data. An index built *before* one of those - /// overlays (`overlay.committed_version > index.dataset_version`) indexed the - /// stale pre-overlay values -- and unlike a live overlay, the compacted - /// fragment no longer signals that staleness to the query path. Drop each - /// rewritten (new) fragment from the coverage of any index covering a field - /// such an overlay supplied, so those rows fall back to a flat scan. - fn prune_overlay_stale_fields_from_indices( - indices: &mut [IndexMetadata], - groups: &[RewriteGroup], - ) { - for group in groups { - // field id -> newest overlay committed_version supplying that field - let mut overlaid_field_versions: HashMap = HashMap::new(); - for old_frag in &group.old_fragments { - for overlay in &old_frag.overlays { - for &field_id in overlay.data_file.fields.iter() { - if field_id < 0 { - // Tombstoned (obsolete) overlay field: supplies nothing. - continue; - } - let entry = overlaid_field_versions.entry(field_id).or_insert(0); - *entry = (*entry).max(overlay.committed_version); - } - } - } - if overlaid_field_versions.is_empty() { - continue; - } - - let new_fragment_ids = group - .new_fragments - .iter() - .map(|f| f.id as u32) - .collect::>(); - for index in indices.iter_mut() { - let is_stale = index.fields.iter().any(|field_id| { - overlaid_field_versions - .get(field_id) - .is_some_and(|&overlay_version| overlay_version > index.dataset_version) - }); - if is_stale && let Some(fragment_bitmap) = &mut index.fragment_bitmap { - for new_id in &new_fragment_ids { - fragment_bitmap.remove(*new_id); - } - } - } - } - } - - fn is_vector_index(index: &IndexMetadata) -> bool { - if let Some(details) = &index.index_details { - details.type_url.ends_with("VectorIndexDetails") - } else { - false - } - } - - /// Remove data files that only contain tombstoned fields (-2) - /// These files no longer contain any live data and can be safely dropped - fn remove_tombstoned_data_files(fragments: &mut [Fragment]) { - for fragment in fragments { - fragment.files.retain(|file| { - // Keep file if it has at least one non-tombstoned field - file.fields.iter().any(|&field_id| field_id != -2) - }); - } - } - - fn retain_relevant_indices( - indices: &mut Vec, - schema: &Schema, - fragments: &[Fragment], - ) { - let field_ids = schema - .fields_pre_order() - .map(|f| f.id) - .collect::>(); - - // Remove indices for fields no longer in schema - indices.retain(|existing_index| { - existing_index - .fields - .iter() - .all(|field_id| field_ids.contains(field_id)) - || is_system_index(existing_index) - }); - - // Fragment bitmaps record which fragments the index was originally built for. - // Operations like updates and data replacement prune these bitmaps, and - // effective_fragment_bitmap intersects with existing fragments at query time. - - // Apply retention logic for indices with empty bitmaps per index name - // (except for fragment reuse indices which are always kept) - let mut indices_by_name: std::collections::HashMap> = - std::collections::HashMap::new(); - - // Group indices by name - for index in indices.iter() { - if index.name != FRAG_REUSE_INDEX_NAME { - indices_by_name - .entry(index.name.clone()) - .or_default() - .push(index); - } - } - - // Build a set of UUIDs to keep based on retention rules - let mut uuids_to_keep = std::collections::HashSet::new(); - - let existing_fragments = fragments - .iter() - .map(|f| f.id as u32) - .collect::(); - - // For each group of indices with the same name - for (_, same_name_indices) in indices_by_name { - if same_name_indices.len() > 1 { - // Separate empty and non-empty indices - let (empty_indices, non_empty_indices): (Vec<_>, Vec<_>) = - same_name_indices.iter().partition(|index| { - index - .effective_fragment_bitmap(&existing_fragments) - .as_ref() - .is_none_or(|bitmap| bitmap.is_empty()) - }); - - if non_empty_indices.is_empty() { - // All indices are empty - for scalar indices, keep only the first (oldest) one - // For vector indices, remove all of them - let mut sorted_indices = empty_indices; - sorted_indices.sort_by_key(|index: &&IndexMetadata| index.dataset_version); // Sort by ascending dataset_version - - // Keep only the first (oldest) if it's not a vector index - if let Some(oldest) = sorted_indices.first() - && !Self::is_vector_index(oldest) - { - uuids_to_keep.insert(oldest.uuid); - } - } else { - // At least one index has non-empty bitmap - keep all non-empty indices - for index in non_empty_indices { - uuids_to_keep.insert(index.uuid); - } - } - } else { - // Single index - keep it unless it's an empty vector index - if let Some(index) = same_name_indices.first() { - let is_empty = index - .effective_fragment_bitmap(&existing_fragments) - .as_ref() - .is_none_or(|bitmap| bitmap.is_empty()); - let is_vector = Self::is_vector_index(index); - - // Keep the index unless it's an empty vector index - if !is_empty || !is_vector { - uuids_to_keep.insert(index.uuid); - } - } - } - } - - // Use Vec::retain to safely remove indices - indices.retain(|index| { - index.name == FRAG_REUSE_INDEX_NAME || uuids_to_keep.contains(&index.uuid) - }); - } - - fn recalculate_fragment_bitmap( - old: &RoaringBitmap, - groups: &[RewriteGroup], - ) -> Result { - let mut new_bitmap = old.clone(); - for group in groups { - let any_in_index = group - .old_fragments - .iter() - .any(|frag| old.contains(frag.id as u32)); - let all_in_index = group - .old_fragments - .iter() - .all(|frag| old.contains(frag.id as u32)); - // Any rewrite group may or may not be covered by the index. However, if any fragment - // in a rewrite group was previously covered by the index then all fragments in the rewrite - // group must have been previously covered by the index. plan_compaction takes care of - // this for us so this should be safe to assume. - if any_in_index { - if all_in_index { - for frag_id in group.old_fragments.iter().map(|frag| frag.id as u32) { - new_bitmap.remove(frag_id); - } - new_bitmap.extend(group.new_fragments.iter().map(|frag| frag.id as u32)); - } else { - return Err(Error::invalid_input( - "The compaction plan included a rewrite group that was a split of indexed and non-indexed data", - )); - } - } - } - Ok(new_bitmap) - } - - fn handle_rewrite_indices( - indices: &mut [IndexMetadata], - rewritten_indices: &[RewrittenIndex], - groups: &[RewriteGroup], - ) -> Result<()> { - let mut modified_indices = HashSet::new(); - - for rewritten_index in rewritten_indices { - if !modified_indices.insert(rewritten_index.old_id) { - return Err(Error::invalid_input(format!( - "An invalid compaction plan must have been generated because multiple tasks modified the same index: {}", - rewritten_index.old_id - ))); - } - - // Skip indices that no longer exist (may have been removed by concurrent operation) - let Some(index) = indices - .iter_mut() - .find(|idx| idx.uuid == rewritten_index.old_id) - else { - continue; - }; - - index.fragment_bitmap = Some(Self::recalculate_fragment_bitmap( - index.fragment_bitmap.as_ref().ok_or_else(|| { - Error::invalid_input(format!( - "Cannot rewrite index {} which did not store fragment bitmap", - index.uuid - )) - })?, - groups, - )?); - index.uuid = rewritten_index.new_id; - // Update file sizes to match the new index files. When not available - // (e.g., from older writers), clear the old file sizes to avoid - // using stale sizes from the pre-remap index. - index.files = rewritten_index.new_index_files.clone(); - } - Ok(()) - } - - fn handle_rewrite_fragments( - final_fragments: &mut Vec, - groups: &[RewriteGroup], - fragment_id: &mut u64, - version: u64, - _next_row_id: Option<&u64>, - ) -> Result<()> { - for group in groups { - // If the old fragments are contiguous, find the range - let replace_range = { - let start = final_fragments - .iter() - .enumerate() - .find(|(_, f)| f.id == group.old_fragments[0].id) - .ok_or_else(|| { - Error::commit_conflict_source( - version, - format!( - "dataset does not contain a fragment a rewrite operation wants to replace: id={}", - group.old_fragments[0].id - ) - .into(), - ) - })? - .0; - - // Verify old_fragments matches contiguous range - let mut i = 1; - loop { - if i == group.old_fragments.len() { - break Some(start..start + i); - } - if final_fragments[start + i].id != group.old_fragments[i].id { - break None; - } - i += 1; - } - }; - - let new_fragments = Self::fragments_with_ids(group.new_fragments.clone(), fragment_id) - .collect::>(); - - // Version metadata for rewritten fragments is handled by the compaction code - // (recalc_versions_for_rewritten_fragments) which preserves version information - // from the original fragments. We don't modify it here. - - if let Some(replace_range) = replace_range { - // Efficiently path using slice - final_fragments.splice(replace_range, new_fragments); - } else { - // Slower path for non-contiguous ranges - for fragment in group.old_fragments.iter() { - final_fragments.retain(|f| f.id != fragment.id); - } - final_fragments.extend(new_fragments); - } - } - Ok(()) - } - - /// collect the pure(the num of row IDs are equal to the physical rows) "rewrite rows" updated fragment ids - fn collect_pure_rewrite_row_update_frags_ids(fragments: &[Fragment]) -> Result> { - let mut pure_update_frag_ids = Vec::new(); - - for fragment in fragments { - let physical_rows = fragment - .physical_rows - .ok_or_else(|| Error::internal("Fragment does not have physical rows"))? - as u64; - - if let Some(row_id_meta) = &fragment.row_id_meta { - let existing_row_count = match row_id_meta { - RowIdMeta::Inline(data) => { - let sequence = read_row_ids(data)?; - sequence.len() as u64 - } - _ => 0, - }; - - // only filter the fragments that match: all the rows have row id, - // which means it does not contain inserted rows in this fragment - if existing_row_count == physical_rows { - pure_update_frag_ids.push(fragment.id); - } - } - } - - Ok(pure_update_frag_ids) - } - - fn assign_row_ids(next_row_id: &mut u64, fragments: &mut [Fragment]) -> Result<()> { - for fragment in fragments { - let physical_rows = fragment - .physical_rows - .ok_or_else(|| Error::internal("Fragment does not have physical rows"))? - as u64; - - if fragment.row_id_meta.is_some() { - // we may meet merge insert case, it only has partial row ids. - // so here, we need to check if the row ids match the physical rows - // if yes, continue - // if not, fill the remaining row ids to the physical rows, then update row_id_meta - - // Check if existing row IDs match the physical rows count - let existing_row_count = match &fragment.row_id_meta { - Some(RowIdMeta::Inline(data)) => { - // Parse the serialized row ID sequence to get the count - let sequence = read_row_ids(data)?; - sequence.len() as u64 - } - _ => 0, - }; - - match existing_row_count.cmp(&physical_rows) { - Ordering::Equal => { - // Row IDs already match physical rows, continue to next fragment - continue; - } - Ordering::Less => { - // Partial row IDs - need to fill the remaining ones - let remaining_rows = physical_rows - existing_row_count; - let new_row_ids = *next_row_id..(*next_row_id + remaining_rows); - - // Merge existing and new row IDs - let combined_sequence = match &fragment.row_id_meta { - Some(RowIdMeta::Inline(data)) => read_row_ids(data)?, - _ => { - return Err(Error::internal( - "Failed to deserialize existing row ID sequence", - )); - } - }; - - let mut row_ids: Vec = combined_sequence.iter().collect(); - for row_id in new_row_ids { - row_ids.push(row_id); - } - let combined_sequence = RowIdSequence::from(row_ids.as_slice()); - - let serialized = write_row_ids(&combined_sequence); - fragment.row_id_meta = Some(RowIdMeta::Inline(serialized)); - *next_row_id += remaining_rows; - } - Ordering::Greater => { - // More row IDs than physical rows - this shouldn't happen - return Err(Error::internal(format!( - "Fragment has more row IDs ({}) than physical rows ({})", - existing_row_count, physical_rows - ))); - } - } - } else { - let row_ids = *next_row_id..(*next_row_id + physical_rows); - let sequence = RowIdSequence::from(row_ids); - // TODO: write to a separate file if large. Possibly share a file with other fragments. - let serialized = write_row_ids(&sequence); - fragment.row_id_meta = Some(RowIdMeta::Inline(serialized)); - *next_row_id += physical_rows; - } - } - Ok(()) - } -} - -impl From<&DataReplacementGroup> for pb::transaction::DataReplacementGroup { - fn from(DataReplacementGroup(fragment_id, new_file): &DataReplacementGroup) -> Self { - Self { - fragment_id: *fragment_id, - new_file: Some(new_file.into()), - } - } -} - -/// Convert a protobug DataReplacementGroup to a rust native DataReplacementGroup -/// this is unfortunately TryFrom instead of From because of the Option in the pb::DataReplacementGroup -impl TryFrom for DataReplacementGroup { - type Error = Error; - - fn try_from(message: pb::transaction::DataReplacementGroup) -> Result { - Ok(Self( - message.fragment_id, - message - .new_file - .ok_or(Error::invalid_input( - "DataReplacementGroup must have a new_file", - ))? - .try_into()?, - )) - } -} - -impl From<&DataOverlayGroup> for pb::transaction::DataOverlayGroup { - fn from(group: &DataOverlayGroup) -> Self { - Self { - fragment_id: group.fragment_id, - overlays: group - .overlays - .iter() - .map(pb::DataOverlayFile::from) - .collect(), - } - } -} - -impl TryFrom for DataOverlayGroup { - type Error = Error; - - fn try_from(message: pb::transaction::DataOverlayGroup) -> Result { - Ok(Self { - fragment_id: message.fragment_id, - overlays: message - .overlays - .into_iter() - .map(DataOverlayFile::try_from) - .collect::>>()?, - }) - } -} - -impl TryFrom for Transaction { - type Error = Error; - - fn try_from(message: pb::Transaction) -> Result { - let operation = match message.operation { - Some(pb::transaction::Operation::Append(pb::transaction::Append { fragments })) => { - Operation::Append { - fragments: fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - } - } - Some(pb::transaction::Operation::Clone(pb::transaction::Clone { - is_shallow, - ref_name, - ref_version, - ref_path, - branch_name, - })) => Operation::Clone { - is_shallow, - ref_name, - ref_version, - ref_path, - branch_name, - }, - Some(pb::transaction::Operation::Delete(pb::transaction::Delete { - updated_fragments, - deleted_fragment_ids, - predicate, - })) => Operation::Delete { - updated_fragments: updated_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - deleted_fragment_ids, - predicate, - }, - Some(pb::transaction::Operation::Overwrite(pb::transaction::Overwrite { - fragments, - schema, - schema_metadata: _schema_metadata, // TODO: handle metadata - config_upsert_values, - initial_bases, - })) => { - let config_upsert_option = if config_upsert_values.is_empty() { - None - } else { - Some(config_upsert_values) - }; - - Operation::Overwrite { - fragments: fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - schema: Schema::try_from(&Fields(schema))?, - config_upsert_values: config_upsert_option, - initial_bases: if initial_bases.is_empty() { - None - } else { - Some(initial_bases.into_iter().map(BasePath::from).collect()) - }, - } - } - Some(pb::transaction::Operation::ReserveFragments( - pb::transaction::ReserveFragments { num_fragments }, - )) => Operation::ReserveFragments { num_fragments }, - Some(pb::transaction::Operation::Rewrite(pb::transaction::Rewrite { - old_fragments, - new_fragments, - groups, - rewritten_indices, - })) => { - let groups = if !groups.is_empty() { - groups - .into_iter() - .map(RewriteGroup::try_from) - .collect::>()? - } else { - vec![RewriteGroup { - old_fragments: old_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - new_fragments: new_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - }] - }; - let rewritten_indices = rewritten_indices - .iter() - .map(RewrittenIndex::try_from) - .collect::>()?; - - Operation::Rewrite { - groups, - rewritten_indices, - frag_reuse_index: None, - } - } - Some(pb::transaction::Operation::CreateIndex(pb::transaction::CreateIndex { - new_indices, - removed_indices, - })) => Operation::CreateIndex { - new_indices: new_indices - .into_iter() - .map(IndexMetadata::try_from) - .collect::>()?, - removed_indices: removed_indices - .into_iter() - .map(IndexMetadata::try_from) - .collect::>()?, - }, - Some(pb::transaction::Operation::Merge(pb::transaction::Merge { - fragments, - schema, - schema_metadata: _schema_metadata, // TODO: handle metadata - })) => Operation::Merge { - fragments: fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - schema: Schema::try_from(&Fields(schema))?, - }, - Some(pb::transaction::Operation::Restore(pb::transaction::Restore { version })) => { - Operation::Restore { version } - } - Some(pb::transaction::Operation::Update(pb::transaction::Update { - removed_fragment_ids, - updated_fragments, - new_fragments, - fields_modified, - compacted_sstables, - fields_for_preserving_frag_bitmap, - update_mode, - inserted_rows, - updated_fragment_offsets, - })) => Operation::Update { - removed_fragment_ids, - updated_fragments: updated_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - new_fragments: new_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - fields_modified, - compacted_sstables: compacted_sstables - .into_iter() - .map(|m| CompactedSsTable::try_from(m).unwrap()) - .collect(), - fields_for_preserving_frag_bitmap, - update_mode: match update_mode { - 0 => Some(UpdateMode::RewriteRows), - 1 => Some(UpdateMode::RewriteColumns), - _ => Some(UpdateMode::RewriteRows), - }, - inserted_rows_filter: inserted_rows - .map(|ik| KeyExistenceFilter::try_from(&ik)) - .transpose()?, - updated_fragment_offsets: { - let m: HashMap = updated_fragment_offsets - .into_iter() - .filter(|(_, list)| !list.values.is_empty()) - .map(|(id, list)| (id, RoaringBitmap::from_iter(list.values))) - .collect(); - if m.is_empty() { - None - } else { - Some(UpdatedFragmentOffsets(m)) - } - }, - }, - Some(pb::transaction::Operation::Project(pb::transaction::Project { schema })) => { - Operation::Project { - schema: Schema::try_from(&Fields(schema))?, - } - } - Some(pb::transaction::Operation::UpdateConfig(update_config)) => { - // Check if new-style fields are present - let has_new_fields = update_config.config_updates.is_some() - || update_config.table_metadata_updates.is_some() - || update_config.schema_metadata_updates.is_some() - || !update_config.field_metadata_updates.is_empty(); - - // Check if old-style fields are present - let has_old_fields = !update_config.upsert_values.is_empty() - || !update_config.delete_keys.is_empty() - || !update_config.schema_metadata.is_empty() - || !update_config.field_metadata.is_empty(); - - // Error if both are present - if has_new_fields && has_old_fields { - return Err(Error::invalid_input_source( - "Cannot mix old and new style UpdateConfig fields".into(), - )); - } - - if has_old_fields { - // Translate old-style to new-style - let config_updates = if !update_config.upsert_values.is_empty() - || !update_config.delete_keys.is_empty() - { - Some(translate_config_updates( - &update_config.upsert_values, - &update_config.delete_keys, - )) - } else { - None - }; - - let schema_metadata_updates = if !update_config.schema_metadata.is_empty() { - Some(translate_schema_metadata_updates( - &update_config.schema_metadata, - )) - } else { - None - }; - - let field_metadata_updates = update_config - .field_metadata - .into_iter() - .map(|(field_id, field_meta_update)| { - ( - field_id as i32, - translate_schema_metadata_updates(&field_meta_update.metadata), - ) - }) - .collect(); - - Operation::UpdateConfig { - config_updates, - table_metadata_updates: None, - schema_metadata_updates, - field_metadata_updates, - } - } else { - // Use new-style fields directly (convert from protobuf) - Operation::UpdateConfig { - config_updates: update_config.config_updates.as_ref().map(UpdateMap::from), - table_metadata_updates: update_config - .table_metadata_updates - .as_ref() - .map(UpdateMap::from), - schema_metadata_updates: update_config - .schema_metadata_updates - .as_ref() - .map(UpdateMap::from), - field_metadata_updates: update_config - .field_metadata_updates - .iter() - .map(|(field_id, pb_update_map)| { - (*field_id, UpdateMap::from(pb_update_map)) - }) - .collect(), - } - } - } - Some(pb::transaction::Operation::DataReplacement( - pb::transaction::DataReplacement { replacements }, - )) => Operation::DataReplacement { - replacements: replacements - .into_iter() - .map(DataReplacementGroup::try_from) - .collect::>>()?, - }, - Some(pb::transaction::Operation::UpdateMemWalState( - pb::transaction::UpdateMemWalState { compacted_sstables }, - )) => Operation::UpdateMemWalState { - compacted_sstables: compacted_sstables - .into_iter() - .map(|m| CompactedSsTable::try_from(m).unwrap()) - .collect(), - }, - Some(pb::transaction::Operation::UpdateBases(pb::transaction::UpdateBases { - new_bases, - })) => Operation::UpdateBases { - new_bases: new_bases.into_iter().map(BasePath::from).collect(), - }, - Some(pb::transaction::Operation::DataOverlay(pb::transaction::DataOverlay { - groups, - })) => Operation::DataOverlay { - groups: groups - .into_iter() - .map(DataOverlayGroup::try_from) - .collect::>>()?, - }, - None => { - return Err(Error::internal( - "Transaction message did not contain an operation".to_string(), - )); - } - }; - Ok(Self { - read_version: message.read_version, - uuid: message.uuid.clone(), - operation, - tag: if message.tag.is_empty() { - None - } else { - Some(message.tag.clone()) - }, - transaction_properties: if message.transaction_properties.is_empty() { - None - } else { - Some(Arc::new(message.transaction_properties)) - }, - }) - } -} - -impl TryFrom<&pb::transaction::rewrite::RewrittenIndex> for RewrittenIndex { - type Error = Error; - - fn try_from(message: &pb::transaction::rewrite::RewrittenIndex) -> Result { - Ok(Self { - old_id: message - .old_id - .as_ref() - .map(Uuid::try_from) - .ok_or_else(|| { - Error::invalid_input("required field (old_id) missing from message".to_string()) - })??, - new_id: message - .new_id - .as_ref() - .map(Uuid::try_from) - .ok_or_else(|| { - Error::invalid_input("required field (new_id) missing from message".to_string()) - })??, - new_index_details: message - .new_index_details - .as_ref() - .ok_or_else(|| { - Error::invalid_input("new_index_details is a required field".to_string()) - })? - .clone(), - new_index_version: message.new_index_version, - new_index_files: if message.new_index_files.is_empty() { - None - } else { - Some( - message - .new_index_files - .iter() - .map(|f| IndexFile { - path: f.path.clone(), - size_bytes: f.size_bytes, - }) - .collect(), - ) - }, - }) - } -} - -impl TryFrom for RewriteGroup { - type Error = Error; - - fn try_from(message: pb::transaction::rewrite::RewriteGroup) -> Result { - Ok(Self { - old_fragments: message - .old_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - new_fragments: message - .new_fragments - .into_iter() - .map(Fragment::try_from) - .collect::>>()?, - }) - } -} - -impl From<&Transaction> for pb::Transaction { - fn from(value: &Transaction) -> Self { - let operation = match &value.operation { - Operation::Append { fragments } => { - pb::transaction::Operation::Append(pb::transaction::Append { - fragments: fragments.iter().map(pb::DataFragment::from).collect(), - }) - } - Operation::Clone { - is_shallow, - ref_name, - ref_version, - ref_path, - branch_name, - } => pb::transaction::Operation::Clone(pb::transaction::Clone { - is_shallow: *is_shallow, - ref_name: ref_name.clone(), - ref_version: *ref_version, - ref_path: ref_path.clone(), - branch_name: branch_name.clone(), - }), - Operation::Delete { - updated_fragments, - deleted_fragment_ids, - predicate, - } => pb::transaction::Operation::Delete(pb::transaction::Delete { - updated_fragments: updated_fragments - .iter() - .map(pb::DataFragment::from) - .collect(), - deleted_fragment_ids: deleted_fragment_ids.clone(), - predicate: predicate.clone(), - }), - Operation::Overwrite { - fragments, - schema, - config_upsert_values, - initial_bases, - } => { - pb::transaction::Operation::Overwrite(pb::transaction::Overwrite { - fragments: fragments.iter().map(pb::DataFragment::from).collect(), - schema: Fields::from(schema).0, - schema_metadata: Default::default(), // TODO: handle metadata - config_upsert_values: config_upsert_values - .clone() - .unwrap_or(Default::default()), - initial_bases: initial_bases - .as_ref() - .map(|paths| { - paths - .iter() - .cloned() - .map(|bp: BasePath| -> pb::BasePath { bp.into() }) - .collect::>() - }) - .unwrap_or_default(), - }) - } - Operation::ReserveFragments { num_fragments } => { - pb::transaction::Operation::ReserveFragments(pb::transaction::ReserveFragments { - num_fragments: *num_fragments, - }) - } - Operation::Rewrite { - groups, - rewritten_indices, - frag_reuse_index: _, - } => pb::transaction::Operation::Rewrite(pb::transaction::Rewrite { - groups: groups - .iter() - .map(pb::transaction::rewrite::RewriteGroup::from) - .collect(), - rewritten_indices: rewritten_indices - .iter() - .map(|rewritten| rewritten.into()) - .collect(), - ..Default::default() - }), - Operation::CreateIndex { - new_indices, - removed_indices, - } => pb::transaction::Operation::CreateIndex(pb::transaction::CreateIndex { - new_indices: new_indices.iter().map(pb::IndexMetadata::from).collect(), - removed_indices: removed_indices - .iter() - .map(pb::IndexMetadata::from) - .collect(), - }), - Operation::Merge { fragments, schema } => { - pb::transaction::Operation::Merge(pb::transaction::Merge { - fragments: fragments.iter().map(pb::DataFragment::from).collect(), - schema: Fields::from(schema).0, - schema_metadata: Default::default(), // TODO: handle metadata - }) - } - Operation::Restore { version } => { - pb::transaction::Operation::Restore(pb::transaction::Restore { version: *version }) - } - Operation::Update { - removed_fragment_ids, - updated_fragments, - new_fragments, - fields_modified, - compacted_sstables, - fields_for_preserving_frag_bitmap, - update_mode, - inserted_rows_filter, - updated_fragment_offsets, - } => pb::transaction::Operation::Update(pb::transaction::Update { - removed_fragment_ids: removed_fragment_ids.clone(), - updated_fragments: updated_fragments - .iter() - .map(pb::DataFragment::from) - .collect(), - new_fragments: new_fragments.iter().map(pb::DataFragment::from).collect(), - fields_modified: fields_modified.clone(), - compacted_sstables: compacted_sstables - .iter() - .map(pb::CompactedSsTable::from) - .collect(), - fields_for_preserving_frag_bitmap: fields_for_preserving_frag_bitmap.clone(), - update_mode: update_mode - .as_ref() - .map(|mode| match mode { - UpdateMode::RewriteRows => 0, - UpdateMode::RewriteColumns => 1, - }) - .unwrap_or(0), - inserted_rows: inserted_rows_filter.as_ref().map(|ik| ik.into()), - updated_fragment_offsets: updated_fragment_offsets - .as_ref() - .map(|UpdatedFragmentOffsets(m)| { - m.iter() - .filter(|(_, b)| !b.is_empty()) - .map(|(frag_id, b)| { - let values: Vec = b.iter().collect(); - (*frag_id, pb::transaction::UInt32List { values }) - }) - .collect::>() - }) - .unwrap_or_default(), - }), - Operation::Project { schema } => { - pb::transaction::Operation::Project(pb::transaction::Project { - schema: Fields::from(schema).0, - }) - } - Operation::UpdateConfig { - config_updates, - table_metadata_updates, - schema_metadata_updates, - field_metadata_updates, - } => pb::transaction::Operation::UpdateConfig(pb::transaction::UpdateConfig { - config_updates: config_updates - .as_ref() - .map(pb::transaction::UpdateMap::from), - table_metadata_updates: table_metadata_updates - .as_ref() - .map(pb::transaction::UpdateMap::from), - schema_metadata_updates: schema_metadata_updates - .as_ref() - .map(pb::transaction::UpdateMap::from), - field_metadata_updates: field_metadata_updates - .iter() - .map(|(field_id, update_map)| { - (*field_id, pb::transaction::UpdateMap::from(update_map)) - }) - .collect(), - // Leave old fields empty - we only write new-style fields - upsert_values: Default::default(), - delete_keys: Default::default(), - schema_metadata: Default::default(), - field_metadata: Default::default(), - }), - Operation::DataReplacement { replacements } => { - pb::transaction::Operation::DataReplacement(pb::transaction::DataReplacement { - replacements: replacements - .iter() - .map(pb::transaction::DataReplacementGroup::from) - .collect(), - }) - } - Operation::DataOverlay { groups } => { - pb::transaction::Operation::DataOverlay(pb::transaction::DataOverlay { - groups: groups - .iter() - .map(pb::transaction::DataOverlayGroup::from) - .collect(), - }) - } - Operation::UpdateMemWalState { compacted_sstables } => { - pb::transaction::Operation::UpdateMemWalState(pb::transaction::UpdateMemWalState { - compacted_sstables: compacted_sstables - .iter() - .map(pb::CompactedSsTable::from) - .collect::>(), - }) - } - Operation::UpdateBases { new_bases } => { - pb::transaction::Operation::UpdateBases(pb::transaction::UpdateBases { - new_bases: new_bases - .iter() - .cloned() - .map(|bp: BasePath| -> pb::BasePath { bp.into() }) - .collect::>(), - }) - } - }; - - let transaction_properties = value - .transaction_properties - .as_ref() - .map(|arc| arc.as_ref().clone()) - .unwrap_or_default(); - Self { - read_version: value.read_version, - uuid: value.uuid.clone(), - operation: Some(operation), - tag: value.tag.clone().unwrap_or("".to_string()), - transaction_properties, - } - } -} - -impl From<&RewrittenIndex> for pb::transaction::rewrite::RewrittenIndex { - fn from(value: &RewrittenIndex) -> Self { - Self { - old_id: Some((&value.old_id).into()), - new_id: Some((&value.new_id).into()), - new_index_details: Some(value.new_index_details.clone()), - new_index_version: value.new_index_version, - new_index_files: value - .new_index_files - .as_ref() - .map(|files| { - files - .iter() - .map(|f| pb::IndexFile { - path: f.path.clone(), - size_bytes: f.size_bytes, - }) - .collect() - }) - .unwrap_or_default(), - } - } -} - -impl From<&RewriteGroup> for pb::transaction::rewrite::RewriteGroup { - fn from(value: &RewriteGroup) -> Self { - Self { - old_fragments: value - .old_fragments - .iter() - .map(pb::DataFragment::from) - .collect(), - new_fragments: value - .new_fragments - .iter() - .map(pb::DataFragment::from) - .collect(), - } - } -} - -/// Validate the operation is valid for the given manifest. -pub fn validate_operation(manifest: Option<&Manifest>, operation: &Operation) -> Result<()> { - let manifest = match (manifest, operation) { - ( - None, - Operation::Overwrite { - fragments, schema, .. - }, - ) => { - // Validate here because we are going to return early. - schema_fragments_valid(None, schema, fragments)?; - - return Ok(()); - } - (None, Operation::Clone { .. }) => return Ok(()), - (Some(manifest), _) => manifest, - (None, _) => { - return Err(Error::invalid_input(format!( - "Cannot apply operation {} to non-existent dataset", - operation.name() - ))); - } - }; - - match operation { - Operation::Append { fragments } => { - // Fragments must contain all fields in the schema - schema_fragments_valid(Some(manifest), &manifest.schema, fragments) - } - Operation::Project { schema } => { - schema_fragments_valid(Some(manifest), schema, manifest.fragments.as_ref()) - } - Operation::Merge { fragments, schema } => { - merge_fragments_valid(manifest, fragments)?; - schema_fragments_valid(Some(manifest), schema, fragments) - } - Operation::Overwrite { - fragments, - schema, - config_upsert_values: None, - initial_bases: _, - } => { - // Pass None for manifest because Overwrite replaces all fragments. - // The old manifest's storage format is irrelevant for validating - // the new fragments (e.g., LEGACY→STABLE transitions). - schema_fragments_valid(None, schema, fragments) - } - Operation::Update { - updated_fragments, - new_fragments, - .. - } => { - schema_fragments_valid(Some(manifest), &manifest.schema, updated_fragments)?; - schema_fragments_valid(Some(manifest), &manifest.schema, new_fragments) - } - _ => Ok(()), - } -} - -fn schema_fragments_valid( - manifest: Option<&Manifest>, - schema: &Schema, - fragments: &[Fragment], -) -> Result<()> { - if let Some(manifest) = manifest - && manifest.data_storage_format.lance_file_version()? == LanceFileVersion::Legacy - { - return schema_fragments_legacy_valid(schema, fragments); - } - // validate that each data file at least contains one field. - for fragment in fragments { - for data_file in &fragment.files { - if data_file.fields.iter().len() == 0 { - return Err(Error::invalid_input(format!( - "Datafile {} does not contain any fields", - data_file.path - ))); - } - } - } - Ok(()) -} - -/// Check that each fragment contains all fields in the schema. -/// It is not required that the schema contains all fields in the fragment. -/// There may be masked fields. -fn schema_fragments_legacy_valid(schema: &Schema, fragments: &[Fragment]) -> Result<()> { - // TODO: add additional validation. Consider consolidating with various - // validate() methods in the codebase. - for fragment in fragments { - for field in schema.fields_pre_order() { - if !fragment - .files - .iter() - .flat_map(|f| f.fields.iter()) - .any(|f_id| f_id == &field.id) - { - return Err(Error::invalid_input(format!( - "Fragment {} does not contain field {:?}", - fragment.id, field - ))); - } - } - } - Ok(()) -} - -/// Returns true if Operation::Merge rewrote this fragment's column data files (Fragment::files -/// changed versus the previous manifest). Used to bump last_updated_at_version_meta only when -/// new column values were materialized to disk. -/// -/// Deletion file changes alone are not treated as rewrites: tombstones remove rows but -/// survivors did not receive new column bytes; stamping last_updated for those rows would be -/// incorrect for CDF. -#[inline] -fn merge_fragment_physically_rewritten(prev: &Fragment, merged: &Fragment) -> bool { - debug_assert_eq!(prev.id, merged.id); - if prev.files.len() != merged.files.len() { - return true; - } - // Compare identity fields only. file_size_bytes is an AtomicU64 cache that - // concurrent scans can populate in place on the manifest's DataFile, so it - // must not be part of the rewrite check. - prev.files.iter().zip(merged.files.iter()).any(|(p, m)| { - p.path != m.path - || p.fields != m.fields - || p.column_indices != m.column_indices - || p.file_major_version != m.file_major_version - || p.file_minor_version != m.file_minor_version - || p.base_id != m.base_id - }) -} - -/// Validate that Merge operations preserve all original fragments. -/// Merge operations should only add columns or rows, not reduce fragments. -/// This ensures fragments correspond at one-to-one with the original fragment list. -fn merge_fragments_valid(manifest: &Manifest, new_fragments: &[Fragment]) -> Result<()> { - let original_fragments = manifest.fragments.as_ref(); - - // Additional validation: ensure we're not accidentally reducing the fragment count - if new_fragments.len() < original_fragments.len() { - return Err(Error::invalid_input(format!( - "Merge operation reduced fragment count from {} to {}. \ - Merge operations should only add columns, not reduce fragments.", - original_fragments.len(), - new_fragments.len() - ))); - } - - // Collect new fragment IDs - let new_fragment_map: HashMap = - new_fragments.iter().map(|f| (f.id, f)).collect(); - - // Check that all original fragments are preserved in the new fragments list - // Validate that each original fragment's metadata is preserved - let mut missing_fragments: Vec = Vec::new(); - for original_fragment in original_fragments { - if let Some(new_fragment) = new_fragment_map.get(&original_fragment.id) { - // Validate physical_rows (row count) hasn't changed - if original_fragment.physical_rows != new_fragment.physical_rows { - return Err(Error::invalid_input(format!( - "Merge operation changed row count for fragment {}. \ - Original: {:?}, New: {:?}. \ - Merge operations should preserve fragment row counts and only add new columns.", - original_fragment.id, - original_fragment.physical_rows, - new_fragment.physical_rows - ))); - } - } else { - missing_fragments.push(original_fragment.id); - } - } - - if !missing_fragments.is_empty() { - return Err(Error::invalid_input(format!( - "Merge operation is missing original fragments: {:?}. \ - Merge operations should preserve all original fragments and only add new columns. \ - Expected fragments: {:?}, but got: {:?}", - missing_fragments, - original_fragments.iter().map(|f| f.id).collect::>(), - new_fragment_map.keys().copied().collect::>() - ))); - } - - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use arrow_array::cast::AsArray; - use arrow_array::types::UInt64Type; - use arrow_array::{Int32Array, RecordBatch, RecordBatchIterator}; - use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; - use chrono::Utc; - use futures::TryStreamExt; - use lance_core::datatypes::Schema as LanceSchema; - use lance_core::utils::address::RowAddress; - use lance_core::utils::tempfile::TempStrDir; - use lance_core::{ROW_ADDR, ROW_CREATED_AT_VERSION, ROW_LAST_UPDATED_AT_VERSION}; - use lance_file::version::LanceFileVersion; - use lance_io::utils::CachedFileSize; - use lance_table::format::overlay::OverlayCoverage; - use lance_table::format::{ - RowDatasetVersionMeta, RowDatasetVersionRun, RowDatasetVersionSequence, RowIdMeta, - }; - use lance_table::rowids::segment::U64Segment; - use lance_table::rowids::write_row_ids; - use std::collections::HashMap; - use std::sync::Arc; - use uuid::Uuid; - - use crate::Dataset; - use crate::dataset::write::WriteParams; - use crate::session::Session; - - fn sample_manifest() -> Manifest { - let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - Manifest::new( - LanceSchema::try_from(&schema).unwrap(), - Arc::new(vec![Fragment::new(0)]), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ) - } - - fn sample_index_metadata(name: &str) -> IndexMetadata { - IndexMetadata { - uuid: Uuid::new_v4(), - fields: vec![0], - name: name.to_string(), - dataset_version: 0, - fragment_bitmap: Some([0].into_iter().collect()), - index_details: None, - index_version: 1, - created_at: Some(Utc::now()), - base_id: None, - files: None, - } - } - - #[test] - fn test_rewrite_fragments() { - let existing_fragments: Vec = (0..10).map(Fragment::new).collect(); - - let mut final_fragments = existing_fragments; - let rewrite_groups = vec![ - // Since these are contiguous, they will be put in the same location - // as 1 and 2. - RewriteGroup { - old_fragments: vec![Fragment::new(1), Fragment::new(2)], - // These two fragments were previously reserved - new_fragments: vec![Fragment::new(15), Fragment::new(16)], - }, - // These are not contiguous, so they will be inserted at the end. - RewriteGroup { - old_fragments: vec![Fragment::new(5), Fragment::new(8)], - // We pretend this id was not reserved. Does not happen in practice today - // but we want to leave the door open. - new_fragments: vec![Fragment::new(0)], - }, - ]; - - let mut fragment_id = 20; - let version = 0; - - Transaction::handle_rewrite_fragments( - &mut final_fragments, - &rewrite_groups, - &mut fragment_id, - version, - None, - ) - .unwrap(); - - assert_eq!(fragment_id, 21); - - let expected_fragments: Vec = vec![ - Fragment::new(0), - Fragment::new(15), - Fragment::new(16), - Fragment::new(3), - Fragment::new(4), - Fragment::new(6), - Fragment::new(7), - Fragment::new(9), - Fragment::new(20), - ]; - - assert_eq!(final_fragments, expected_fragments); - } - - #[test] - fn test_merge_fragments_valid() { - // Create a simple schema for testing - let schema = ArrowSchema::new(vec![ - ArrowField::new("id", DataType::Int32, false), - ArrowField::new("name", DataType::Utf8, false), - ]); - - // Create original fragments - let original_fragments = vec![Fragment::new(1), Fragment::new(2), Fragment::new(3)]; - - // Create a manifest with original fragments - let manifest = Manifest::new( - LanceSchema::try_from(&schema).unwrap(), - Arc::new(original_fragments), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - - // Test 1: Empty fragments should fail - let empty_fragments = vec![]; - let result = merge_fragments_valid(&manifest, &empty_fragments); - assert!(result.is_err()); - assert!( - result - .unwrap_err() - .to_string() - .contains("reduced fragment count") - ); - - // Test 2: Missing original fragments should fail - let missing_fragments = vec![ - Fragment::new(1), - Fragment::new(2), - // Fragment 3 is missing - Fragment::new(4), // New fragment - ]; - let result = merge_fragments_valid(&manifest, &missing_fragments); - assert!(result.is_err()); - assert!( - result - .unwrap_err() - .to_string() - .contains("missing original fragments") - ); - - // Test 3: Reduced fragment count should fail - let reduced_fragments = vec![ - Fragment::new(1), - Fragment::new(2), - // Fragment 3 is missing, no new fragments added - ]; - let result = merge_fragments_valid(&manifest, &reduced_fragments); - assert!(result.is_err()); - assert!( - result - .unwrap_err() - .to_string() - .contains("reduced fragment count") - ); - - // Test 4: Valid merge with all original fragments plus new ones should succeed - let valid_fragments = vec![ - Fragment::new(1), - Fragment::new(2), - Fragment::new(3), - Fragment::new(4), // New fragment - Fragment::new(5), // Another new fragment - ]; - let result = merge_fragments_valid(&manifest, &valid_fragments); - assert!(result.is_ok()); - - // Test 5: Same fragments (no new ones) should succeed - let same_fragments = vec![Fragment::new(1), Fragment::new(2), Fragment::new(3)]; - let result = merge_fragments_valid(&manifest, &same_fragments); - assert!(result.is_ok()); - } - - #[test] - fn test_create_index_build_manifest_keeps_unremoved_same_name_indices() { - let manifest = sample_manifest(); - let first_index = sample_index_metadata("vector_idx"); - let second_index = sample_index_metadata("vector_idx"); - let third_index = sample_index_metadata("vector_idx"); - - let transaction = Transaction::new( - manifest.version, - Operation::CreateIndex { - new_indices: vec![third_index.clone()], - removed_indices: vec![second_index.clone()], - }, - None, - ); - - let (_, final_indices) = transaction - .build_manifest( - Some(&manifest), - vec![first_index.clone(), second_index.clone()], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert_eq!(final_indices.len(), 2); - assert!(final_indices.iter().any(|idx| idx.uuid == first_index.uuid)); - assert!(final_indices.iter().any(|idx| idx.uuid == third_index.uuid)); - assert!( - !final_indices - .iter() - .any(|idx| idx.uuid == second_index.uuid) - ); - } - - #[test] - fn test_create_index_build_manifest_deduplicates_relisted_indices_by_uuid() { - let manifest = sample_manifest(); - let first_index = sample_index_metadata("vector_idx"); - let second_index = sample_index_metadata("vector_idx"); - let third_index = sample_index_metadata("vector_idx"); - - let transaction = Transaction::new( - manifest.version, - Operation::CreateIndex { - new_indices: vec![first_index.clone(), third_index.clone()], - removed_indices: vec![second_index.clone()], - }, - None, - ); - - let (_, final_indices) = transaction - .build_manifest( - Some(&manifest), - vec![first_index.clone(), second_index.clone()], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert_eq!(final_indices.len(), 2); - assert_eq!( - final_indices - .iter() - .filter(|idx| idx.uuid == first_index.uuid) - .count(), - 1 - ); - assert!(final_indices.iter().any(|idx| idx.uuid == third_index.uuid)); - assert!( - !final_indices - .iter() - .any(|idx| idx.uuid == second_index.uuid) - ); - } - - #[test] - fn test_remove_tombstoned_data_files() { - // Create a fragment with mixed data files: some normal, some fully tombstoned - let mut fragment = Fragment::new(1); - - // Add a normal data file with valid field IDs - fragment.files.push(DataFile { - path: "normal.lance".to_string(), - fields: Arc::from([1, 2, 3]), - column_indices: Arc::from([]), - file_major_version: 2, - file_minor_version: 0, - file_size_bytes: CachedFileSize::new(1000), - base_id: None, - }); - - // Add a data file with all fields tombstoned - fragment.files.push(DataFile { - path: "all_tombstoned.lance".to_string(), - fields: Arc::from([-2, -2, -2]), - column_indices: Arc::from([]), - file_major_version: 2, - file_minor_version: 0, - file_size_bytes: CachedFileSize::new(500), - base_id: None, - }); - - // Add a data file with mixed tombstoned and valid fields - fragment.files.push(DataFile { - path: "mixed.lance".to_string(), - fields: Arc::from([4, -2, 5]), - column_indices: Arc::from([]), - file_major_version: 2, - file_minor_version: 0, - file_size_bytes: CachedFileSize::new(750), - base_id: None, - }); - - // Add another fully tombstoned file - fragment.files.push(DataFile { - path: "another_tombstoned.lance".to_string(), - fields: Arc::from([-2_i32]), - column_indices: Arc::from([]), - file_major_version: 2, - file_minor_version: 0, - file_size_bytes: CachedFileSize::new(250), - base_id: None, - }); - - let mut fragments = vec![fragment]; - - // Apply the cleanup - Transaction::remove_tombstoned_data_files(&mut fragments); - - // Should have removed the two fully tombstoned files - assert_eq!(fragments[0].files.len(), 2); - assert_eq!(fragments[0].files[0].path, "normal.lance"); - assert_eq!(fragments[0].files[1].path, "mixed.lance"); - } - - #[test] - fn test_assign_row_ids_new_fragment() { - // Test assigning row IDs to a fragment without existing row IDs - let mut fragments = vec![Fragment { - id: 1, - physical_rows: Some(100), - row_id_meta: None, - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }]; - let mut next_row_id = 0; - - Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); - - assert_eq!(next_row_id, 100); - assert!(fragments[0].row_id_meta.is_some()); - - if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { - let sequence = read_row_ids(data).unwrap(); - assert_eq!(sequence.len(), 100); - let row_ids: Vec = sequence.iter().collect(); - assert_eq!(row_ids, (0..100).collect::>()); - } else { - panic!("Expected inline row ID metadata"); - } - } - - #[test] - fn test_assign_row_ids_existing_complete() { - // Test with fragment that already has complete row IDs - let existing_sequence = RowIdSequence::from(0..50); - let serialized = write_row_ids(&existing_sequence); - - let mut fragments = vec![Fragment { - id: 1, - physical_rows: Some(50), - row_id_meta: Some(RowIdMeta::Inline(serialized)), - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }]; - let mut next_row_id = 100; - - Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); - - // next_row_id should not change - assert_eq!(next_row_id, 100); - - if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { - let sequence = read_row_ids(data).unwrap(); - assert_eq!(sequence.len(), 50); - let row_ids: Vec = sequence.iter().collect(); - assert_eq!(row_ids, (0..50).collect::>()); - } else { - panic!("Expected inline row ID metadata"); - } - } - - #[test] - fn test_assign_row_ids_partial_existing() { - // Test with fragment that has partial row IDs (merge insert case) - let existing_sequence = RowIdSequence::from(0..30); - let serialized = write_row_ids(&existing_sequence); - - let mut fragments = vec![Fragment { - id: 1, - physical_rows: Some(50), // More physical rows than existing row IDs - row_id_meta: Some(RowIdMeta::Inline(serialized)), - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }]; - let mut next_row_id = 100; - - Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); - - // next_row_id should advance by 20 (50 - 30) - assert_eq!(next_row_id, 120); - - if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { - let sequence = read_row_ids(data).unwrap(); - assert_eq!(sequence.len(), 50); - let row_ids: Vec = sequence.iter().collect(); - // Should contain original 0-29 plus new 100-119 - let mut expected = (0..30).collect::>(); - expected.extend(100..120); - assert_eq!(row_ids, expected); - } else { - panic!("Expected inline row ID metadata"); - } - } - - #[test] - fn test_assign_row_ids_excess_row_ids() { - // Test error case where fragment has more row IDs than physical rows - let existing_sequence = RowIdSequence::from(0..60); - let serialized = write_row_ids(&existing_sequence); - - let mut fragments = vec![Fragment { - id: 1, - physical_rows: Some(50), // Less physical rows than existing row IDs - row_id_meta: Some(RowIdMeta::Inline(serialized)), - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }]; - let mut next_row_id = 100; - - let result = Transaction::assign_row_ids(&mut next_row_id, &mut fragments); - - assert!(result.is_err()); - if let Err(Error::Internal { message, .. }) = result { - assert!(message.contains("more row IDs (60) than physical rows (50)")); - } else { - panic!("Expected Internal error about excess row IDs"); - } - } - - #[test] - fn test_assign_row_ids_multiple_fragments() { - // Test with multiple fragments, some with existing row IDs, some without - let existing_sequence = RowIdSequence::from(500..520); - let serialized = write_row_ids(&existing_sequence); - - let mut fragments = vec![ - Fragment { - id: 1, - physical_rows: Some(30), // No existing row IDs - row_id_meta: None, - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }, - Fragment { - id: 2, - physical_rows: Some(25), // Partial existing row IDs - row_id_meta: Some(RowIdMeta::Inline(serialized)), - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }, - ]; - let mut next_row_id = 1000; - - Transaction::assign_row_ids(&mut next_row_id, &mut fragments).unwrap(); - - // Should advance by 30 (first fragment) + 5 (second fragment partial) - assert_eq!(next_row_id, 1035); - - // Check first fragment - if let Some(RowIdMeta::Inline(data)) = &fragments[0].row_id_meta { - let sequence = read_row_ids(data).unwrap(); - assert_eq!(sequence.len(), 30); - let row_ids: Vec = sequence.iter().collect(); - assert_eq!(row_ids, (1000..1030).collect::>()); - } else { - panic!("Expected inline row ID metadata for first fragment"); - } - - // Check second fragment - if let Some(RowIdMeta::Inline(data)) = &fragments[1].row_id_meta { - let sequence = read_row_ids(data).unwrap(); - assert_eq!(sequence.len(), 25); - let row_ids: Vec = sequence.iter().collect(); - // Should contain original 500-519 plus new 1030-1034 - let mut expected = (500..520).collect::>(); - expected.extend(1030..1035); - assert_eq!(row_ids, expected); - } else { - panic!("Expected inline row ID metadata for second fragment"); - } - } - - #[test] - fn test_assign_row_ids_missing_physical_rows() { - // Test error case where fragment doesn't have physical_rows set - let mut fragments = vec![Fragment { - id: 1, - physical_rows: None, - row_id_meta: None, - files: vec![], - overlays: vec![], - deletion_file: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }]; - let mut next_row_id = 0; - - let result = Transaction::assign_row_ids(&mut next_row_id, &mut fragments); - - assert!(result.is_err()); - if let Err(Error::Internal { message, .. }) = result { - assert!(message.contains("Fragment does not have physical rows")); - } else { - panic!("Expected Internal error about missing physical rows"); - } - } - - // Helper functions for retain_relevant_indices tests - fn create_test_index( - name: &str, - field_id: i32, - dataset_version: u64, - fragment_bitmap: Option, - is_vector: bool, - ) -> IndexMetadata { - use prost_types::Any; - use std::sync::Arc; - use uuid::Uuid; - - let index_details = if is_vector { - Some(Arc::new(Any { - type_url: "type.googleapis.com/lance.index.VectorIndexDetails".to_string(), - value: vec![], - })) - } else { - Some(Arc::new(Any { - type_url: "type.googleapis.com/lance.index.ScalarIndexDetails".to_string(), - value: vec![], - })) - }; - - IndexMetadata { - uuid: Uuid::new_v4(), - fields: vec![field_id], - name: name.to_string(), - dataset_version, - fragment_bitmap, - index_details, - index_version: 1, - created_at: None, - base_id: None, - files: None, - } - } - - fn create_system_index(name: &str, field_id: i32) -> IndexMetadata { - use prost_types::Any; - use std::sync::Arc; - use uuid::Uuid; - - IndexMetadata { - uuid: Uuid::new_v4(), - fields: vec![field_id], - name: name.to_string(), - dataset_version: 1, - fragment_bitmap: Some(RoaringBitmap::from_iter([1, 2])), - index_details: Some(Arc::new(Any { - type_url: "type.googleapis.com/lance.index.SystemIndexDetails".to_string(), - value: vec![], - })), - index_version: 1, - created_at: None, - base_id: None, - files: None, - } - } - - fn create_test_schema(field_ids: &[i32]) -> Schema { - use arrow_schema::{DataType, Field as ArrowField, Schema as ArrowSchema}; - use lance_core::datatypes::Schema as LanceSchema; - - let fields: Vec = field_ids - .iter() - .map(|id| ArrowField::new(format!("field_{}", id), DataType::Int32, false)) - .collect(); - - let arrow_schema = ArrowSchema::new(fields); - let mut lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); - - // Assign field IDs - for (i, field_id) in field_ids.iter().enumerate() { - lance_schema.mut_field_by_id(i as i32).unwrap().id = *field_id; - } - - lance_schema - } - - #[test] - fn test_retain_indices_removes_missing_fields() { - let schema = create_test_schema(&[1, 2]); - let fragments = vec![Fragment::new(1), Fragment::new(2)]; - - let mut indices = vec![ - create_test_index("idx1", 1, 1, Some(RoaringBitmap::from_iter([1])), false), - create_test_index("idx2", 2, 1, Some(RoaringBitmap::from_iter([1])), false), - create_test_index("idx3", 99, 1, Some(RoaringBitmap::from_iter([1])), false), // Field doesn't exist - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - assert_eq!(indices.len(), 2); - assert!(indices.iter().all(|idx| idx.fields[0] != 99)); - } - - #[test] - fn test_retain_indices_keeps_system_indices() { - use lance_index::mem_wal::MEM_WAL_INDEX_NAME; - - let schema = create_test_schema(&[1, 2]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_system_index(FRAG_REUSE_INDEX_NAME, 99), // Field doesn't exist but should be kept - create_system_index(MEM_WAL_INDEX_NAME, 99), // Field doesn't exist but should be kept - create_test_index("regular_idx", 99, 1, Some(RoaringBitmap::new()), false), // Should be removed - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - assert_eq!(indices.len(), 2); - assert!(indices.iter().any(|idx| idx.name == FRAG_REUSE_INDEX_NAME)); - assert!(indices.iter().any(|idx| idx.name == MEM_WAL_INDEX_NAME)); - } - - #[test] - fn test_retain_indices_keeps_fragment_reuse_index() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_system_index(FRAG_REUSE_INDEX_NAME, 1), - create_test_index("other_idx", 1, 1, Some(RoaringBitmap::new()), false), - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Fragment reuse index should always be kept - assert!(indices.iter().any(|idx| idx.name == FRAG_REUSE_INDEX_NAME)); - } - - #[test] - fn test_retain_single_empty_scalar_index() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![create_test_index( - "scalar_idx", - 1, - 1, - Some(RoaringBitmap::new()), // Empty bitmap - false, - )]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Single empty scalar index should be kept - assert_eq!(indices.len(), 1); - } - - #[test] - fn test_retain_single_empty_vector_index() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![create_test_index( - "vector_idx", - 1, - 1, - Some(RoaringBitmap::new()), // Empty bitmap - true, - )]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Single empty vector index should be removed - assert_eq!(indices.len(), 0); - } - - #[test] - fn test_retain_single_nonempty_index() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut scalar_indices = vec![create_test_index( - "scalar_idx", - 1, - 1, - Some(RoaringBitmap::from_iter([1])), - false, - )]; - - let mut vector_indices = vec![create_test_index( - "vector_idx", - 1, - 1, - Some(RoaringBitmap::from_iter([1])), - true, - )]; - - Transaction::retain_relevant_indices(&mut scalar_indices, &schema, &fragments); - Transaction::retain_relevant_indices(&mut vector_indices, &schema, &fragments); - - // Both should be kept - assert_eq!(scalar_indices.len(), 1); - assert_eq!(vector_indices.len(), 1); - } - - #[test] - fn test_retain_single_index_with_none_bitmap() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut scalar_indices = vec![create_test_index("scalar_idx", 1, 1, None, false)]; - let mut vector_indices = vec![create_test_index("vector_idx", 1, 1, None, true)]; - - Transaction::retain_relevant_indices(&mut scalar_indices, &schema, &fragments); - Transaction::retain_relevant_indices(&mut vector_indices, &schema, &fragments); - - // Scalar should be kept, vector should be removed - assert_eq!(scalar_indices.len(), 1); - assert_eq!(vector_indices.len(), 0); - } - - #[test] - fn test_retain_multiple_empty_scalar_indices_keeps_oldest() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_test_index("idx", 1, 3, Some(RoaringBitmap::new()), false), - create_test_index("idx", 1, 1, Some(RoaringBitmap::new()), false), // Oldest - create_test_index("idx", 1, 2, Some(RoaringBitmap::new()), false), - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Should keep only the oldest (dataset_version = 1) - assert_eq!(indices.len(), 1); - assert_eq!(indices[0].dataset_version, 1); - } - - #[test] - fn test_retain_multiple_empty_vector_indices_removes_all() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_test_index("vec_idx", 1, 1, Some(RoaringBitmap::new()), true), - create_test_index("vec_idx", 1, 2, Some(RoaringBitmap::new()), true), - create_test_index("vec_idx", 1, 3, Some(RoaringBitmap::new()), true), - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // All empty vector indices should be removed - assert_eq!(indices.len(), 0); - } - - #[test] - fn test_retain_mixed_empty_nonempty_keeps_nonempty() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_test_index("idx", 1, 1, Some(RoaringBitmap::new()), false), // Empty - create_test_index("idx", 1, 2, Some(RoaringBitmap::from_iter([1])), false), // Non-empty - create_test_index("idx", 1, 3, Some(RoaringBitmap::new()), false), // Empty - create_test_index("idx", 1, 4, Some(RoaringBitmap::from_iter([1])), false), // Non-empty - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Should keep only non-empty indices - assert_eq!(indices.len(), 2); - assert!( - indices - .iter() - .all(|idx| idx.dataset_version == 2 || idx.dataset_version == 4) - ); - } - - #[test] - fn test_retain_mixed_empty_nonempty_vector_keeps_nonempty() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_test_index("vec_idx", 1, 1, Some(RoaringBitmap::new()), true), // Empty - create_test_index("vec_idx", 1, 2, Some(RoaringBitmap::from_iter([1])), true), // Non-empty - create_test_index("vec_idx", 1, 3, Some(RoaringBitmap::new()), true), // Empty - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Should keep only non-empty index - assert_eq!(indices.len(), 1); - assert_eq!(indices[0].dataset_version, 2); - } - - #[test] - fn test_retain_fragment_bitmap_with_nonexistent_fragments() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1), Fragment::new(2)]; // Only fragments 1 and 2 exist - - let mut indices = vec![create_test_index( - "idx", - 1, - 1, - Some(RoaringBitmap::from_iter([1, 2, 3, 4])), // References non-existent fragments 3, 4 - false, - )]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Should still keep the index (effective bitmap will be intersection with existing) - assert_eq!(indices.len(), 1); - // Original bitmap should be unchanged - assert_eq!( - indices[0].fragment_bitmap.as_ref().unwrap(), - &RoaringBitmap::from_iter([1, 2, 3, 4]) - ); - } - - #[test] - fn test_retain_effective_empty_bitmap_single_index() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(5), Fragment::new(6)]; - - // Bitmap references fragments that don't exist, so effective bitmap is empty - let mut scalar_indices = vec![create_test_index( - "scalar_idx", - 1, - 1, - Some(RoaringBitmap::from_iter([1, 2, 3])), - false, - )]; - - let mut vector_indices = vec![create_test_index( - "vector_idx", - 1, - 1, - Some(RoaringBitmap::from_iter([1, 2, 3])), - true, - )]; - - Transaction::retain_relevant_indices(&mut scalar_indices, &schema, &fragments); - Transaction::retain_relevant_indices(&mut vector_indices, &schema, &fragments); - - // Scalar should be kept (single index, even if effective bitmap is empty) - // Vector should be removed (empty effective bitmap) - assert_eq!(scalar_indices.len(), 1); - assert_eq!(vector_indices.len(), 0); - } - - #[test] - fn test_retain_different_index_names() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_test_index("idx_a", 1, 1, Some(RoaringBitmap::new()), false), - create_test_index("idx_b", 1, 1, Some(RoaringBitmap::new()), true), - create_test_index("idx_c", 1, 1, Some(RoaringBitmap::from_iter([1])), false), - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // idx_a (empty scalar) should be kept, idx_b (empty vector) removed, idx_c (non-empty) kept - assert_eq!(indices.len(), 2); - assert!(indices.iter().any(|idx| idx.name == "idx_a")); - assert!(indices.iter().any(|idx| idx.name == "idx_c")); - assert!(!indices.iter().any(|idx| idx.name == "idx_b")); - } - - #[test] - fn test_retain_empty_indices_vec() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices: Vec = vec![]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - assert_eq!(indices.len(), 0); - } - - #[test] - fn test_retain_all_indices_removed() { - let schema = create_test_schema(&[1]); - let fragments = vec![Fragment::new(1)]; - - let mut indices = vec![ - create_test_index("vec1", 1, 1, Some(RoaringBitmap::new()), true), - create_test_index("vec2", 1, 1, Some(RoaringBitmap::new()), true), - create_test_index("idx3", 99, 1, Some(RoaringBitmap::from_iter([1])), false), // Bad field - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - assert_eq!(indices.len(), 0); - } - - #[test] - fn test_retain_complex_scenario() { - let schema = create_test_schema(&[1, 2]); - let fragments = vec![Fragment::new(1), Fragment::new(2)]; - - let mut indices = vec![ - // System index - should always be kept - create_system_index(FRAG_REUSE_INDEX_NAME, 1), - // Group "idx_a" - all empty scalars, keep oldest - create_test_index("idx_a", 1, 3, Some(RoaringBitmap::new()), false), - create_test_index("idx_a", 1, 1, Some(RoaringBitmap::new()), false), // Oldest - create_test_index("idx_a", 1, 2, Some(RoaringBitmap::new()), false), - // Group "vec_b" - all empty vectors, remove all - create_test_index("vec_b", 1, 1, Some(RoaringBitmap::new()), true), - create_test_index("vec_b", 1, 2, Some(RoaringBitmap::new()), true), - // Group "idx_c" - mixed empty/non-empty, keep non-empty - create_test_index("idx_c", 2, 1, Some(RoaringBitmap::new()), false), - create_test_index("idx_c", 2, 2, Some(RoaringBitmap::from_iter([1])), false), // Keep - create_test_index("idx_c", 2, 3, Some(RoaringBitmap::from_iter([2])), false), // Keep - // Single non-empty - keep - create_test_index("idx_d", 1, 1, Some(RoaringBitmap::from_iter([1, 2])), false), - // Index with bad field - remove - create_test_index("idx_e", 99, 1, Some(RoaringBitmap::from_iter([1])), false), - ]; - - Transaction::retain_relevant_indices(&mut indices, &schema, &fragments); - - // Expected: frag_reuse, idx_a (oldest), idx_c (2 non-empty), idx_d = 5 total - assert_eq!(indices.len(), 5); - - // Verify system index kept - assert!(indices.iter().any(|idx| idx.name == FRAG_REUSE_INDEX_NAME)); - - // Verify idx_a kept oldest only - let idx_a_indices: Vec<_> = indices.iter().filter(|idx| idx.name == "idx_a").collect(); - assert_eq!(idx_a_indices.len(), 1); - assert_eq!(idx_a_indices[0].dataset_version, 1); - - // Verify vec_b all removed - assert!(!indices.iter().any(|idx| idx.name == "vec_b")); - - // Verify idx_c kept non-empty only - let idx_c_indices: Vec<_> = indices.iter().filter(|idx| idx.name == "idx_c").collect(); - assert_eq!(idx_c_indices.len(), 2); - assert!( - idx_c_indices - .iter() - .all(|idx| idx.dataset_version == 2 || idx.dataset_version == 3) - ); - - // Verify idx_d kept - assert!(indices.iter().any(|idx| idx.name == "idx_d")); - - // Verify idx_e removed (bad field) - assert!(!indices.iter().any(|idx| idx.name == "idx_e")); - } - - #[test] - fn test_handle_rewrite_indices_skips_missing_index() { - use uuid::Uuid; - - // Create an empty indices list - let mut indices = vec![]; - - // Create rewritten_indices referring to a non-existent index - let rewritten_indices = vec![RewrittenIndex { - old_id: Uuid::new_v4(), - new_id: Uuid::new_v4(), - new_index_details: prost_types::Any { - type_url: String::new(), - value: vec![], - }, - new_index_version: 1, - new_index_files: None, - }]; - - // Should succeed (skip missing index) instead of error - let result = Transaction::handle_rewrite_indices(&mut indices, &rewritten_indices, &[]); - assert!(result.is_ok()); - assert!(indices.is_empty()); - } - - /// When a fragment has no existing last_updated_at_version_meta (None), a - /// partial RewriteColumns refresh must leave it as None rather than fabricating - /// prev_version for unmatched rows. - #[test] - fn test_partial_rewrite_skips_fragment_with_no_version_meta() { - let row_ids = RowIdSequence::from([10u64, 11, 12, 13, 14].as_slice()); - let row_id_meta = Some(RowIdMeta::Inline(write_row_ids(&row_ids))); - - let (major, minor) = lance_file::version::LanceFileVersion::Stable.to_numbers(); - let data_file = DataFile::new("data.lance", vec![0], vec![0], major, minor, None, None); - - let fragment = Fragment { - id: 1, - files: vec![data_file], - overlays: vec![], - deletion_file: None, - row_id_meta, - physical_rows: Some(5), - last_updated_at_version_meta: None, - created_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![fragment.clone()]); - - // Simulate a RewriteColumns update that matched offsets 1 and 3 - let off_map = HashMap::from([(1u64, RoaringBitmap::from_iter([1u32, 3]))]); - let tx = Transaction::new( - manifest.version, - Operation::Update { - removed_fragment_ids: vec![], - updated_fragments: vec![fragment], - new_fragments: vec![], - fields_modified: vec![], - compacted_sstables: vec![], - fields_for_preserving_frag_bitmap: vec![], - update_mode: Some(UpdateMode::RewriteColumns), - inserted_rows_filter: None, - updated_fragment_offsets: Some(UpdatedFragmentOffsets(off_map)), - }, - None, - ); - - let (out, _) = tx - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert!( - out.fragments[0].last_updated_at_version_meta.is_none(), - "fragment with no prior version metadata must not have fabricated prev_version stamped on unmatched rows" - ); - } - - /// Partial RewriteColumns refresh in `build_manifest`: only matched physical - /// rows get `last_updated_at_version` bumped; same-fragment unmatched rows and - /// untouched fragments keep both version sequences. - #[tokio::test] - async fn test_build_manifest_partial_last_updated_rewrite_columns_stable_row_ids() { - let dir = TempStrDir::default(); - let uri = dir.as_str(); - - let schema = Arc::new(ArrowSchema::new(vec![ - ArrowField::new("i", DataType::Int32, false), - ArrowField::new("x", DataType::Int32, false), - ])); - let batch0 = RecordBatch::try_new( - schema.clone(), - vec![ - Arc::new(Int32Array::from_iter_values(0..8)), - Arc::new(Int32Array::from(vec![0_i32; 8])), - ], - ) - .unwrap(); - let reader0 = RecordBatchIterator::new(vec![Ok(batch0)], schema.clone()); - let write_params = WriteParams { - enable_stable_row_ids: true, - data_storage_version: Some(LanceFileVersion::Stable), - ..Default::default() - }; - let mut dataset = Dataset::write(reader0, uri, Some(write_params)) - .await - .unwrap(); - - let batch1 = RecordBatch::try_new( - schema.clone(), - vec![ - Arc::new(Int32Array::from_iter_values(100..108)), - Arc::new(Int32Array::from(vec![0_i32; 8])), - ], - ) - .unwrap(); - let reader1 = RecordBatchIterator::new(vec![Ok(batch1)], schema.clone()); - dataset.append(reader1, None).await.unwrap(); - - let frags = dataset.get_fragments(); - assert_eq!( - frags.len(), - 2, - "expected two fragments (append creates a new fragment)" - ); - - async fn scan_row_versions(ds: &Dataset) -> HashMap<(u32, u32), (u64, u64)> { - let mut scanner = ds.scan(); - scanner - .project(&[ - ROW_ADDR, - ROW_LAST_UPDATED_AT_VERSION, - ROW_CREATED_AT_VERSION, - ]) - .unwrap(); - let batches = scanner - .try_into_stream() - .await - .unwrap() - .try_collect::>() - .await - .unwrap(); - let mut out = HashMap::new(); - for batch in batches { - let addrs = batch - .column_by_name(ROW_ADDR) - .unwrap() - .as_primitive::(); - let last = batch - .column_by_name(ROW_LAST_UPDATED_AT_VERSION) - .unwrap() - .as_primitive::(); - let created = batch - .column_by_name(ROW_CREATED_AT_VERSION) - .unwrap() - .as_primitive::(); - for row in 0..batch.num_rows() { - let addr = RowAddress::from(addrs.value(row)); - out.insert( - (addr.fragment_id(), addr.row_offset()), - (last.value(row), created.value(row)), - ); - } - } - out - } - - let before = scan_row_versions(&dataset).await; - assert_eq!(before.len(), 16); - - // Update only rows i in {2, 4, 6} within fragment 0 (physical offsets 2, 4, 6). - let update_schema = Arc::new(ArrowSchema::new(vec![ - ArrowField::new("i", DataType::Int32, false), - ArrowField::new("x", DataType::Int32, false), - ])); - let update_batch = RecordBatch::try_new( - update_schema.clone(), - vec![ - Arc::new(Int32Array::from(vec![2, 4, 6])), - Arc::new(Int32Array::from(vec![99, 99, 99])), - ], - ) - .unwrap(); - let right: Box = Box::new( - RecordBatchIterator::new(vec![Ok(update_batch)].into_iter(), update_schema), - ); - - let mut frag0 = dataset.get_fragment(0).unwrap(); - let u = frag0 - .update_columns_with_offsets(right, "i", "i") - .await - .unwrap(); - assert_eq!(u.matched_offsets.iter().count(), 3); - for off in [2_u32, 4, 6] { - assert!(u.matched_offsets.contains(off)); - } - - let updated_fragment_offsets = Some(UpdatedFragmentOffsets(HashMap::from([( - u.fragment.id, - u.matched_offsets, - )]))); - - let op = Operation::Update { - removed_fragment_ids: vec![], - updated_fragments: vec![u.fragment], - new_fragments: vec![], - fields_modified: u.fields_modified, - compacted_sstables: Vec::new(), - fields_for_preserving_frag_bitmap: vec![], - update_mode: Some(UpdateMode::RewriteColumns), - inserted_rows_filter: None, - updated_fragment_offsets, - }; - - let read_v = dataset.version().version; - let dataset = Dataset::commit( - uri, - op, - Some(read_v), - None, - None, - Arc::new(Session::default()), - true, - ) - .await - .unwrap(); - - let new_v = dataset.version().version; - assert_eq!(new_v, read_v + 1); - - let after = scan_row_versions(&dataset).await; - for off in 0..8_u32 { - let key = (0, off); - let (last_before, created_before) = before[&key]; - let (last_after, created_after) = after[&key]; - assert_eq!(created_after, created_before); - if off == 2 || off == 4 || off == 6 { - assert_eq!( - last_after, new_v, - "matched row offset {off} should advance last_updated to new version" - ); - } else { - assert_eq!( - last_after, last_before, - "unmatched row offset {off} in fragment 0 should keep last_updated" - ); - } - } - - for off in 0..8_u32 { - let key = (1, off); - assert_eq!( - after[&key], before[&key], - "fragment 1 row offset {off}: both version columns unchanged" - ); - } - } - - /// Regression test for https://github.com/lance-format/lance/issues/6417 - /// - /// When overwriting a LEGACY dataset with STABLE-format fragments, the - /// validation should not use the old manifest's format. STABLE fragments - /// omit struct parent fields, which the strict legacy check rejects. - #[test] - fn test_overwrite_legacy_to_stable_with_struct_fields() { - use arrow_schema::Fields; - - // Schema: id (field 0), name (field 1), address (field 2, struct parent), - // city (field 3), country (field 4) - let arrow_schema = ArrowSchema::new(vec![ - ArrowField::new("id", DataType::Int32, false), - ArrowField::new("name", DataType::Utf8, false), - ArrowField::new( - "address", - DataType::Struct(Fields::from(vec![ - ArrowField::new("city", DataType::Utf8, false), - ArrowField::new("country", DataType::Utf8, false), - ])), - false, - ), - ]); - let schema = LanceSchema::try_from(&arrow_schema).unwrap(); - - // Old manifest is LEGACY format - let legacy_manifest = Manifest::new( - schema.clone(), - Arc::new(vec![Fragment::new(0)]), - DataStorageFormat::new(LanceFileVersion::Legacy), - HashMap::new(), - ); - - // New fragments in STABLE format omit struct parent field (id=2), - // only including leaf fields: id=0, name=1, city=3, country=4 - let stable_fragment = Fragment { - id: 0, - files: vec![DataFile::new( - "data.lance", - vec![0, 1, 3, 4], // no field 2 (struct parent) - vec![0, 1, 2, 3], - lance_file::format::MAJOR_VERSION as u32, - lance_file::format::MINOR_VERSION as u32, - None, - None, - )], - physical_rows: Some(10), - overlays: vec![], - deletion_file: None, - row_id_meta: None, - last_updated_at_version_meta: None, - created_at_version_meta: None, - }; - - let operation = Operation::Overwrite { - fragments: vec![stable_fragment], - schema, - config_upsert_values: None, - initial_bases: None, - }; - - // This should succeed — the old manifest's LEGACY format should not - // cause strict validation of the new STABLE fragments. - validate_operation(Some(&legacy_manifest), &operation).unwrap(); - } - - /// Existing fragments use id >= 1 to avoid collision with `Fragment::new(0)` - /// used by `sample_manifest`. New (updated) fragments use id = 10. - fn make_stable_row_id_manifest(fragments: Vec) -> Manifest { - let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let mut manifest = Manifest::new( - LanceSchema::try_from(&schema).unwrap(), - Arc::new(fragments), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - manifest.reader_feature_flags = FLAG_STABLE_ROW_IDS; - manifest.next_row_id = 1000; - manifest.version = 4; - manifest - } - - fn update_txn(new_fragments: Vec) -> Transaction { - Transaction::new( - 4, - Operation::Update { - removed_fragment_ids: vec![], - updated_fragments: vec![], - new_fragments, - fields_modified: vec![], - compacted_sstables: vec![], - fields_for_preserving_frag_bitmap: vec![], - update_mode: None, - inserted_rows_filter: None, - updated_fragment_offsets: None, - }, - None, - ) - } - - fn created_at_versions(manifest: &Manifest, frag_id: u64) -> Vec { - let frag = manifest.fragments.iter().find(|f| f.id == frag_id).unwrap(); - let seq = frag - .created_at_version_meta - .as_ref() - .unwrap() - .load_sequence() - .unwrap(); - seq.versions().collect() - } - - fn last_updated_at_versions(manifest: &Manifest, frag_id: u64) -> Vec { - let frag = manifest.fragments.iter().find(|f| f.id == frag_id).unwrap(); - let seq = frag - .last_updated_at_version_meta - .as_ref() - .unwrap() - .load_sequence() - .unwrap(); - seq.versions().collect() - } - - #[test] - fn merge_build_manifest_refreshes_last_updated_when_data_files_change_stable_row_ids() { - use lance_file::version::LanceFileVersion; - use lance_table::feature_flags::FLAG_STABLE_ROW_IDS; - - let (major, minor) = LanceFileVersion::Stable.to_numbers(); - let mk_file = |path: &str| DataFile::new(path, vec![0], vec![0], major, minor, None, None); - - let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); - - let row_ids = RowIdSequence::from([100u64, 101, 102, 103, 104].as_slice()); - let row_id_meta = Some(RowIdMeta::Inline(write_row_ids(&row_ids))); - - let prev_fragment = Fragment { - id: 0, - files: vec![mk_file("before.lance")], - overlays: vec![], - deletion_file: None, - row_id_meta, - physical_rows: Some(5), - last_updated_at_version_meta: None, - created_at_version_meta: None, - }; - - let mut manifest = Manifest::new( - lance_schema.clone(), - Arc::new(vec![prev_fragment.clone()]), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - manifest.reader_feature_flags |= FLAG_STABLE_ROW_IDS; - manifest.next_row_id = 100; - - let merged_fragment = Fragment { - files: vec![mk_file("after.lance")], - ..prev_fragment - }; - - let tx = Transaction::new( - manifest.version, - Operation::Merge { - fragments: vec![merged_fragment], - schema: lance_schema, - }, - None, - ); - - let (out, _) = tx - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert_eq!(out.version, 2); - let frag = &out.fragments[0]; - let seq = frag - .last_updated_at_version_meta - .as_ref() - .unwrap() - .load_sequence() - .unwrap(); - assert_eq!(seq.version_at(0).unwrap(), 2); - assert_eq!(seq.version_at(4).unwrap(), 2); - } - - #[test] - fn merge_build_manifest_skips_refresh_when_carry_forward_stable_row_ids() { - use lance_file::version::LanceFileVersion; - use lance_table::feature_flags::FLAG_STABLE_ROW_IDS; - use lance_table::rowids::version::{RowDatasetVersionMeta, RowDatasetVersionSequence}; - - let (major, minor) = LanceFileVersion::Stable.to_numbers(); - let data_file = DataFile::new("same.lance", vec![0], vec![0], major, minor, None, None); - - let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); - - let row_ids = RowIdSequence::from([200u64, 201, 202, 203, 204].as_slice()); - let row_id_meta = Some(RowIdMeta::Inline(write_row_ids(&row_ids))); - - let uniform_v1 = RowDatasetVersionSequence::from_uniform_row_count(5, 1); - let meta_v1 = RowDatasetVersionMeta::from_sequence(&uniform_v1).unwrap(); - - let prev_fragment = Fragment { - id: 0, - files: vec![data_file.clone()], - overlays: vec![], - deletion_file: None, - row_id_meta: row_id_meta.clone(), - physical_rows: Some(5), - last_updated_at_version_meta: Some(meta_v1.clone()), - created_at_version_meta: None, - }; - - let mut manifest = Manifest::new( - lance_schema.clone(), - Arc::new(vec![prev_fragment]), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - manifest.reader_feature_flags |= FLAG_STABLE_ROW_IDS; - manifest.next_row_id = 100; - - let merged_fragment = Fragment { - id: 0, - files: vec![data_file], - overlays: vec![], - deletion_file: None, - row_id_meta, - physical_rows: Some(5), - last_updated_at_version_meta: Some(meta_v1), - created_at_version_meta: None, - }; - - let tx = Transaction::new( - manifest.version, - Operation::Merge { - fragments: vec![merged_fragment], - schema: lance_schema, - }, - None, - ); - - let (out, _) = tx - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - let seq = out.fragments[0] - .last_updated_at_version_meta - .as_ref() - .unwrap() - .load_sequence() - .unwrap(); - assert_eq!(seq.version_at(0).unwrap(), 1); - assert_eq!(seq.version_at(4).unwrap(), 1); - } - - #[test] - fn merge_build_manifest_no_last_updated_refresh_without_stable_row_ids() { - use lance_file::version::LanceFileVersion; - use lance_table::feature_flags::FLAG_STABLE_ROW_IDS; - - let (major, minor) = LanceFileVersion::Stable.to_numbers(); - let mk_file = |path: &str| DataFile::new(path, vec![0], vec![0], major, minor, None, None); - - let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); - - let prev_fragment = Fragment { - id: 0, - files: vec![mk_file("before.lance")], - overlays: vec![], - deletion_file: None, - row_id_meta: None, - physical_rows: Some(5), - last_updated_at_version_meta: None, - created_at_version_meta: None, - }; - - let manifest = Manifest::new( - lance_schema.clone(), - Arc::new(vec![prev_fragment.clone()]), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - assert_eq!( - manifest.reader_feature_flags & FLAG_STABLE_ROW_IDS, - 0, - "manifest must not use stable row IDs for this guard test" - ); - - let merged_fragment = Fragment { - files: vec![mk_file("after.lance")], - ..prev_fragment - }; - - let tx = Transaction::new( - manifest.version, - Operation::Merge { - fragments: vec![merged_fragment], - schema: lance_schema, - }, - None, - ); - - let (out, _) = tx - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert!( - out.fragments[0].last_updated_at_version_meta.is_none(), - "without stable row IDs, Merge must not populate per-row last_updated metadata" - ); - } - - #[test] - fn merge_build_manifest_sets_both_version_meta_for_new_fragment_id_stable_row_ids() { - use lance_file::version::LanceFileVersion; - use lance_table::feature_flags::FLAG_STABLE_ROW_IDS; - - let (major, minor) = LanceFileVersion::Stable.to_numbers(); - let mk_file = |path: &str| DataFile::new(path, vec![0], vec![0], major, minor, None, None); - - let arrow_schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let lance_schema = LanceSchema::try_from(&arrow_schema).unwrap(); - - // Existing fragment (id=0) with stable row IDs - let row_ids_0 = RowIdSequence::from([10u64, 11, 12].as_slice()); - let existing_fragment = Fragment { - id: 0, - files: vec![mk_file("existing.lance")], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&row_ids_0))), - physical_rows: Some(3), - last_updated_at_version_meta: None, - created_at_version_meta: None, - }; - - let mut manifest = Manifest::new( - lance_schema.clone(), - Arc::new(vec![existing_fragment.clone()]), - DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - manifest.reader_feature_flags |= FLAG_STABLE_ROW_IDS; - manifest.next_row_id = 100; - manifest.version = 1; - - // New fragment (id=1) not present in prev manifest — exercises the None branch - let row_ids_1 = RowIdSequence::from([20u64, 21, 22, 23].as_slice()); - let new_fragment = Fragment { - id: 1, - files: vec![mk_file("new.lance")], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&row_ids_1))), - physical_rows: Some(4), - last_updated_at_version_meta: None, - created_at_version_meta: None, - }; - - let tx = Transaction::new( - manifest.version, - Operation::Merge { - fragments: vec![existing_fragment, new_fragment], - schema: lance_schema, - }, - None, - ); - - let (out, _) = tx - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert_eq!(out.version, 2); - - let new_frag = out.fragments.iter().find(|f| f.id == 1).unwrap(); - - // last_updated_at_version must be set to the commit version - let last_updated_seq = new_frag - .last_updated_at_version_meta - .as_ref() - .expect("new fragment must have last_updated_at_version_meta") - .load_sequence() - .unwrap(); - assert_eq!(last_updated_seq.version_at(0).unwrap(), 2); - assert_eq!(last_updated_seq.version_at(3).unwrap(), 2); - - // created_at_version must also be set — must not be None - let created_seq = new_frag - .created_at_version_meta - .as_ref() - .expect("new fragment must have created_at_version_meta") - .load_sequence() - .unwrap(); - assert_eq!(created_seq.version_at(0).unwrap(), 2); - assert_eq!(created_seq.version_at(3).unwrap(), 2); - } - - #[test] - fn test_update_version_tracking_preserves_created_at() { - let existing_seq = RowIdSequence::from([100u64, 101, 102].as_slice()); - let created_at_seq = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..3), - version: 5, - }], - }; - let existing_fragment = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), - physical_rows: Some(3), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&created_at_seq).unwrap(), - ), - last_updated_at_version_meta: None, - }; - - let new_seq = RowIdSequence::from([100u64, 102].as_slice()); - let new_fragment = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![existing_fragment]); - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - assert_eq!(created_at_versions(&result, 10), vec![5, 5]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); - } - - #[test] - fn test_update_version_tracking_mixed_origins() { - let frag_a_seq = RowIdSequence::from([10u64, 11].as_slice()); - let frag_a_created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..2), - version: 2, - }], - }; - let frag_b_seq = RowIdSequence::from([20u64, 21, 22].as_slice()); - let frag_b_created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..3), - version: 3, - }], - }; - - let manifest = make_stable_row_id_manifest(vec![ - Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&frag_a_seq))), - physical_rows: Some(2), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&frag_a_created).unwrap(), - ), - last_updated_at_version_meta: None, - }, - Fragment { - id: 2, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&frag_b_seq))), - physical_rows: Some(3), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&frag_b_created).unwrap(), - ), - last_updated_at_version_meta: None, - }, - ]); - - // New fragment has rows from both original fragments: row 11 from frag_a, row 20 from frag_b - let new_seq = RowIdSequence::from([11u64, 20].as_slice()); - let new_fragment = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Row 11 came from frag_a (offset 1, version 2), row 20 came from frag_b (offset 0, version 3) - assert_eq!(created_at_versions(&result, 10), vec![2, 3]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); - } - - #[test] - fn test_update_version_tracking_insert_branch_gets_new_version() { - // Simulates the INSERT branch (NOT MATCHED) of a MERGE INTO commit: - // the new fragment contains a mix of rewritten rows (UPDATE branch, row ID - // present in existing fragments) and freshly inserted rows (INSERT branch, - // row ID not present in any existing fragment). - // - // UPDATE branch row (10): created_at must be copied from the source fragment. - // INSERT branch row (999): created_at must equal new_version (the merge commit - // version), because the row first appeared in this commit. - let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); - let existing_created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..2), - version: 5, - }], - }; - let existing_fragment = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), - physical_rows: Some(2), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&existing_created).unwrap(), - ), - last_updated_at_version_meta: None, - }; - - // New fragment has row 10 (UPDATE branch) and row 999 (INSERT branch) - let new_seq = RowIdSequence::from([10u64, 999].as_slice()); - let new_fragment = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - // update_txn uses read_version 4 → new_version is 5 - let manifest = make_stable_row_id_manifest(vec![existing_fragment]); - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Row 10 (UPDATE branch): created_at copied from source (version 5). - // Row 999 (INSERT branch): created_at == new_version (5). - assert_eq!(created_at_versions(&result, 10), vec![5, 5]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); - } - - #[test] - fn test_update_version_tracking_merge_into_distinguishes_insert_and_update_branch() { - // Verifies the MERGE INTO correctness contract when UPDATE branch rows and INSERT - // branch rows have *different* source created_at values, so we can distinguish - // which row got which value. - // - // Existing fragment (id=1): row IDs [10, 11], created_at = version 3. - // New fragment (id=20): row IDs [10, 500, 11, 501]. - // - Rows 10 and 11: UPDATE branch (present in existing fragment) → created_at = 3. - // - Rows 500 and 501: INSERT branch (no source) → created_at = new_version = 5. - let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); - let existing_created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..2), - version: 3, - }], - }; - let existing_fragment = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), - physical_rows: Some(2), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&existing_created).unwrap(), - ), - last_updated_at_version_meta: None, - }; - - let new_seq = RowIdSequence::from([10u64, 500, 11, 501].as_slice()); - let new_fragment = Fragment { - id: 20, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(4), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - // update_txn uses read_version 4 → new_version is 5 - let manifest = make_stable_row_id_manifest(vec![existing_fragment]); - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // UPDATE branch rows (10, 11): created_at preserved from source (version 3). - // INSERT branch rows (500, 501): created_at == new_version (5). - assert_eq!(created_at_versions(&result, 20), vec![3, 5, 3, 5]); - // All rows in the new fragment get last_updated == new_version. - assert_eq!(last_updated_at_versions(&result, 20), vec![5, 5, 5, 5]); - } - - #[test] - fn test_update_version_tracking_source_fragment_no_created_at_defaults_to_1() { - // Source fragment has row_id_meta but no created_at_version_meta. - // The row IS found in the lookup, but the version defaults to 1. - let existing_seq = RowIdSequence::from([50u64, 51].as_slice()); - let existing_fragment = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let new_seq = RowIdSequence::from([50u64].as_slice()); - let new_fragment = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(1), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![existing_fragment]); - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Row 50 is found in source but source has no created_at_version_meta → default 1 - assert_eq!(created_at_versions(&result, 10), vec![1]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5]); - } - - #[test] - fn test_update_version_tracking_no_row_id_meta_fallback() { - let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); - let existing_fragment = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let new_fragment = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: None, - physical_rows: Some(3), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![existing_fragment]); - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Fragment starts with no row_id_meta → assign_row_ids gives it fresh IDs → - // those IDs have no source in existing fragments (INSERT branch) → - // created_at == new_version (5) for each row. - assert_eq!(created_at_versions(&result, 10), vec![5, 5, 5]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5, 5]); - } - - #[test] - fn test_update_version_tracking_corrupt_created_at_defaults_to_1() { - let existing_seq = RowIdSequence::from([10u64, 11].as_slice()); - let existing_fragment = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&existing_seq))), - physical_rows: Some(2), - created_at_version_meta: Some(RowDatasetVersionMeta::Inline(Arc::from( - vec![0xFFu8; 8].as_slice(), - ))), - last_updated_at_version_meta: None, - }; - - let new_seq = RowIdSequence::from([10u64].as_slice()); - let new_fragment = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(1), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![existing_fragment]); - let (result, _) = update_txn(vec![new_fragment]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Corrupt metadata causes decode to fail → falls back to UNKNOWN_CREATED_AT_VERSION (1) - assert_eq!(created_at_versions(&result, 10), vec![1]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5]); - } - - // --- Proposal 1: range pre-filter --- - - /// Fragments whose row-ID range lies entirely outside the needed set must not - /// affect the result. Here fragment 1 has IDs [1000, 1001] which are far above - /// the needed range [10, 11]; it is skipped by the range pre-filter and its - /// created_at version (version 99) must never appear in the output. - #[test] - fn test_update_version_tracking_range_filter_skips_non_overlapping_fragment() { - // Fragment in range – IDs [10, 11], created_at = 5 - let in_range_seq = RowIdSequence::from([10u64, 11].as_slice()); - let in_range_created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..2), - version: 5, - }], - }; - let in_range_frag = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&in_range_seq))), - physical_rows: Some(2), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&in_range_created).unwrap(), - ), - last_updated_at_version_meta: None, - }; - - // Fragment outside range – IDs [1000, 1001], created_at = 99 (must never appear) - let out_of_range_seq = RowIdSequence::from([1000u64, 1001].as_slice()); - let out_of_range_created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..2), - version: 99, - }], - }; - let out_of_range_frag = Fragment { - id: 2, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&out_of_range_seq))), - physical_rows: Some(2), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&out_of_range_created).unwrap(), - ), - last_updated_at_version_meta: None, - }; - - // New fragment rewrites both rows from the in-range fragment - let new_seq = RowIdSequence::from([10u64, 11].as_slice()); - let new_frag = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![in_range_frag, out_of_range_frag]); - let (result, _) = update_txn(vec![new_frag]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Both rows originate from the in-range fragment (version 5). - // The out-of-range fragment's version 99 must not appear. - assert_eq!(created_at_versions(&result, 10), vec![5, 5]); - assert_eq!(last_updated_at_versions(&result, 10), vec![5, 5]); - } - - /// When the needed row IDs fall exactly at the boundary of a fragment's range, - /// the range pre-filter must NOT skip the fragment (boundary values are inclusive). - #[test] - fn test_update_version_tracking_range_filter_boundary_inclusive() { - // Fragment IDs [10, 11, 12], created_at = 7 - let seq = RowIdSequence::from([10u64, 11, 12].as_slice()); - let created = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..3), - version: 7, - }], - }; - let existing = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&seq))), - physical_rows: Some(3), - created_at_version_meta: Some(RowDatasetVersionMeta::from_sequence(&created).unwrap()), - last_updated_at_version_meta: None, - }; - - // New fragment takes the boundary IDs: 10 (min) and 12 (max) - let new_seq = RowIdSequence::from([10u64, 12].as_slice()); - let new_frag = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![existing]); - let (result, _) = update_txn(vec![new_frag]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Boundary IDs must be found and resolved correctly - assert_eq!(created_at_versions(&result, 10), vec![7, 7]); - } - - // --- Proposal 2: version sequence cache --- - - /// When multiple updated rows all originate from the same source fragment, - /// the created_at version sequence for that fragment must be decoded exactly - /// once (not once per row). The observable correctness requirement is that - /// all rows get the right version regardless of how many there are. - #[test] - fn test_update_version_tracking_many_rows_same_source_fragment() { - // Source fragment: 100 rows with IDs 0..100, mixed versions (2 runs). - // First 50 rows at version 3, next 50 rows at version 4. - let src_ids: Vec = (0u64..100).collect(); - let src_seq = RowIdSequence::from(src_ids.as_slice()); - let src_created = RowDatasetVersionSequence { - runs: vec![ - RowDatasetVersionRun { - span: U64Segment::Range(0..50), - version: 3, - }, - RowDatasetVersionRun { - span: U64Segment::Range(0..50), - version: 4, - }, - ], - }; - let src_frag = Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&src_seq))), - physical_rows: Some(100), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&src_created).unwrap(), - ), - last_updated_at_version_meta: None, - }; - - // New fragment rewrites all 100 rows preserving their stable IDs. - let new_seq = RowIdSequence::from(src_ids.as_slice()); - let new_frag = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(100), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let manifest = make_stable_row_id_manifest(vec![src_frag]); - let (result, _) = update_txn(vec![new_frag]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - let versions = created_at_versions(&result, 10); - assert_eq!(versions.len(), 100); - // First 50 rows came from version 3, next 50 from version 4 - assert!(versions[..50].iter().all(|&v| v == 3)); - assert!(versions[50..].iter().all(|&v| v == 4)); - } - - /// Rows originating from multiple distinct source fragments must each get - /// the version from their own source, even when all cached together. - #[test] - fn test_update_version_tracking_cache_multiple_source_fragments() { - let seq_a = RowIdSequence::from([10u64, 11, 12].as_slice()); - let created_a = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..3), - version: 2, - }], - }; - let seq_b = RowIdSequence::from([20u64, 21, 22].as_slice()); - let created_b = RowDatasetVersionSequence { - runs: vec![RowDatasetVersionRun { - span: U64Segment::Range(0..3), - version: 8, - }], - }; - - let manifest = make_stable_row_id_manifest(vec![ - Fragment { - id: 1, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&seq_a))), - physical_rows: Some(3), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&created_a).unwrap(), - ), - last_updated_at_version_meta: None, - }, - Fragment { - id: 2, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&seq_b))), - physical_rows: Some(3), - created_at_version_meta: Some( - RowDatasetVersionMeta::from_sequence(&created_b).unwrap(), - ), - last_updated_at_version_meta: None, - }, - ]); - - // New fragment takes rows from both sources: 12 (frag A, offset 2) and 20 (frag B, offset 0) - let new_seq = RowIdSequence::from([12u64, 20].as_slice()); - let new_frag = Fragment { - id: 10, - files: vec![], - overlays: vec![], - deletion_file: None, - row_id_meta: Some(RowIdMeta::Inline(write_row_ids(&new_seq))), - physical_rows: Some(2), - created_at_version_meta: None, - last_updated_at_version_meta: None, - }; - - let (result, _) = update_txn(vec![new_frag]) - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - // Row 12 → frag A offset 2 → version 2; row 20 → frag B offset 0 → version 8 - assert_eq!(created_at_versions(&result, 10), vec![2, 8]); - } - - #[test] - fn test_encode_version_runs_empty() { - let runs = encode_version_runs(&[]); - assert!(runs.is_empty()); - } - - #[test] - fn test_encode_version_runs_single_run() { - let runs = encode_version_runs(&[3, 3, 3]); - assert_eq!(runs.len(), 1); - assert_eq!(runs[0].version, 3); - } - - #[test] - fn test_encode_version_runs_alternating() { - let runs = encode_version_runs(&[1, 2, 1, 2]); - assert_eq!(runs.len(), 4); - assert_eq!(runs[0].version, 1); - assert_eq!(runs[1].version, 2); - assert_eq!(runs[2].version, 1); - assert_eq!(runs[3].version, 2); - } - - fn table_metadata_update(entries: Vec<(&str, Option<&str>)>, replace: bool) -> Operation { - Operation::UpdateConfig { - config_updates: None, - table_metadata_updates: Some(UpdateMap { - update_entries: entries.into_iter().map(UpdateMapEntry::from).collect(), - replace, - }), - schema_metadata_updates: None, - field_metadata_updates: HashMap::new(), - } - } - - #[test] - fn test_table_metadata_conflicts_on_same_key() { - let left = table_metadata_update(vec![("key", Some("1"))], false); - let same_key = table_metadata_update(vec![("key", Some("2"))], false); - let different_key = table_metadata_update(vec![("other", Some("2"))], false); - let replace = table_metadata_update(vec![("other", Some("2"))], true); - - assert!(left.modifies_same_metadata(&same_key)); - assert!(!left.modifies_same_metadata(&different_key)); - assert!(left.modifies_same_metadata(&replace)); - } - - #[test] - fn test_data_overlay_operation_roundtrips() { - // A DataOverlay operation survives the protobuf round-trip, preserving - // the target fragment, the overlay's coverage, and its committed_version. - let mut bitmap = roaring::RoaringBitmap::new(); - bitmap.insert(1); - bitmap.insert(4); - let overlay = DataOverlayFile { - data_file: DataFile::new_legacy_from_fields("overlay-0.lance", vec![3], None), - coverage: OverlayCoverage::dense(bitmap.clone()), - committed_version: 6, - }; - let pb_overlay = pb::DataOverlayFile::from(&overlay); - - let message = pb::Transaction { - read_version: 1, - uuid: Uuid::new_v4().to_string(), - operation: Some(pb::transaction::Operation::DataOverlay( - pb::transaction::DataOverlay { - groups: vec![pb::transaction::DataOverlayGroup { - fragment_id: 7, - overlays: vec![pb_overlay], - }], - }, - )), - ..Default::default() - }; - - let txn = Transaction::try_from(message).unwrap(); - match txn.operation { - Operation::DataOverlay { groups } => { - assert_eq!(groups.len(), 1); - assert_eq!(groups[0].fragment_id, 7); - assert_eq!(groups[0].overlays.len(), 1); - assert_eq!(groups[0].overlays[0].committed_version, 6); - assert_eq!( - *groups[0].overlays[0].coverage_for_field(0).unwrap(), - bitmap - ); - } - other => panic!("expected DataOverlay, got {other:?}"), - } - } - - fn overlay_with_field(field: i32, committed_version: u64) -> DataOverlayFile { - DataOverlayFile { - data_file: DataFile::new_legacy_from_fields("o.lance", vec![field], None), - coverage: OverlayCoverage::dense(roaring::RoaringBitmap::from_iter([0u32])), - committed_version, - } - } - - #[test] - fn test_prune_overlay_stale_fields_from_indices() { - // Fragment 0 carried an overlay on field 1 committed at v5, and was - // fully compacted into new fragment 7. - let mut old_frag = Fragment::new(0); - old_frag.overlays = vec![overlay_with_field(1, 5)]; - let groups = vec![RewriteGroup { - old_fragments: vec![old_frag], - new_fragments: vec![Fragment::new(7)], - }]; - - // Post-remap state: every index already covers the new fragment (7). - let covering = || Some(RoaringBitmap::from_iter([7u32])); - let mut indices = vec![ - // Stale: covers the overlaid field 1, built (v2) before the overlay. - create_test_index("stale", 1, 2, covering(), false), - // Not stale: covers field 1 but built at the overlay's version (v5); - // `committed_version > dataset_version` is false at equality. - create_test_index("fresh", 1, 5, covering(), false), - // Unrelated: covers field 2, which the overlay never touched. - create_test_index("unrelated", 2, 2, covering(), false), - ]; - - Transaction::prune_overlay_stale_fields_from_indices(&mut indices, &groups); - - assert!( - !indices[0].fragment_bitmap.as_ref().unwrap().contains(7), - "stale index must drop the rewritten fragment from its coverage" - ); - assert!( - indices[1].fragment_bitmap.as_ref().unwrap().contains(7), - "an index built at/after the overlay is not stale" - ); - assert!( - indices[2].fragment_bitmap.as_ref().unwrap().contains(7), - "an index on an un-overlaid field is unaffected" - ); - } - - #[test] - fn test_data_overlay_build_manifest_multi_fragment() { - // Overlays targeting two distinct fragments are each applied and stamped. - // A targeted fragment already carrying an overlay (committed at v3) gets - // the new overlay appended and stamped while its existing overlay is - // preserved, and a fragment the operation does not target is passed - // through with its existing overlays untouched. - let mut frag0 = Fragment::new(0); - frag0.overlays = vec![overlay_with_field(5, 3)]; // targeted, pre-existing at v3 - let frag1 = Fragment::new(1); - let mut frag2 = Fragment::new(2); - frag2.overlays = vec![overlay_with_field(9, 3)]; // untargeted, committed at v3 - let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let mut manifest = Manifest::new( - LanceSchema::try_from(&schema).unwrap(), - Arc::new(vec![frag0, frag1, frag2]), - lance_table::format::DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - // The pre-existing overlays were committed at v3, so the current - // manifest must be at least that version; the new commit then stamps - // its overlay at v4, keeping the fragment's overlays newest-last. - manifest.version = 3; - - let txn = Transaction::new( - manifest.version, - Operation::DataOverlay { - groups: vec![ - DataOverlayGroup { - fragment_id: 0, - overlays: vec![overlay_with_field(1, 0)], - }, - DataOverlayGroup { - fragment_id: 1, - overlays: vec![overlay_with_field(2, 0)], - }, - ], - }, - None, - ); - - let (result, _) = txn - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - let frag = |id: u64| { - result - .fragments - .iter() - .find(|f| f.id == id) - .unwrap_or_else(|| panic!("fragment {id} missing from result")) - }; - // The already-overlaid target keeps its v3 overlay and appends the new - // one, stamped to the new version. - assert_eq!(frag(0).overlays.len(), 2); - assert_eq!(frag(0).overlays[0].committed_version, 3); - assert_eq!(frag(0).overlays[1].committed_version, result.version); - // The fresh target gets its overlay, stamped to the new version. - assert_eq!(frag(1).overlays.len(), 1); - assert_eq!(frag(1).overlays[0].committed_version, result.version); - // The untargeted fragment is unchanged: same overlay, original version. - assert_eq!(frag(2).overlays.len(), 1); - assert_eq!(frag(2).overlays[0].committed_version, 3); - assert!(result.version > manifest.version); - } - - #[test] - fn test_data_replacement_tombstones_overlaid_fields() { - // A DataReplacement writing new base values for field 5 must stop any - // overlay from shadowing those cells: field 5 is tombstoned in place - // (preserving the overlay's field 3), and an overlay covering only field - // 5 is dropped entirely. - let mut fragment = Fragment::new(0); - fragment.files = vec![ - DataFile::new_legacy_from_fields("f3.lance", vec![3], None), - DataFile::new_legacy_from_fields("f5.lance", vec![5], None), - ]; - fragment.overlays = vec![ - DataOverlayFile { - data_file: DataFile::new_legacy_from_fields("o35.lance", vec![3, 5], None), - coverage: OverlayCoverage::sparse(vec![ - roaring::RoaringBitmap::from_iter([0u32]), - roaring::RoaringBitmap::from_iter([0u32]), - ]), - committed_version: 3, - }, - DataOverlayFile { - data_file: DataFile::new_legacy_from_fields("o5.lance", vec![5], None), - coverage: OverlayCoverage::dense(roaring::RoaringBitmap::from_iter([0u32])), - committed_version: 3, - }, - ]; - - let schema = ArrowSchema::new(vec![ArrowField::new("id", DataType::Int32, false)]); - let manifest = Manifest::new( - LanceSchema::try_from(&schema).unwrap(), - Arc::new(vec![fragment]), - lance_table::format::DataStorageFormat::new(LanceFileVersion::V2_0), - HashMap::new(), - ); - - let txn = Transaction::new( - manifest.version, - Operation::DataReplacement { - replacements: vec![DataReplacementGroup( - 0, - DataFile::new_legacy_from_fields("f5-new.lance", vec![5], None), - )], - }, - None, - ); - - let (result, _) = txn - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - let frag = &result.fragments[0]; - // The base data file for field 5 was swapped in. - assert!(frag.files.iter().any(|f| f.path == "f5-new.lance")); - // The [3, 5] overlay keeps field 3 and tombstones field 5; the [5]-only - // overlay is dropped. - assert_eq!(frag.overlays.len(), 1); - assert_eq!(frag.overlays[0].data_file.fields.as_ref(), &[3, -2]); - } - - #[test] - fn test_data_overlay_build_manifest_merges_duplicate_groups() { - // Two groups targeting the same fragment must both survive (a HashMap - // collapse would have dropped the first). - let manifest = sample_manifest(); - let txn = Transaction::new( - manifest.version, - Operation::DataOverlay { - groups: vec![ - DataOverlayGroup { - fragment_id: 0, - overlays: vec![overlay_with_field(1, 0)], - }, - DataOverlayGroup { - fragment_id: 0, - overlays: vec![overlay_with_field(2, 0)], - }, - ], - }, - None, - ); - - let (result, _) = txn - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap(); - - let overlays = &result.fragments[0].overlays; - assert_eq!(overlays.len(), 2); - assert_eq!(overlays[0].data_file.fields.as_ref(), [1i32].as_slice()); - assert_eq!(overlays[1].data_file.fields.as_ref(), [2i32].as_slice()); - } - - #[test] - fn test_data_overlay_build_manifest_rejects_unknown_fragment() { - let manifest = sample_manifest(); - let txn = Transaction::new( - manifest.version, - Operation::DataOverlay { - groups: vec![DataOverlayGroup { - fragment_id: 99, - overlays: vec![overlay_with_field(1, 0)], - }], - }, - None, - ); - let err = txn - .build_manifest( - Some(&manifest), - vec![], - "txn", - &ManifestWriteConfig::default(), - ) - .unwrap_err(); - assert!(err.to_string().contains("does not exist"), "{err}"); - } - - #[test] - fn test_data_overlay_operation_eq() { - let overlay = |field: i32| Operation::DataOverlay { - groups: vec![DataOverlayGroup { - fragment_id: 0, - overlays: vec![overlay_with_field(field, 1)], - }], - }; - // Reflexive and value-based (the arm previously returned false for self). - assert_eq!(overlay(1), overlay(1)); - assert_ne!(overlay(1), overlay(2)); - // Not equal to a different operation kind (previously returned true vs Rewrite). - let rewrite = Operation::Rewrite { - groups: vec![], - rewritten_indices: vec![], - frag_reuse_index: None, - }; - assert_ne!(overlay(1), rewrite); - } -} diff --git a/rust/lance/src/dataset/write/merge_insert/inserted_rows.rs b/rust/lance/src/dataset/write/merge_insert/inserted_rows.rs index 805073e75e2..2ac73148538 100644 --- a/rust/lance/src/dataset/write/merge_insert/inserted_rows.rs +++ b/rust/lance/src/dataset/write/merge_insert/inserted_rows.rs @@ -2,712 +2,8 @@ // SPDX-FileCopyrightText: Copyright The Lance Authors //! Key existence tracking for merge insert conflict detection. +//! +//! The implementation lives in [`lance_table::format::key_existence`] because the +//! filter is serialized into the transaction protobuf. -use std::collections::HashSet; -use std::collections::hash_map::DefaultHasher; -use std::hash::{Hash, Hasher}; - -use arrow_array::cast::AsArray; -use arrow_array::{ - Array, BinaryArray, LargeBinaryArray, LargeListArray, LargeStringArray, ListArray, RecordBatch, - StringArray, StructArray, -}; -use arrow_schema::DataType; -use lance_core::Result; -use lance_core::deepsize::DeepSizeOf; -use lance_core::utils::bloomfilter::sbbf::{Sbbf, SbbfBuilder}; -use lance_table::format::pb; - -// Default bloom filter config: 8192 items @ 0.00057 fpp -> 16KiB filter -pub const BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS: u64 = 8192; -pub const BLOOM_FILTER_DEFAULT_PROBABILITY: f64 = 0.00057; - -/// Key value for conflict detection. -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub enum KeyValue { - String(String), - Int64(i64), - UInt64(u64), - Binary(Vec), - List(Vec), - Struct(Vec), - Composite(Vec), -} - -impl KeyValue { - pub fn to_bytes(&self) -> Vec { - match self { - Self::String(s) => s.as_bytes().to_vec(), - Self::Int64(i) => i.to_le_bytes().to_vec(), - Self::UInt64(u) => u.to_le_bytes().to_vec(), - Self::Binary(b) => b.clone(), - Self::List(values) | Self::Struct(values) | Self::Composite(values) => { - let mut result = Vec::new(); - for value in values { - result.extend_from_slice(&value.to_bytes()); - result.push(0); - } - result - } - } - } - - pub fn hash_value(&self) -> u64 { - let mut hasher = DefaultHasher::new(); - self.to_bytes().hash(&mut hasher); - hasher.finish() - } -} - -/// Builder for KeyExistenceFilter using Split Block Bloom Filter. -#[derive(Debug, Clone)] -pub struct KeyExistenceFilterBuilder { - sbbf: Sbbf, - field_ids: Vec, - item_count: usize, -} - -impl KeyExistenceFilterBuilder { - pub fn new(field_ids: Vec) -> Self { - let sbbf = SbbfBuilder::new() - .expected_items(BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS) - .false_positive_probability(BLOOM_FILTER_DEFAULT_PROBABILITY) - .build() - .expect("Failed to build SBBF"); - Self { - sbbf, - field_ids, - item_count: 0, - } - } - - pub fn insert(&mut self, key: KeyValue) -> Result<()> { - self.sbbf.insert(&key.to_bytes()[..]); - self.item_count += 1; - Ok(()) - } - - pub fn contains(&self, key: &KeyValue) -> bool { - self.sbbf.check(&key.to_bytes()[..]) - } - - pub fn might_intersect(&self, other: &Self) -> Result { - self.sbbf - .might_intersect(&other.sbbf) - .map_err(|e| lance_core::Error::invalid_input(e.to_string())) - } - - pub fn field_ids(&self) -> &[i32] { - &self.field_ids - } - - pub fn estimated_size_bytes(&self) -> usize { - self.sbbf.size_bytes() - } - - pub fn len(&self) -> usize { - self.item_count - } - - pub fn is_empty(&self) -> bool { - self.item_count == 0 - } - - pub fn build(&self) -> KeyExistenceFilter { - KeyExistenceFilter { - field_ids: self.field_ids.clone(), - filter: FilterType::Bloom { - bitmap: self.sbbf.to_bytes(), - num_bits: (self.sbbf.size_bytes() as u32) * 8, - number_of_items: BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS, - probability: BLOOM_FILTER_DEFAULT_PROBABILITY, - }, - } - } -} - -impl From<&KeyExistenceFilterBuilder> for pb::transaction::KeyExistenceFilter { - fn from(builder: &KeyExistenceFilterBuilder) -> Self { - Self { - field_ids: builder.field_ids.clone(), - data: Some(pb::transaction::key_existence_filter::Data::Bloom( - pb::transaction::BloomFilter { - bitmap: builder.sbbf.to_bytes(), - num_bits: (builder.sbbf.size_bytes() as u32) * 8, - number_of_items: BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS, - probability: BLOOM_FILTER_DEFAULT_PROBABILITY, - }, - )), - } - } -} - -/// Filter type for key existence data. -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub enum FilterType { - ExactSet(HashSet), - Bloom { - bitmap: Vec, - num_bits: u32, - number_of_items: u64, - probability: f64, - }, -} - -/// Tracks keys of inserted rows for conflict detection. -/// Only created when ON columns match the schema's unenforced primary key. -#[derive(Debug, Clone, DeepSizeOf, PartialEq)] -pub struct KeyExistenceFilter { - pub field_ids: Vec, - pub filter: FilterType, -} - -impl KeyExistenceFilter { - pub fn from_bloom_filter(bloom: &KeyExistenceFilterBuilder) -> Self { - bloom.build() - } - - /// Check if two filters intersect. Returns (has_intersection, might_be_false_positive). - /// Errors if bloom filter configs don't match. - pub fn intersects(&self, other: &Self) -> Result<(bool, bool)> { - match (&self.filter, &other.filter) { - (FilterType::ExactSet(a), FilterType::ExactSet(b)) => { - Ok((a.iter().any(|h| b.contains(h)), false)) - } - (FilterType::ExactSet(_), FilterType::Bloom { .. }) - | (FilterType::Bloom { .. }, FilterType::ExactSet(_)) => { - // Can't compare different hash schemes, assume intersection - Ok((true, true)) - } - ( - FilterType::Bloom { - bitmap: a_bits, - number_of_items: a_num_items, - probability: a_prob, - .. - }, - FilterType::Bloom { - bitmap: b_bits, - number_of_items: b_num_items, - probability: b_prob, - .. - }, - ) => { - if a_num_items != b_num_items || (a_prob - b_prob).abs() > f64::EPSILON { - return Err(lance_core::Error::invalid_input(format!( - "Bloom filter config mismatch: ({}, {}) vs ({}, {})", - a_num_items, a_prob, b_num_items, b_prob - ))); - } - let has = Sbbf::bytes_might_intersect(a_bits, b_bits) - .map_err(|e| lance_core::Error::invalid_input(e.to_string()))?; - Ok((has, has)) - } - } - } -} - -impl From<&KeyExistenceFilter> for pb::transaction::KeyExistenceFilter { - fn from(filter: &KeyExistenceFilter) -> Self { - match &filter.filter { - FilterType::ExactSet(hashes) => Self { - field_ids: filter.field_ids.clone(), - data: Some(pb::transaction::key_existence_filter::Data::Exact( - pb::transaction::ExactKeySetFilter { - key_hashes: hashes.iter().copied().collect(), - }, - )), - }, - FilterType::Bloom { - bitmap, - num_bits, - number_of_items, - probability, - } => Self { - field_ids: filter.field_ids.clone(), - data: Some(pb::transaction::key_existence_filter::Data::Bloom( - pb::transaction::BloomFilter { - bitmap: bitmap.clone(), - num_bits: *num_bits, - number_of_items: *number_of_items, - probability: *probability, - }, - )), - }, - } - } -} - -impl TryFrom<&pb::transaction::KeyExistenceFilter> for KeyExistenceFilter { - type Error = lance_core::Error; - - fn try_from(message: &pb::transaction::KeyExistenceFilter) -> Result { - let filter = match message.data.as_ref() { - Some(pb::transaction::key_existence_filter::Data::Exact(exact)) => { - FilterType::ExactSet(exact.key_hashes.iter().copied().collect()) - } - Some(pb::transaction::key_existence_filter::Data::Bloom(b)) => { - // Use defaults for backwards compatibility - let number_of_items = if b.number_of_items == 0 { - BLOOM_FILTER_DEFAULT_NUMBER_OF_ITEMS - } else { - b.number_of_items - }; - let probability = if b.probability == 0.0 { - BLOOM_FILTER_DEFAULT_PROBABILITY - } else { - b.probability - }; - FilterType::Bloom { - bitmap: b.bitmap.clone(), - num_bits: b.num_bits, - number_of_items, - probability, - } - } - None => FilterType::ExactSet(HashSet::new()), - }; - Ok(Self { - field_ids: message.field_ids.clone(), - filter, - }) - } -} - -/// Extract key value from a batch row. Returns None if null or unsupported type. -pub fn extract_key_value_from_batch( - batch: &RecordBatch, - row_idx: usize, - on_columns: &[String], -) -> Option { - let mut parts: Vec = Vec::with_capacity(on_columns.len()); - - for col_name in on_columns { - let (col_idx, _) = batch.schema().column_with_name(col_name)?; - let column = batch.column(col_idx); - - if column.is_null(row_idx) { - return None; - } - - let key_part = extract_key_value(column, row_idx)?; - parts.push(key_part); - } - - if parts.is_empty() { - None - } else if parts.len() == 1 { - Some(parts.into_iter().next().unwrap()) - } else { - Some(KeyValue::Composite(parts)) - } -} - -fn extract_key_value(array: &dyn Array, row_idx: usize) -> Option { - let v = match array.data_type() { - DataType::Utf8 => { - let arr = array.as_any().downcast_ref::()?; - KeyValue::String(arr.value(row_idx).to_string()) - } - DataType::LargeUtf8 => { - let arr = array.as_any().downcast_ref::()?; - KeyValue::String(arr.value(row_idx).to_string()) - } - DataType::UInt64 => { - let arr = array.as_primitive::(); - KeyValue::UInt64(arr.value(row_idx)) - } - DataType::Int64 => { - let arr = array.as_primitive::(); - KeyValue::Int64(arr.value(row_idx)) - } - DataType::UInt32 => { - let arr = array.as_primitive::(); - KeyValue::UInt64(arr.value(row_idx) as u64) - } - DataType::Int32 => { - let arr = array.as_primitive::(); - KeyValue::Int64(arr.value(row_idx) as i64) - } - DataType::Binary => { - let arr = array.as_any().downcast_ref::()?; - KeyValue::Binary(arr.value(row_idx).to_vec()) - } - DataType::LargeBinary => { - let arr = array.as_any().downcast_ref::()?; - KeyValue::Binary(arr.value(row_idx).to_vec()) - } - DataType::List(_) => { - let list_array = array.as_any().downcast_ref::().unwrap(); - let values = list_array.value(row_idx); - - let mut elements = Vec::with_capacity(values.len()); - for i in 0..values.len() { - if values.is_null(i) { - return None; - } - let element = extract_key_value(&values, i)?; - elements.push(element); - } - KeyValue::List(elements) - } - DataType::LargeList(_) => { - let list_array = array.as_any().downcast_ref::().unwrap(); - let values = list_array.value(row_idx); - - let mut elements = Vec::with_capacity(values.len()); - for i in 0..values.len() { - if values.is_null(i) { - return None; - } - let element = extract_key_value(&values, i)?; - elements.push(element); - } - KeyValue::List(elements) - } - DataType::Struct(_) => { - let struct_array = array.as_any().downcast_ref::()?; - let mut elements = Vec::with_capacity(struct_array.num_columns()); - for i in 0..struct_array.num_columns() { - let child = struct_array.column(i); - if child.is_null(row_idx) { - return None; - } - let field_value = extract_key_value(child.as_ref(), row_idx)?; - elements.push(field_value); - } - KeyValue::Struct(elements) - } - _ => return None, - }; - Some(v) -} - -#[cfg(test)] -mod tests { - use super::*; - use std::sync::Arc; - - use arrow_array::builder::{Int32Builder, ListBuilder, StringBuilder}; - use arrow_array::{Int32Array, RecordBatch, StringArray, StructArray}; - use arrow_schema::{Field, Schema}; - - #[test] - fn test_extract_key_value_from_batch_list_int() { - let values_builder = Int32Builder::new(); - let mut list_builder = ListBuilder::new(values_builder); - - list_builder.append_value([Some(1), Some(2)]); - list_builder.append_value([Some(3), Some(4), Some(5)]); - - let list_array = list_builder.finish(); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - list_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) - .expect("batch should be valid"); - - let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("first row should produce a key"); - let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) - .expect("second row should produce a key"); - - match &key0 { - KeyValue::List(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::Int64(1)); - assert_eq!(values[1], KeyValue::Int64(2)); - } - other => panic!("expected list key, got {:?}", other), - } - - match &key1 { - KeyValue::List(values) => { - assert_eq!(values.len(), 3); - assert_eq!(values[0], KeyValue::Int64(3)); - assert_eq!(values[1], KeyValue::Int64(4)); - assert_eq!(values[2], KeyValue::Int64(5)); - } - other => panic!("expected list key, got {:?}", other), - } - - assert_ne!( - key0.hash_value(), - key1.hash_value(), - "different list values should hash differently", - ); - } - - #[test] - fn test_extract_key_value_from_batch_empty_list() { - let values_builder = Int32Builder::new(); - let mut list_builder = ListBuilder::new(values_builder); - - list_builder.append_value(std::iter::empty::>()); - - let list_array = list_builder.finish(); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - list_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) - .expect("batch should be valid"); - - let key = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("empty list should still produce a key"); - - match key { - KeyValue::List(values) => { - assert!(values.is_empty(), "expected empty list"); - } - other => panic!("expected list key, got {:?}", other), - } - } - - #[test] - fn test_extract_key_value_from_batch_list_utf8() { - let values_builder = StringBuilder::new(); - let mut list_builder = ListBuilder::new(values_builder); - - list_builder.append_value([Some("a"), Some("bc")]); - list_builder.append_value([Some("de")]); - - let list_array = list_builder.finish(); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - list_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) - .expect("batch should be valid"); - - let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("first row should produce a key"); - let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) - .expect("second row should produce a key"); - - match &key0 { - KeyValue::List(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::String("a".to_string())); - assert_eq!(values[1], KeyValue::String("bc".to_string())); - } - other => panic!("expected list key, got {:?}", other), - } - - match &key1 { - KeyValue::List(values) => { - assert_eq!(values.len(), 1); - assert_eq!(values[0], KeyValue::String("de".to_string())); - } - other => panic!("expected list key, got {:?}", other), - } - - assert_ne!( - key0.hash_value(), - key1.hash_value(), - "different list values should hash differently", - ); - } - - #[test] - fn test_extract_key_value_from_batch_list_with_null_child() { - let values_builder = Int32Builder::new(); - let mut list_builder = ListBuilder::new(values_builder); - - list_builder.append_value([Some(1), Some(2)]); - list_builder.append_value([Some(3), None]); - - let list_array = list_builder.finish(); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - list_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(list_array)]) - .expect("batch should be valid"); - - let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("first row should produce a key"); - let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]); - - match &key0 { - KeyValue::List(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::Int64(1)); - assert_eq!(values[1], KeyValue::Int64(2)); - } - other => panic!("expected list key, got {:?}", other), - } - - assert!( - key1.is_none(), - "list row with a null child should not produce a key", - ); - } - - #[test] - fn test_extract_key_value_from_batch_struct_int() { - let a_values = Int32Array::from(vec![1, 3]); - let b_values = Int32Array::from(vec![2, 4]); - - let struct_array = StructArray::from(vec![ - ( - Arc::new(Field::new("a", arrow_schema::DataType::Int32, false)), - Arc::new(a_values) as Arc, - ), - ( - Arc::new(Field::new("b", arrow_schema::DataType::Int32, false)), - Arc::new(b_values) as Arc, - ), - ]); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - struct_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(struct_array)]) - .expect("batch should be valid"); - - let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("first row should produce a key"); - let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) - .expect("second row should produce a key"); - - match &key0 { - KeyValue::Struct(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::Int64(1)); - assert_eq!(values[1], KeyValue::Int64(2)); - } - other => panic!("expected struct key, got {:?}", other), - } - - match &key1 { - KeyValue::Struct(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::Int64(3)); - assert_eq!(values[1], KeyValue::Int64(4)); - } - other => panic!("expected struct key, got {:?}", other), - } - - assert_ne!( - key0.hash_value(), - key1.hash_value(), - "different struct values should hash differently", - ); - } - - #[test] - fn test_extract_key_value_from_batch_struct_utf8() { - let first_names = StringArray::from(vec!["alice", "bob"]); - let last_names = StringArray::from(vec!["smith", "jones"]); - - let struct_array = StructArray::from(vec![ - ( - Arc::new(Field::new("first", arrow_schema::DataType::Utf8, false)), - Arc::new(first_names) as Arc, - ), - ( - Arc::new(Field::new("last", arrow_schema::DataType::Utf8, false)), - Arc::new(last_names) as Arc, - ), - ]); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - struct_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(struct_array)]) - .expect("batch should be valid"); - - let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("first row should produce a key"); - let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]) - .expect("second row should produce a key"); - - match &key0 { - KeyValue::Struct(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::String("alice".to_string())); - assert_eq!(values[1], KeyValue::String("smith".to_string())); - } - other => panic!("expected struct key, got {:?}", other), - } - - match &key1 { - KeyValue::Struct(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::String("bob".to_string())); - assert_eq!(values[1], KeyValue::String("jones".to_string())); - } - other => panic!("expected struct key, got {:?}", other), - } - - assert_ne!( - key0.hash_value(), - key1.hash_value(), - "different struct values should hash differently", - ); - } - - #[test] - fn test_extract_key_value_from_batch_struct_with_null_child() { - let a_values = Int32Array::from(vec![Some(1), None]); - let b_values = Int32Array::from(vec![Some(2), Some(3)]); - - let struct_array = StructArray::from(vec![ - ( - Arc::new(Field::new("a", arrow_schema::DataType::Int32, true)), - Arc::new(a_values) as Arc, - ), - ( - Arc::new(Field::new("b", arrow_schema::DataType::Int32, true)), - Arc::new(b_values) as Arc, - ), - ]); - - let schema = Arc::new(Schema::new(vec![Field::new( - "id", - struct_array.data_type().clone(), - false, - )])); - - let batch = RecordBatch::try_new(schema, vec![Arc::new(struct_array)]) - .expect("batch should be valid"); - - let key0 = extract_key_value_from_batch(&batch, 0, &[String::from("id")]) - .expect("first row should produce a key"); - let key1 = extract_key_value_from_batch(&batch, 1, &[String::from("id")]); - - match &key0 { - KeyValue::Struct(values) => { - assert_eq!(values.len(), 2); - assert_eq!(values[0], KeyValue::Int64(1)); - assert_eq!(values[1], KeyValue::Int64(2)); - } - other => panic!("expected struct key, got {:?}", other), - } - - assert!( - key1.is_none(), - "struct row with a null child should not produce a key", - ); - } -} +pub use lance_table::format::key_existence::*; diff --git a/rust/lance/src/index/mem_wal.rs b/rust/lance/src/index/mem_wal.rs index 699338085b4..9f60e36a04f 100644 --- a/rust/lance/src/index/mem_wal.rs +++ b/rust/lance/src/index/mem_wal.rs @@ -3,125 +3,22 @@ //! MemWAL Index operations. //! -//! The MemWAL Index stores: -//! - Configuration (sharding_specs, maintained_indexes) -//! - SSTable compaction progress -//! - Shard state snapshots (eventually consistent) -//! -//! Writers no longer update the index on every write. Instead, they update -//! shard manifests directly. This module provides functions to: -//! - Load the MemWAL index -//! - Update compacted SSTables (called during merge-insert commits) - -use std::sync::Arc; - -use lance_core::{Error, Result}; -use lance_index::mem_wal::{CompactedSsTable, MEM_WAL_INDEX_NAME, MemWalIndex, MemWalIndexDetails}; -use lance_table::format::{IndexMetadata, pb}; -use uuid::Uuid; - -/// Load MemWalIndexDetails from an IndexMetadata. -pub(crate) fn load_mem_wal_index_details(index: IndexMetadata) -> Result { - if let Some(details_any) = index.index_details.as_ref() { - if !details_any.type_url.ends_with("MemWalIndexDetails") { - return Err(Error::index(format!( - "Index details is not for the MemWAL index, but {}", - details_any.type_url - ))); - } - - Ok(MemWalIndexDetails::try_from( - details_any.to_msg::()?, - )?) - } else { - Err(Error::index("Index details not found for the MemWAL index")) - } -} - -/// Open the MemWAL index from its metadata. -pub(crate) fn open_mem_wal_index(index: IndexMetadata) -> Result> { - Ok(Arc::new(MemWalIndex::new(load_mem_wal_index_details( - index, - )?))) -} - -/// Update `compacted_sstables` in the MemWAL index. -/// -/// This is called during merge-insert commits to atomically record which -/// SSTables have been compacted into the base table. -pub(crate) fn update_mem_wal_index_compacted_sstables( - indices: &mut Vec, - dataset_version: u64, - new_compacted_sstables: Vec, -) -> Result<()> { - if new_compacted_sstables.is_empty() { - return Ok(()); - } +//! The index data structures and the helpers that read and update the index's +//! `IndexMetadata` entry live in [`lance_table::system_index::mem_wal`]; this +//! module holds the dataset-level operations built on top of them. - let pos = indices - .iter() - .position(|idx| idx.name == MEM_WAL_INDEX_NAME); - - let new_meta = if let Some(pos) = pos { - let current_meta = indices.remove(pos); - let mut details = load_mem_wal_index_details(current_meta)?; - - // Update compacted_sstables - for each shard, keep the higher generation - for new_sstable in new_compacted_sstables { - if let Some(existing) = details - .compacted_sstables - .iter_mut() - .find(|sstable| sstable.shard_id == new_sstable.shard_id) - { - if new_sstable.generation > existing.generation { - existing.generation = new_sstable.generation; - } - } else { - details.compacted_sstables.push(new_sstable); - } - } - - new_mem_wal_index_meta(dataset_version, details)? - } else { - // Create a MemWAL index containing only compaction progress. - let details = MemWalIndexDetails { - compacted_sstables: new_compacted_sstables, - ..Default::default() - }; - new_mem_wal_index_meta(dataset_version, details)? - }; - - indices.push(new_meta); - Ok(()) -} - -/// Create a new MemWAL index metadata entry. -pub(crate) fn new_mem_wal_index_meta( - dataset_version: u64, - details: MemWalIndexDetails, -) -> Result { - Ok(IndexMetadata { - uuid: Uuid::new_v4(), - name: MEM_WAL_INDEX_NAME.to_string(), - fields: vec![], - dataset_version, - fragment_bitmap: None, - index_details: Some(Arc::new(prost_types::Any::from_msg( - &pb::MemWalIndexDetails::from(&details), - )?)), - index_version: 0, - created_at: Some(chrono::Utc::now()), - base_id: None, - // Memory WAL index is inline (no files) - files: None, - }) -} +pub(crate) use lance_table::system_index::mem_wal::{ + load_mem_wal_index_details, new_mem_wal_index_meta, open_mem_wal_index, +}; #[cfg(test)] mod tests { use super::*; + use lance_index::mem_wal::{CompactedSsTable, MEM_WAL_INDEX_NAME, MemWalIndexDetails}; + use lance_table::system_index::mem_wal::update_mem_wal_index_compacted_sstables; use std::sync::Arc; + use uuid::Uuid; use crate::index::DatasetIndexExt; use arrow_array::{Int32Array, RecordBatch}; diff --git a/rust/lance/src/io/commit.rs b/rust/lance/src/io/commit.rs index 2441e31749d..a024ffb75fa 100644 --- a/rust/lance/src/io/commit.rs +++ b/rust/lance/src/io/commit.rs @@ -250,8 +250,12 @@ async fn do_commit_new_dataset( (new_manifest, updated_indices) } } else { - let (manifest, indices) = - transaction.build_manifest(None, vec![], &transaction_file, write_config)?; + let (manifest, indices) = transaction.build_manifest( + None, + vec![], + &transaction_file, + &write_config.to_build_config(), + )?; (manifest, indices) }; @@ -811,7 +815,7 @@ pub(crate) async fn do_commit_detached_transaction( commit_handler, &dataset.base, version, - write_config, + &write_config.to_build_config(), &transaction_file, &dataset.manifest, ) @@ -821,7 +825,7 @@ pub(crate) async fn do_commit_detached_transaction( Some(dataset.manifest.as_ref()), dataset.load_indices().await?.as_ref().clone(), &transaction_file, - write_config, + &write_config.to_build_config(), )?, }; @@ -1009,7 +1013,7 @@ pub(crate) async fn commit_transaction( commit_handler, &dataset.base, version, - write_config, + &write_config.to_build_config(), transaction_file, &dataset.manifest, ) @@ -1019,7 +1023,7 @@ pub(crate) async fn commit_transaction( Some(dataset.manifest.as_ref()), dataset.load_indices().await?.as_ref().clone(), transaction_file, - write_config, + &write_config.to_build_config(), )?, }; diff --git a/rust/lance/src/io/commit/conflict_resolver.rs b/rust/lance/src/io/commit/conflict_resolver.rs index daed9798816..7bbc8a84003 100644 --- a/rust/lance/src/io/commit/conflict_resolver.rs +++ b/rust/lance/src/io/commit/conflict_resolver.rs @@ -16,6 +16,7 @@ use lance_index::mem_wal::{CompactedSsTable, MEM_WAL_INDEX_NAME}; use lance_select::{RowAddrTreeMap, RowSetOps}; use lance_table::format::IndexMetadata; use lance_table::format::overlay::OverlayCoverage; +use lance_table::transaction::{Conflict, Region}; use lance_table::{format::Fragment, io::deletion::write_deletion_file}; use roaring::RoaringBitmap; use std::{ @@ -24,6 +25,17 @@ use std::{ sync::Arc, }; +/// Regions whose *legacy* rebase needs side effects derived from the concurrent +/// operation's contents, which no action family supplies yet. +/// +/// While a region is listed here, a concurrent action-based operation touching it +/// is answered with a retry instead of being compared by footprint. A slice that +/// wants its region removed has to supply those side effects first: fragments +/// drive deletion-file rewrites and frag-reuse index collection, indices drive +/// index invalidation, and generations drive compacted-SSTable collection. +const REGIONS_WITH_REBASE_SIDE_EFFECTS: [Region; 3] = + [Region::Fragments, Region::Indices, Region::Generations]; + #[derive(Debug)] pub struct TransactionRebase<'a> { transaction: Transaction, @@ -56,7 +68,8 @@ impl<'a> TransactionRebase<'a> { | Operation::UpdateMemWalState { .. } | Operation::Clone { .. } | Operation::Restore { .. } - | Operation::UpdateBases { .. } => Ok(Self { + | Operation::UpdateBases { .. } + | Operation::UserOperation(..) => Ok(Self { transaction, affected_rows, initial_fragments: HashMap::new(), @@ -221,6 +234,46 @@ impl<'a> TransactionRebase<'a> { /// Will return an error if the transaction is not valid. Otherwise, it will /// return Ok(()). pub fn check_txn(&mut self, other_transaction: &Transaction, other_version: u64) -> Result<()> { + // A V2 `other` is decided by footprint intersection, before any legacy + // dispatch. This ordering is load-bearing: several `check_*_txn` bodies do + // not merely decide, they mutate rebase state derived from the other + // operation's contents (`check_delete_txn` marks fragments needing a + // deletion-file rewrite, `check_create_index_txn` collects conflicting + // frag-reuse indices, `check_update_mem_wal_state_txn` collects compacted + // SSTables). Most end in a permissive arm, so a `UserOperation` reaching + // them would be silently allowed while the side effect it should have + // triggered was never computed. + if matches!(&other_transaction.operation, Operation::UserOperation(..)) { + let other_mask = other_transaction.operation.writes(); + // Until an action family supplies those rebase side effects, an `other` + // that touches their regions is answered with a retry. A V2 operation + // confined to `Bases` provably needs none — that is what makes AddBase a + // safe first slice, and the gate every later slice must clear before + // removing its region from this set. The failure mode is a spurious retry. + // + // The gate also applies when *self* is action-based, where no legacy + // rebase is involved and it is therefore stricter than it needs to be. + // Keeping one rule is worth the occasional extra retry until a region + // leaves the set. + if other_mask.intersects_any(®IONS_WITH_REBASE_SIDE_EFFECTS) { + return Err(self.retryable_conflict_err(other_transaction, other_version)); + } + return match self + .transaction + .operation + .writes() + .conflicts_with(&other_mask) + { + None => Ok(()), + Some(Conflict::Retryable) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } + Some(Conflict::Incompatible) => { + Err(self.incompatible_conflict_err(other_transaction, other_version)) + } + }; + } + let op = &self.transaction.operation; match op { Operation::Delete { .. } => self.check_delete_txn(other_transaction, other_version), @@ -255,6 +308,25 @@ impl<'a> TransactionRebase<'a> { Operation::UpdateBases { .. } => { self.check_add_bases_txn(other_transaction, other_version) } + // Legacy `other`, V2 self: the same footprint intersection, without the + // frontier check. The regions in that set are ones whose *legacy rebase* + // needs side effects derived from the other operation; here it is this + // side that is action-based, and an action re-applies from scratch on + // retry rather than carrying rebase state forward. + Operation::UserOperation(..) => { + match op + .writes() + .conflicts_with(&other_transaction.operation.writes()) + { + None => Ok(()), + Some(Conflict::Retryable) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } + Some(Conflict::Incompatible) => { + Err(self.incompatible_conflict_err(other_transaction, other_version)) + } + } + } } } @@ -265,6 +337,12 @@ impl<'a> TransactionRebase<'a> { ) -> Result<()> { if let Operation::Delete { .. } = &self.transaction.operation { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::CreateIndex { .. } | Operation::ReserveFragments { .. } | Operation::Clone { .. } @@ -419,6 +497,12 @@ impl<'a> TransactionRebase<'a> { } match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::CreateIndex { .. } | Operation::ReserveFragments { .. } | Operation::Project { .. } @@ -581,6 +665,12 @@ impl<'a> TransactionRebase<'a> { } = &self.transaction.operation { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Append { .. } | Operation::Clone { .. } // An overlay committed after this index's version is newer than @@ -759,6 +849,12 @@ impl<'a> TransactionRebase<'a> { } = &self.transaction.operation { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } // Rewrite is only compatible with operations that don't touch // existing fragments or update fragments we don't touch. Operation::Append { .. } @@ -949,6 +1045,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Overwrite { .. } => { if self .transaction @@ -998,6 +1100,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } // Append is not compatible with any operation that completely // overwrites the schema. Operation::Overwrite { .. } @@ -1028,6 +1136,12 @@ impl<'a> TransactionRebase<'a> { ) -> Result<()> { if let Operation::DataReplacement { replacements } = &self.transaction.operation { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Append { .. } | Operation::Clone { .. } | Operation::UpdateConfig { .. } @@ -1189,6 +1303,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Append { .. } | Operation::CreateIndex { .. } | Operation::ReserveFragments { .. } @@ -1287,6 +1407,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::CreateIndex { .. } | Operation::ReserveFragments { .. } | Operation::Clone { .. } @@ -1317,6 +1443,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Append { .. } | Operation::Delete { .. } | Operation::Overwrite { .. } @@ -1344,6 +1476,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Overwrite { .. } | Operation::Restore { .. } => { Err(self.incompatible_conflict_err(other_transaction, other_version)) } @@ -1370,6 +1508,12 @@ impl<'a> TransactionRebase<'a> { other_version: u64, ) -> Result<()> { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } // Project is compatible with anything that doesn't change the schema Operation::Append { .. } | Operation::Update { .. } @@ -1406,6 +1550,12 @@ impl<'a> TransactionRebase<'a> { } = &self.transaction.operation { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::Overwrite { .. } => { // Updates to schema metadata or field metadata conflict with any kind // of overwrite. @@ -1466,6 +1616,12 @@ impl<'a> TransactionRebase<'a> { } = &self.transaction.operation { match &other_transaction.operation { + // Unreachable: check_txn answers a V2 `other` by footprint before + // dispatching here. Fail safe rather than fall through to an arm + // that would silently permit it. + Operation::UserOperation(..) => { + Err(self.retryable_conflict_err(other_transaction, other_version)) + } Operation::UpdateMemWalState { compacted_sstables: other_compacted_sstables, } => { @@ -1618,7 +1774,11 @@ impl<'a> TransactionRebase<'a> { | Operation::Clone { .. } | Operation::UpdateConfig { .. } | Operation::UpdateMemWalState { .. } - | Operation::UpdateBases { .. } => Ok(self.transaction), + | Operation::UpdateBases { .. } + // AddBase needs no rebase mutation: `build_manifest` re-mints the base + // id from the current manifest on every retry, so the id simply comes + // out right. That is the property the minting design buys. + | Operation::UserOperation(..) => Ok(self.transaction), } } @@ -3094,6 +3254,23 @@ mod tests { }; for (other, expected_conflict) in other_transactions.iter().zip(expected_conflicts) { + // This table doubles as the oracle for `Operation::writes()`, which + // restates it as a footprint. Over-declaring is safe — it costs a + // spurious retry — so the assertion runs one way only: if two + // footprints are disjoint, the table must have said `Compatible`. + // An under-declared mask, the unsafe direction, fails here. + if operation + .writes() + .conflicts_with(&other.operation.writes()) + .is_none() + { + assert_eq!( + expected_conflict, &Compatible, + "the footprints of {:?} and {:?} are disjoint, but the table expects {:?}", + operation, other.operation, expected_conflict + ); + } + match expected_conflict { Compatible => { let result = rebase.check_txn(other, 1); @@ -3949,6 +4126,233 @@ mod tests { } } + /// An action-based operation that mints one base. + fn add_base_operation(name: Option<&str>, path: &str) -> Operation { + use lance_table::transaction::{Action, AddBase, UserAction, UserOperation}; + + Operation::UserOperation(UserOperation { + description: format!("ALTER TABLE t ADD BASE {path}"), + uuid: uuid::Uuid::new_v4().to_string(), + read_version: 1, + actions: vec![UserAction { + description: "register base".to_string(), + actions: vec![Action::AddBase(AddBase { + local: 0, + name: name.map(str::to_string), + is_dataset_root: false, + path: path.to_string(), + })], + }], + }) + } + + async fn check_against(dataset: &Dataset, ours: Operation, other: Operation) -> Result<()> { + let mut rebase = + TransactionRebase::try_new(dataset, Transaction::new_from_version(1, ours), None) + .await + .unwrap(); + rebase.check_txn(&Transaction::new_from_version(1, other), 2) + } + + #[tokio::test] + async fn test_user_operation_add_base_commits() { + let dataset = test_dataset(10, 2).await; + let version = dataset.version().version; + + let dataset = CommitBuilder::new(Arc::new(dataset)) + .execute(Transaction::new_from_version( + version, + add_base_operation(Some("warm"), "s3://bucket/warm"), + )) + .await + .unwrap(); + + let bases = &dataset.manifest().base_paths; + assert_eq!(bases.len(), 1); + let (id, base) = bases.iter().next().unwrap(); + assert_eq!(*id, 1); + assert_eq!(base.name.as_deref(), Some("warm")); + assert_eq!(base.path, "s3://bucket/warm"); + } + + #[tokio::test] + async fn test_concurrent_add_bases_with_distinct_names_get_distinct_ids() { + // Both transactions are planned against the same version, so the second + // commit rebases over the first. No rebase mutation is needed: the id is + // re-minted from the current manifest, which is the property minting buys. + let dataset = Arc::new(test_dataset(10, 2).await); + let version = dataset.version().version; + + let first = CommitBuilder::new(dataset.clone()) + .execute(Transaction::new_from_version( + version, + add_base_operation(Some("warm"), "s3://bucket/warm"), + )) + .await + .unwrap(); + + let second = CommitBuilder::new(Arc::new(first)) + .execute(Transaction::new_from_version( + version, + add_base_operation(Some("cold"), "s3://bucket/cold"), + )) + .await + .unwrap(); + + let bases = &second.manifest().base_paths; + assert_eq!(bases.len(), 2, "both bases should survive: {bases:?}"); + let mut named: Vec<_> = bases + .iter() + .map(|(id, base)| (*id, base.name.clone().unwrap())) + .collect(); + named.sort(); + assert_eq!( + named, + vec![(1, "warm".to_string()), (2, "cold".to_string())], + "ids must be distinct and minted in commit order" + ); + } + + #[tokio::test] + async fn test_add_base_conflicts_are_incompatible_on_name_and_path() { + let dataset = test_dataset(10, 2).await; + + for (ours, other, what) in [ + ( + add_base_operation(Some("warm"), "s3://bucket/a"), + add_base_operation(Some("warm"), "s3://bucket/b"), + "name", + ), + ( + add_base_operation(Some("a"), "s3://bucket/warm"), + add_base_operation(Some("b"), "s3://bucket/warm"), + "path", + ), + ] { + let result = check_against(&dataset, ours, other).await; + assert!( + matches!(result, Err(Error::IncompatibleTransaction { .. })), + "a taken base {what} is not freed by a retry, so it must be \ + incompatible; got {result:?}" + ); + } + } + + #[tokio::test] + async fn test_add_base_commutes_with_distinct_add_base_and_with_append() { + let dataset = test_dataset(10, 2).await; + + let commuting = [ + add_base_operation(Some("cold"), "s3://bucket/cold"), + Operation::Append { fragments: vec![] }, + ]; + for other in commuting { + check_against( + &dataset, + add_base_operation(Some("warm"), "s3://bucket/warm"), + other.clone(), + ) + .await + .unwrap_or_else(|e| panic!("AddBase should commute with {other:?}: {e:?}")); + + // And the same in the other direction: a legacy self against a V2 other + // takes the frontier path in `check_txn`. + check_against( + &dataset, + other.clone(), + add_base_operation(Some("warm"), "s3://bucket/warm"), + ) + .await + .unwrap_or_else(|e| panic!("{other:?} should commute with AddBase: {e:?}")); + } + } + + #[tokio::test] + async fn test_add_base_conflicts_with_restore_retryably() { + // Restore replaces the manifest wholesale and does not merge base_paths + // forward, so re-adding the base on top of the restored state is the correct + // resolution. (Legacy `check_add_bases_txn` returns Ok here and silently + // drops the base; that pre-existing gap is untouched.) + let dataset = test_dataset(10, 2).await; + + for (ours, other) in [ + ( + add_base_operation(Some("warm"), "s3://bucket/warm"), + Operation::Restore { version: 1 }, + ), + ( + Operation::Restore { version: 1 }, + add_base_operation(Some("warm"), "s3://bucket/warm"), + ), + ] { + let result = check_against(&dataset, ours, other).await; + assert!( + matches!(result, Err(Error::RetryableCommitConflict { .. })), + "expected a retryable conflict, got {result:?}" + ); + } + } + + #[tokio::test] + async fn test_add_base_conflicts_with_legacy_update_bases_across_vocabularies() { + let dataset = test_dataset(10, 2).await; + let legacy = |name: &str, path: &str| Operation::UpdateBases { + new_bases: vec![lance_table::format::BasePath { + id: 0, + path: path.to_string(), + name: Some(name.to_string()), + is_dataset_root: false, + }], + }; + + // Same name across the two vocabularies, in both directions. + for (ours, other) in [ + ( + add_base_operation(Some("warm"), "s3://bucket/a"), + legacy("warm", "s3://bucket/b"), + ), + ( + legacy("warm", "s3://bucket/b"), + add_base_operation(Some("warm"), "s3://bucket/a"), + ), + ] { + let result = check_against(&dataset, ours, other).await; + assert!( + matches!(result, Err(Error::IncompatibleTransaction { .. })), + "expected an incompatible conflict, got {result:?}" + ); + } + + // Distinct names commute across vocabularies too. + check_against( + &dataset, + add_base_operation(Some("warm"), "s3://bucket/a"), + legacy("cold", "s3://bucket/b"), + ) + .await + .unwrap(); + } + + #[tokio::test] + async fn test_legacy_self_against_fragment_touching_other_is_unaffected() { + // The frontier rule keys on the *other* operation's regions. A V2 other + // confined to Bases needs none of the legacy rebase side effects, so a + // fragment-touching legacy self is compared by footprint and commutes. + let dataset = test_dataset(10, 2).await; + + check_against( + &dataset, + Operation::Delete { + updated_fragments: vec![], + deleted_fragment_ids: vec![0], + predicate: "a > 5".to_string(), + }, + add_base_operation(Some("warm"), "s3://bucket/warm"), + ) + .await + .unwrap(); + } + #[tokio::test] async fn test_add_bases_multiple_bases() { let dataset = test_dataset(10, 2).await; @@ -4087,6 +4491,9 @@ mod tests { | Operation::UpdateBases { .. } | Operation::Restore { .. } | Operation::UpdateMemWalState { .. } => Box::new(std::iter::empty()), + // TODO: the only action today is AddBase, which touches no fragments. + // A fragment-level action must derive the ids from the action list here. + Operation::UserOperation(..) => Box::new(std::iter::empty()), Operation::Delete { updated_fragments, deleted_fragment_ids,