diff --git a/be/src/format_v2/table/lance_reader.cpp b/be/src/format_v2/table/lance_reader.cpp index 724495bf990329..52cf45533e5548 100644 --- a/be/src/format_v2/table/lance_reader.cpp +++ b/be/src/format_v2/table/lance_reader.cpp @@ -697,6 +697,55 @@ void LanceTableReader::_init_scanner_profile() { TUnit::UNIT, LANCE_READER_PROFILE, 1)}, }; _lance_time_metrics = { + // Partition stages accumulate across concurrent work and overlap their parent timers. + // DistanceTopK includes fused candidate filtering, scoring, and heap updates. + {"index_open_time", ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexOpenTime", + LANCE_READER_PROFILE, 1)}, + {"index_partition_load_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexPartitionLoadTime", + LANCE_READER_PROFILE, 1)}, + {"index_partition_prepare_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexPartitionPrepareTime", + LANCE_READER_PROFILE, 1)}, + {"index_prefilter_wait_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexPrefilterWaitTime", + LANCE_READER_PROFILE, 1)}, + {"index_cpu_queue_wait_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexCpuQueueWaitTime", + LANCE_READER_PROFILE, 1)}, + {"index_search_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexSearchTime", + LANCE_READER_PROFILE, 1)}, + {"index_query_prepare_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexQueryPrepareTime", + LANCE_READER_PROFILE, 1)}, + {"index_distance_topk_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexDistanceTopKTime", + LANCE_READER_PROFILE, 1)}, + {"index_result_materialize_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexResultMaterializeTime", + LANCE_READER_PROFILE, 1)}, + {"ANNIVFPartitionExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceANNPartitionExecTime", + LANCE_READER_PROFILE, 1)}, + {"ANNSubIndexExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceANNSubIndexExecTime", + LANCE_READER_PROFILE, 1)}, + {"ANNIvfBatchExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceANNBatchExecTime", + LANCE_READER_PROFILE, 1)}, + {"SortExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceSortComputeTime", + LANCE_READER_PROFILE, 1)}, + {"SortPreservingMergeExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceSortMergeComputeTime", + LANCE_READER_PROFILE, 1)}, + {"TakeExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceTakeExecTime", LANCE_READER_PROFILE, + 1)}, + {"KNNVectorDistanceExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceVectorDistanceComputeTime", + LANCE_READER_PROFILE, 1)}, // These are wall times in the ANN row-id loader. LoadTime includes input polling // and set construction; it must not be added to its component timers. {"prefilter_load_time", diff --git a/be/test/format_v2/table/lance_reader_test.cpp b/be/test/format_v2/table/lance_reader_test.cpp index a751e3aa80e86b..93531688634ca2 100644 --- a/be/test/format_v2/table/lance_reader_test.cpp +++ b/be/test/format_v2/table/lance_reader_test.cpp @@ -912,6 +912,15 @@ TEST(LanceTableReaderVectorSearchTest, MultiVectorScoresFiltersOffsetsAndIndexed } EXPECT_TRUE(reader.close().ok()); if (indexed) { + // Warm searches still perform scoring even when every index partition is cached. + for (const char* name : + {"LanceIndexPartitionLoadTime", "LanceIndexCpuQueueWaitTime", + "LanceIndexSearchTime", "LanceIndexQueryPrepareTime", + "LanceIndexDistanceTopKTime", "LanceIndexResultMaterializeTime"}) { + auto* counter = profile.get_counter(name); + ASSERT_NE(nullptr, counter) << name; + EXPECT_GT(counter->value(), 0) << name; + } // Read metrics after close: lance-c publishes its final execution summary // when the stream is released, including for an early top-k stop. for (const char* name : {"LancePrefilterLoads", "LancePrefilterInputRows", diff --git a/docs/lance-ann-profile.md b/docs/lance-ann-profile.md new file mode 100644 index 00000000000000..51fdc15d095ef6 --- /dev/null +++ b/docs/lance-ann-profile.md @@ -0,0 +1,67 @@ + + +# Lance ANN profile timings + +`FileScannerV2` accumulates scanner initialization, open, block-read, and close +wall time. Range acquisition and split preparation are nested within those +calls, so their counters are not additional time. Scanner worker scheduling +wait is reported separately. `LanceScannerReadTime` measures time spent calling +the Lance scanner, including Rust execution and waits. Doris scanner CPU time +does not include work done on Lance's CPU pool. + +The following counters expose the work within an indexed vector search: + +| Counter | Scope | +| --- | --- | +| `LanceIndexOpenTime` | Index-handle lookup/open, including metadata reads on a miss. | +| `LanceIVFPartitionRankingTime` | Partition ranking, including its CPU dispatch wait. | +| `LanceIndexPartitionLoadTime` | Partition cache lookup, coalesced-load wait, and read/decode on a miss. Also measured on cache hits. | +| `LanceIndexPartitionPrepareTime` | Partition load and per-partition filter preparation. On the streaming path, shared-filter waiting overlaps loading. | +| `LanceIndexPrefilterWaitTime` | Waiting for the shared prefilter to become ready. This is distinct from building the filter. | +| `LanceIndexCpuQueueWaitTime` | Delay before a dispatched search or result-materialization CPU task starts. | +| `LanceIndexSearchTime` | Search of prepared partitions on the CPU pool, including query preparation and any per-partition result construction. | +| `LanceIndexQueryPrepareTime` | Distance-calculator / lookup-table construction in the IVF flat sub-index (including quantized storage). | +| `LanceIndexDistanceTopKTime` | Candidate filtering, distance evaluation, and heap updates in that sub-index. These operations are fused in fast-scan paths. | +| `LanceIndexResultMaterializeTime` | Converting result heaps into Arrow arrays and batches, excluding final global sorting. | +| `LanceANNPartitionExecTime`, `LanceANNSubIndexExecTime`, `LanceANNBatchExecTime` | Baseline elapsed times reported by the corresponding Lance ANN operators. These include asynchronous waits. | +| `LanceSortComputeTime`, `LanceSortMergeComputeTime` | DataFusion sort / sort-preserving merge operator compute times. | +| `LanceTakeExecTime` | Baseline time reported by Lance's take operator within the scan plan. Doris second-phase row-ID fetch has separate counters. | +| `LanceVectorDistanceComputeTime` | Baseline reported by the vector-distance operator, e.g. refinement or an unindexed tail. | + +Timings accumulate across partitions, tasks, and index segments. They are +**nested and may overlap**, so summing them does not reconstruct query wall time. +In particular, partition preparation contains loading; search contains query +preparation and distance/TopK work; ANN operator baselines contain downstream +search stages and waits. A zero counter can mean the corresponding operator or +path was not used. Detailed sub-index timers currently cover IVF flat sub-indices; +other sub-indices are visible through the encompassing search timer. + +`LancePrefilterLoadTime` includes `LancePrefilterInputTime` and +`LancePrefilterBuildTime`; do not add these three together. A segment-scoped +search without a predicate can avoid constructing the row-ID allowlist, while +still respecting deletions and fragment visibility. Filter-readiness waiting can +therefore remain nonzero even when the row-ID materialization counters are zero. + +For a warm query with no execution I/O and no prefilter materialization, inspect +CPU queue wait, query preparation, distance/TopK, and result/sort timers. For cold +queries, inspect partition load together with execution bytes, requests, and +partition cache misses. Use repeated queries and the operator-level elapsed +times to assess latency; cumulative parallel stage times alone are not a critical +path trace. diff --git a/thirdparty/download-thirdparty.sh b/thirdparty/download-thirdparty.sh index e40a6f3d8b872e..431d5a5d013443 100755 --- a/thirdparty/download-thirdparty.sh +++ b/thirdparty/download-thirdparty.sh @@ -718,30 +718,23 @@ if [[ " ${TP_ARCHIVES[*]} " =~ " AZURE " ]]; then echo "Finished patching ${AZURE_SOURCE}" fi -# Apply Doris lance-c patches as one chain to the pinned release archive. +# Foyer remains a local patch until its cache interface is accepted upstream. +# All search fixes are supplied by the immutable lance-c dependency revision. if [[ " ${TP_ARCHIVES[*]} " =~ " LANCE_C " ]]; then - cd "${TP_SOURCE_DIR}/${LANCE_C_SOURCE}" - LANCE_C_PATCHED_MARK="${PATCHED_MARK}_community_pr83_prefilter" - # Older source caches carry a different PR #73 and cannot accept this chain incrementally. - if [[ -f "${PATCHED_MARK}" && ! -f "${LANCE_C_PATCHED_MARK}" ]]; then - echo "The lance-c patch chain changed; remove ${TP_SOURCE_DIR}/${LANCE_C_SOURCE} and rebuild." - exit 1 + foyer_patch_checksum="$(cksum < "${TP_PATCH_DIR}/lance-c-foyer.patch")" + foyer_patch_marker="${TP_SOURCE_DIR}/${LANCE_C_SOURCE}/${PATCHED_MARK}_foyer" + # A new local patch must also replace previously patched cached sources. + # Empty markers from older builds cannot identify the applied patch version. + if [[ -f "${foyer_patch_marker}" ]] && + [[ "$(cat "${foyer_patch_marker}")" != "${foyer_patch_checksum}" ]]; then + rm -rf "${TP_SOURCE_DIR}/${LANCE_C_SOURCE}" + "${TAR_CMD}" xzf "${TP_SOURCE_DIR}/${LANCE_C_NAME}" -C "${TP_SOURCE_DIR}/" fi - if [[ ! -f "${LANCE_C_PATCHED_MARK}" ]]; then - # PR #77 provides Lance v11 for the following community patches. PR #83 - # retains PR #79's scalar-segment path when adding multi-vector execution. - # The final patch pins the full-snapshot prefilter fix and its execution metrics. - for lance_patch in pr-74 pr-75-pr-78 pr-77 pr-73 pr-79 pr-80 pr-83 prefilter; do - patch --batch --forward --reject-file=- --fuzz=0 --no-backup-if-mismatch -s \ - -p1 <"${TP_PATCH_DIR}/${LANCE_C_SOURCE}-${lance_patch}.patch" - done - touch "${PATCHED_MARK}" "${LANCE_C_PATCHED_MARK}" - fi - # Cached sources may carry the earlier prefilter pin; upgrade FTS metrics independently. - if [[ ! -f "${PATCHED_MARK}_prefilter_fts" ]]; then + cd "${TP_SOURCE_DIR}/${LANCE_C_SOURCE}" + if [[ ! -f "${PATCHED_MARK}_foyer" ]]; then patch --batch --forward --reject-file=- --fuzz=0 --no-backup-if-mismatch -s \ - -p1 <"${TP_PATCH_DIR}/${LANCE_C_SOURCE}-prefilter-fts.patch" - touch "${PATCHED_MARK}_prefilter_fts" + -p1 <"${TP_PATCH_DIR}/lance-c-foyer.patch" + printf '%s\n' "${foyer_patch_checksum}" > "${PATCHED_MARK}_foyer" fi cd - echo "Finished patching ${LANCE_C_SOURCE}" diff --git a/thirdparty/patches/lance-c-0.1.9-pr-74.patch b/thirdparty/patches/lance-c-0.1.9-pr-74.patch deleted file mode 100644 index 24c6d33457d84c..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-74.patch +++ /dev/null @@ -1,1159 +0,0 @@ -From b07f970bf2cf3f983cc6043bc72a0fe444c0fd4c Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 3 Sep 2026 17:16:34 +0800 -Subject: [PATCH 1/2] fts - ---- - include/lance/lance.h | 74 ++++++++--- - include/lance/lance.hpp | 44 +++++-- - src/fts_query.rs | 235 +++++++++++++++++++++++++++++------ - src/scanner.rs | 72 ++++++++--- - tests/c_api_test.rs | 264 +++++++++++++++++++++++++++++++++++++--- - 5 files changed, 595 insertions(+), 94 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 3bf291f..0c76edc 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1695,39 +1695,85 @@ typedef enum { - LANCE_FTS_COVERAGE_INDEX_ONLY = 1, - } LanceFtsCoverageMode; - -+/** How the analyzed terms of one Match query are combined. */ -+typedef enum { -+ /** At least one analyzed term must match. */ -+ LANCE_FTS_MATCH_OPERATOR_OR = 0, -+ /** Every analyzed term must match. */ -+ LANCE_FTS_MATCH_OPERATOR_AND = 1, -+} LanceFtsMatchOperator; -+ -+/** -+ * Prepare an OR Match query context for one column. -+ * -+ * @deprecated Use lance_dataset_prepare_fts_match_query() to select the Match -+ * operator explicitly. This compatibility API is equivalent to -+ * LANCE_FTS_MATCH_OPERATOR_OR. -+ */ -+LanceFtsQueryContext* lance_dataset_prepare_fts_query( -+ const LanceDataset* dataset, -+ const char* column, -+ const char* query, -+ uint32_t max_fuzzy_distance, -+ int32_t coverage_mode -+); -+ - /** -- * Prepare an immutable, process-local FTS query context for one column. -+ * Prepare an immutable, process-local Match query context for one column. - * - * Preparation pins the dataset handle's current snapshot, enumerates all - * committed FTS segments for `column`, checks fragment coverage, opens those -- * segments, and computes one query-specific global BM25 scorer across their -- * indexed documents. The context can then be shared by any number of scanners -- * created from the exact same process-local dataset snapshot. It has no -- * serialization or cross-process transport format. Reopening the same URI and -- * manifest version creates a different identity and cannot reuse the context, -- * because storage options and object-store endpoints may differ. -+ * segments, and prepares one global BM25 scorer across their indexed -+ * documents. `match_operator` supports both AND and OR. -+ * -+ * The context can be shared by scanners created from the exact same -+ * process-local dataset snapshot. It has no serialization or cross-process -+ * transport format. Reopening the same URI and manifest version creates a -+ * different identity and cannot reuse the context because storage options and -+ * object-store endpoints may differ. - * - * In LANCE_FTS_COVERAGE_INDEX_ONLY mode, unindexed fragments are allowed and - * excluded from both the scorer corpus and query results. In STRICT mode any - * unindexed fragment makes this call fail. - * -- * Prepared contexts currently support exact Match queries only. -- * `max_fuzzy_distance` must be zero because fuzzy execution requires its -- * canonical expanded vocabulary to be prepared together with the scorer. -- * This restriction does not apply to lance_scanner_full_text_search(). -- * -- * @param max_fuzzy_distance Must be zero for prepared query contexts. -+ * @param match_operator Fixed-width LanceFtsMatchOperator discriminant. -+ * @param max_fuzzy_distance Reserved for prepared fuzzy matching and currently -+ * must be 0. The parameter is retained so enabling -+ * canonical cross-segment fuzzy vocabulary injection -+ * later does not require another C ABI change. - * @param coverage_mode Fixed-width LanceFtsCoverageMode discriminant. - * @return Context handle on success, or NULL on error. - */ --LanceFtsQueryContext* lance_dataset_prepare_fts_query( -+LanceFtsQueryContext* lance_dataset_prepare_fts_match_query( - const LanceDataset* dataset, - const char* column, - const char* query, -+ int32_t match_operator, - uint32_t max_fuzzy_distance, - int32_t coverage_mode - ); - -+/** -+ * Prepare an immutable, process-local Phrase query context for one column. -+ * -+ * The selected FTS index must store token positions. `slop == 0` requires an -+ * exact phrase; a positive value permits that many intervening positions. -+ * Dataset identity, coverage, sharing, and segment-scoped execution follow the -+ * same contract as lance_dataset_prepare_fts_match_query(). -+ * -+ * @param slop Maximum number of intervening token positions permitted between -+ * adjacent phrase terms. -+ * @param coverage_mode Fixed-width LanceFtsCoverageMode discriminant. -+ * @return Context handle on success, or NULL on error. -+ */ -+LanceFtsQueryContext* lance_dataset_prepare_fts_phrase_query( -+ const LanceDataset* dataset, -+ const char* column, -+ const char* query, -+ uint32_t slop, -+ int32_t coverage_mode -+); -+ - /** - * Close a context handle. NULL-safe. Scanners that already attached this - * context retain shared ownership and remain valid. -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 6cf245f..973216a 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -128,6 +128,11 @@ enum class FtsCoverageMode : int32_t { - IndexOnly = LANCE_FTS_COVERAGE_INDEX_ONLY, - }; - -+enum class FtsMatchOperator : int32_t { -+ Or = LANCE_FTS_MATCH_OPERATOR_OR, -+ And = LANCE_FTS_MATCH_OPERATOR_AND, -+}; -+ - /// Tunable parameters for Dataset::write. Numeric fields default-out via 0; - /// `data_storage_version` defaults out via `std::nullopt`. - /// -@@ -764,18 +769,43 @@ class Dataset { - /// Create a Scanner builder for this dataset. - Scanner scan() const; - -- /// Prepare a query-specific global BM25 scorer over the committed FTS -- /// segments of this pinned snapshot. IndexOnly permits unindexed fragments; -- /// Strict rejects them. Prepared contexts currently require -- /// `max_fuzzy_distance == 0`. The context can only be attached to scanners -- /// created from this exact process-local dataset snapshot. -+ /// Compatibility wrapper for an OR Match query. -+ [[deprecated("Use prepare_fts_match_query() to select the Match operator")]] - FtsQueryContext prepare_fts_query( - const std::string& column, - const std::string& query, - uint32_t max_fuzzy_distance = 0, - FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { -- auto* context = lance_dataset_prepare_fts_query( -- handle_.get(), column.c_str(), query.c_str(), max_fuzzy_distance, -+ return prepare_fts_match_query(column, query, FtsMatchOperator::Or, -+ max_fuzzy_distance, coverage_mode); -+ } -+ -+ /// Prepare a Match query with a global BM25 scorer. AND and OR are -+ /// supported. `max_fuzzy_distance` is reserved and currently must be zero; -+ /// keeping it here avoids another API change when canonical cross-segment -+ /// fuzzy vocabulary injection becomes available. -+ FtsQueryContext prepare_fts_match_query( -+ const std::string& column, -+ const std::string& query, -+ FtsMatchOperator match_operator = FtsMatchOperator::Or, -+ uint32_t max_fuzzy_distance = 0, -+ FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { -+ auto* context = lance_dataset_prepare_fts_match_query( -+ handle_.get(), column.c_str(), query.c_str(), -+ static_cast(match_operator), max_fuzzy_distance, -+ static_cast(coverage_mode)); -+ if (!context) check_error(); -+ return FtsQueryContext(context); -+ } -+ -+ /// Prepare a Phrase query. Its FTS index must store token positions. -+ FtsQueryContext prepare_fts_phrase_query( -+ const std::string& column, -+ const std::string& query, -+ uint32_t slop = 0, -+ FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { -+ auto* context = lance_dataset_prepare_fts_phrase_query( -+ handle_.get(), column.c_str(), query.c_str(), slop, - static_cast(coverage_mode)); - if (!context) check_error(); - return FtsQueryContext(context); -diff --git a/src/fts_query.rs b/src/fts_query.rs -index cd194c7..3bf7f3e 100644 ---- a/src/fts_query.rs -+++ b/src/fts_query.rs -@@ -14,7 +14,9 @@ use lance_core::{Error, Result}; - use lance_index::IndexCriteria; - use lance_index::metrics::NoOpMetricsCollector; - use lance_index::scalar::FullTextSearchQuery; --use lance_index::scalar::inverted::query::{FtsQuery, collect_query_tokens}; -+use lance_index::scalar::inverted::query::{ -+ FtsQuery, MatchQuery, Operator, PhraseQuery, collect_query_tokens, -+}; - use lance_index::scalar::inverted::{InvertedIndex, MemBM25Scorer, build_global_bm25_scorer}; - use lance_table::format::IndexMetadata; - use uuid::Uuid; -@@ -48,12 +50,53 @@ impl TryFrom for LanceFtsCoverageMode { - } - } - -+/// Operator used to combine the analyzed terms of a Match query. -+#[repr(i32)] -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub enum LanceFtsMatchOperator { -+ /// At least one analyzed term must match. -+ Or = 0, -+ /// Every analyzed term must match. -+ And = 1, -+} -+ -+impl TryFrom for LanceFtsMatchOperator { -+ type Error = Error; -+ -+ fn try_from(value: i32) -> Result { -+ match value { -+ 0 => Ok(Self::Or), -+ 1 => Ok(Self::And), -+ _ => Err(Error::invalid_input(format!( -+ "invalid match_operator {value}; expected 0 (OR) or 1 (AND)" -+ ))), -+ } -+ } -+} -+ -+impl From for Operator { -+ fn from(value: LanceFtsMatchOperator) -> Self { -+ match value { -+ LanceFtsMatchOperator::Or => Self::Or, -+ LanceFtsMatchOperator::And => Self::And, -+ } -+ } -+} -+ -+/// Query-specific state that must be shared by every segment-scoped scan. -+pub(crate) enum PreparedFtsQuery { -+ /// Exact Match queries share one corpus-wide scorer. -+ Match(Arc), -+ /// Phrase does not expand terms, so a shared global scorer is sufficient. -+ Phrase(Arc), -+} -+ - /// Rust-owned immutable state behind [`LanceFtsQueryContext`]. - pub(crate) struct FtsQueryContextInner { - pub(crate) dataset: Arc, - pub(crate) query: FullTextSearchQuery, - pub(crate) segments: Vec, -- pub(crate) scorer: Arc, -+ pub(crate) prepared: PreparedFtsQuery, - } - - impl FtsQueryContextInner { -@@ -87,7 +130,7 @@ fn invalid_input(message: impl Into) -> Error { - async fn prepare_fts_query_context( - dataset: Arc, - column: String, -- query_text: String, -+ query: FullTextSearchQuery, - coverage_mode: LanceFtsCoverageMode, - ) -> Result { - let logical_index = dataset -@@ -193,34 +236,78 @@ async fn prepare_fts_query_context( - ))); - } - -- let query = FullTextSearchQuery::new(query_text).with_column(column.clone())?; -- let match_query = match &query.query { -- FtsQuery::Match(query) => query, -+ let prepared = match &query.query { -+ FtsQuery::Match(match_query) => { -+ let mut tokenizer = indices[0].tokenizer(); -+ let query_tokens = collect_query_tokens(&match_query.terms, &mut tokenizer); -+ let params = query -+ .params() -+ .with_fuzziness(match_query.fuzziness) -+ .with_max_expansions(match_query.max_expansions) -+ .with_prefix_length(match_query.prefix_length); -+ PreparedFtsQuery::Match(Arc::new( -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ )) -+ } -+ FtsQuery::Phrase(phrase_query) => { -+ if !expected_params.has_positions() { -+ return Err(invalid_input(format!( -+ "FTS index '{}' for column '{column}' does not store token positions required by Phrase queries; recreate the index with positions enabled", -+ logical_index.name -+ ))); -+ } -+ let mut tokenizer = indices[0].tokenizer(); -+ let query_tokens = collect_query_tokens(&phrase_query.terms, &mut tokenizer); -+ let params = query.params().with_phrase_slop(Some(phrase_query.slop)); -+ PreparedFtsQuery::Phrase(Arc::new( -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ )) -+ } - _ => { - return Err(Error::internal( -- "prepared FTS query unexpectedly produced a non-Match query".to_string(), -+ "prepared FTS query must be a single-column Match or Phrase query".to_string(), - )); - } - }; -- let mut tokenizer = indices[0].tokenizer(); -- let query_tokens = collect_query_tokens(&match_query.terms, &mut tokenizer); -- let params = query -- .params() -- .with_fuzziness(match_query.fuzziness) -- .with_max_expansions(match_query.max_expansions) -- .with_prefix_length(match_query.prefix_length); -- let scorer = Arc::new(build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?); - - Ok(FtsQueryContextInner { - dataset, - query, - segments, -- scorer, -+ prepared, - }) - } - --/// Prepare a process-local global BM25 scorer and the committed segment list --/// for one single-column Match query against the dataset's pinned snapshot. -+unsafe fn parse_query_inputs( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ coverage_mode: i32, -+) -> Result<(Arc, String, String, LanceFtsCoverageMode)> { -+ if dataset.is_null() || column.is_null() || query.is_null() { -+ return Err(invalid_input("dataset, column, and query must not be NULL")); -+ } -+ let column = unsafe { helpers::parse_c_string(column)? } -+ .filter(|value| !value.is_empty()) -+ .ok_or_else(|| invalid_input("column must not be empty"))? -+ .to_string(); -+ let query = unsafe { helpers::parse_c_string(query)? } -+ .filter(|value| !value.is_empty()) -+ .ok_or_else(|| invalid_input("query must not be empty"))? -+ .to_string(); -+ let coverage_mode = LanceFtsCoverageMode::try_from(coverage_mode)?; -+ let snapshot = unsafe { &*dataset }.snapshot(); -+ Ok((snapshot, column, query, coverage_mode)) -+} -+ -+fn into_context(inner: FtsQueryContextInner) -> *mut LanceFtsQueryContext { -+ Box::into_raw(Box::new(LanceFtsQueryContext { -+ inner: Arc::new(inner), -+ })) -+} -+ -+/// Compatibility API for an OR Match query. -+#[deprecated(note = "use lance_dataset_prepare_fts_match_query to select the Match operator")] - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_dataset_prepare_fts_query( - dataset: *const LanceDataset, -@@ -231,46 +318,120 @@ pub unsafe extern "C" fn lance_dataset_prepare_fts_query( - ) -> *mut LanceFtsQueryContext { - ffi_try!( - unsafe { -- prepare_fts_query_inner(dataset, column, query, max_fuzzy_distance, coverage_mode) -+ prepare_fts_match_query_inner( -+ dataset, -+ column, -+ query, -+ LanceFtsMatchOperator::Or as i32, -+ max_fuzzy_distance, -+ coverage_mode, -+ ) -+ }, -+ null -+ ) -+} -+ -+/// Prepare a process-local Match query context. AND and OR are supported. -+/// `max_fuzzy_distance` is retained for the future prepared-fuzzy path but -+/// must be zero with the currently pinned Lance revision. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_dataset_prepare_fts_match_query( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ match_operator: i32, -+ max_fuzzy_distance: u32, -+ coverage_mode: i32, -+) -> *mut LanceFtsQueryContext { -+ ffi_try!( -+ unsafe { -+ prepare_fts_match_query_inner( -+ dataset, -+ column, -+ query, -+ match_operator, -+ max_fuzzy_distance, -+ coverage_mode, -+ ) - }, - null - ) - } - --unsafe fn prepare_fts_query_inner( -+unsafe fn prepare_fts_match_query_inner( - dataset: *const LanceDataset, - column: *const c_char, - query: *const c_char, -+ match_operator: i32, - max_fuzzy_distance: u32, - coverage_mode: i32, - ) -> Result<*mut LanceFtsQueryContext> { -- if dataset.is_null() || column.is_null() || query.is_null() { -- return Err(invalid_input("dataset, column, and query must not be NULL")); -- } -- let column = unsafe { helpers::parse_c_string(column)? } -- .filter(|value| !value.is_empty()) -- .ok_or_else(|| invalid_input("column must not be empty"))? -- .to_string(); -- let query = unsafe { helpers::parse_c_string(query)? } -- .filter(|value| !value.is_empty()) -- .ok_or_else(|| invalid_input("query must not be empty"))? -- .to_string(); -- let coverage_mode = LanceFtsCoverageMode::try_from(coverage_mode)?; -+ let (snapshot, column, query_text, coverage_mode) = -+ unsafe { parse_query_inputs(dataset, column, query, coverage_mode)? }; -+ let operator: Operator = LanceFtsMatchOperator::try_from(match_operator)?.into(); -+ // The parameter remains in the public API so callers do not need another -+ // ABI change when Lance-C moves to a Lance revision that can inject the -+ // same canonical fuzzy vocabulary into every segment-scoped scan. The -+ // pinned Lance revision can share only the scorer, so accepting fuzzy here -+ // would allow different segments to choose different capped expansions. - if max_fuzzy_distance != 0 { - return Err(invalid_input(format!( -- "max_fuzzy_distance must be 0 for prepared FTS query contexts, got {max_fuzzy_distance}; fuzzy queries require a canonical prepared BM25 vocabulary" -+ "max_fuzzy_distance must be 0 for prepared FTS with the pinned Lance revision, got {max_fuzzy_distance}; the parameter is reserved until canonical fuzzy vocabulary injection is available" - ))); - } -- let snapshot = unsafe { &*dataset }.snapshot(); -+ let query = FullTextSearchQuery::new_query( -+ MatchQuery::new(query_text) -+ .with_column(Some(column.clone())) -+ .with_operator(operator) -+ .with_fuzziness(Some(0)) -+ .into(), -+ ); - let inner = block_on(prepare_fts_query_context( - snapshot, - column, - query, - coverage_mode, - ))?; -- Ok(Box::into_raw(Box::new(LanceFtsQueryContext { -- inner: Arc::new(inner), -- }))) -+ Ok(into_context(inner)) -+} -+ -+/// Prepare a process-local Phrase query context. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_dataset_prepare_fts_phrase_query( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ slop: u32, -+ coverage_mode: i32, -+) -> *mut LanceFtsQueryContext { -+ ffi_try!( -+ unsafe { prepare_fts_phrase_query_inner(dataset, column, query, slop, coverage_mode) }, -+ null -+ ) -+} -+ -+unsafe fn prepare_fts_phrase_query_inner( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ slop: u32, -+ coverage_mode: i32, -+) -> Result<*mut LanceFtsQueryContext> { -+ let (snapshot, column, query_text, coverage_mode) = -+ unsafe { parse_query_inputs(dataset, column, query, coverage_mode)? }; -+ let query = FullTextSearchQuery::new_query( -+ PhraseQuery::new(query_text) -+ .with_column(Some(column.clone())) -+ .with_slop(slop) -+ .into(), -+ ); -+ let inner = block_on(prepare_fts_query_context( -+ snapshot, -+ column, -+ query, -+ coverage_mode, -+ ))?; -+ Ok(into_context(inner)) - } - - /// Close a context handle. NULL-safe. Scanners that already attached the -diff --git a/src/scanner.rs b/src/scanner.rs -index 0c29b17..cbf1b13 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -18,7 +18,7 @@ use lance::Dataset; - use lance::dataset::scanner::{ - DatasetRecordBatchStream, ExecutionStatsCallback, ExecutionSummaryCounts, - }; --use lance::io::exec::fts::{FlatMatchQueryExec, MatchQueryExec}; -+use lance::io::exec::fts::{FlatMatchQueryExec, MatchQueryExec, PhraseQueryExec}; - use lance_core::Result; - use lance_index::scalar::FullTextSearchQuery; - use lance_io::stream::RecordBatchStream; -@@ -33,7 +33,8 @@ use crate::error::{ - panic_payload_message, set_lance_error, set_last_error, swallow_unwind, - }; - use crate::fts_query::{ -- FtsQueryContextInner, LanceFtsQueryContext, clone_context, parse_segment_uuids, -+ FtsQueryContextInner, LanceFtsQueryContext, PreparedFtsQuery, clone_context, -+ parse_segment_uuids, - }; - use crate::helpers; - use crate::runtime::{RT, block_on}; -@@ -328,17 +329,17 @@ impl PreparedScanner { - let (plan, rewritten) = rewrite_prepared_fts_plan( - plan, - &distributed_fts.segments, -- &distributed_fts.context.scorer, -+ &distributed_fts.context.prepared, - selected_segments_have_current_fragments, - )?; -- if rewritten.match_query_execs > 1 -+ if rewritten.indexed_query_execs > 1 - || rewritten.flat_match_query_execs > 1 -- || rewritten.match_query_execs + rewritten.flat_match_query_execs == 0 -- || (selected_segments_have_current_fragments && rewritten.match_query_execs != 1) -+ || rewritten.indexed_query_execs + rewritten.flat_match_query_execs == 0 -+ || (selected_segments_have_current_fragments && rewritten.indexed_query_execs != 1) - { - return Err(lance_core::Error::internal(format!( -- "unexpected prepared FTS plan for selected segments with current fragment coverage {selected_segments_have_current_fragments}: rewrote {} MatchQueryExec node(s) and removed {} FlatMatchQueryExec node(s)", -- rewritten.match_query_execs, rewritten.flat_match_query_execs -+ "unexpected prepared FTS plan for selected segments with current fragment coverage {selected_segments_have_current_fragments}: rewrote {} indexed FTS query node(s) and removed {} FlatMatchQueryExec node(s)", -+ rewritten.indexed_query_execs, rewritten.flat_match_query_execs - ))); - } - let stream = lance_datafusion::exec::execute_plan( -@@ -420,14 +421,14 @@ fn segments_have_current_fragments( - - #[derive(Default)] - struct PreparedFtsPlanRewriteCounts { -- match_query_execs: usize, -+ indexed_query_execs: usize, - flat_match_query_execs: usize, - } - - fn rewrite_prepared_fts_plan( - plan: Arc, - segments: &[IndexMetadata], -- scorer: &Arc, -+ prepared: &PreparedFtsQuery, - selected_segments_have_current_fragments: bool, - ) -> Result<(Arc, PreparedFtsPlanRewriteCounts)> { - // Lance's ordinary FTS planner adds a flat-search branch for fragments not -@@ -438,7 +439,7 @@ fn rewrite_prepared_fts_plan( - return Ok(( - Arc::new(EmptyExec::new(plan.schema())), - PreparedFtsPlanRewriteCounts { -- match_query_execs: 0, -+ indexed_query_execs: 0, - flat_match_query_execs: 1, - }, - )); -@@ -454,11 +455,11 @@ fn rewrite_prepared_fts_plan( - let (new_child, child_rewritten) = rewrite_prepared_fts_plan( - Arc::clone(child), - segments, -- scorer, -+ prepared, - selected_segments_have_current_fragments, - )?; - new_children.push(new_child); -- rewritten.match_query_execs += child_rewritten.match_query_execs; -+ rewritten.indexed_query_execs += child_rewritten.indexed_query_execs; - rewritten.flat_match_query_execs += child_rewritten.flat_match_query_execs; - } - plan.with_new_children(new_children).map_err(|error| { -@@ -469,10 +470,15 @@ fn rewrite_prepared_fts_plan( - }; - - if let Some(exec) = rebuilt.downcast_ref::() { -- rewritten.match_query_execs += 1; -+ rewritten.indexed_query_execs += 1; - if !selected_segments_have_current_fragments { - return Ok((Arc::new(EmptyExec::new(rebuilt.schema())), rewritten)); - } -+ let PreparedFtsQuery::Match(scorer) = prepared else { -+ return Err(lance_core::Error::internal( -+ "prepared Phrase state cannot be attached to MatchQueryExec".to_string(), -+ )); -+ }; - let replacement = MatchQueryExec::new_with_segments( - Arc::clone(exec.dataset()), - exec.query().clone(), -@@ -483,6 +489,26 @@ fn rewrite_prepared_fts_plan( - .with_base_scorer(Arc::clone(scorer)); - return Ok((Arc::new(replacement), rewritten)); - } -+ if let Some(exec) = rebuilt.downcast_ref::() { -+ rewritten.indexed_query_execs += 1; -+ if !selected_segments_have_current_fragments { -+ return Ok((Arc::new(EmptyExec::new(rebuilt.schema())), rewritten)); -+ } -+ let PreparedFtsQuery::Phrase(scorer) = prepared else { -+ return Err(lance_core::Error::internal( -+ "prepared Match state cannot be attached to PhraseQueryExec".to_string(), -+ )); -+ }; -+ let replacement = PhraseQueryExec::new_with_segments( -+ Arc::clone(exec.dataset()), -+ exec.query().clone(), -+ exec.params().clone(), -+ exec.prefilter_source().clone(), -+ segments.to_vec(), -+ ) -+ .with_base_scorer(Arc::clone(scorer)); -+ return Ok((Arc::new(replacement), rewritten)); -+ } - Ok((rebuilt, rewritten)) - } - -@@ -2152,7 +2178,8 @@ mod tests { - use crate::dataset::{lance_dataset_close, lance_dataset_open}; - use crate::error::{lance_last_error_code, lance_last_error_message}; - use crate::fts_query::{ -- LanceFtsCoverageMode, lance_dataset_prepare_fts_query, lance_fts_query_context_close, -+ LanceFtsCoverageMode, LanceFtsMatchOperator, lance_dataset_prepare_fts_match_query, -+ lance_fts_query_context_close, - }; - use std::ffi::{CStr, CString}; - use std::sync::atomic::{AtomicI32, AtomicUsize}; -@@ -2276,10 +2303,11 @@ mod tests { - let column = CString::new("name").unwrap(); - let query = CString::new("a").unwrap(); - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::IndexOnly as i32, - ) -@@ -2293,7 +2321,6 @@ mod tests { - let prepared = unsafe { &*scanner }.build_scanner().unwrap(); - let distributed = prepared.distributed_fts.as_ref().unwrap(); - let segments = distributed.segments.clone(); -- let scorer = Arc::clone(&distributed.context.scorer); - let plan = block_on(prepared.scanner.create_plan()).unwrap(); - assert_eq!( - prepared_fts_plan_shape(&plan), -@@ -2303,9 +2330,14 @@ mod tests { - - let has_current_fragments = - segments_have_current_fragments(&distributed.context.dataset, &segments).unwrap(); -- let (rewritten, counts) = -- rewrite_prepared_fts_plan(plan, &segments, &scorer, has_current_fragments).unwrap(); -- assert_eq!(counts.match_query_execs, 1); -+ let (rewritten, counts) = rewrite_prepared_fts_plan( -+ plan, -+ &segments, -+ &distributed.context.prepared, -+ has_current_fragments, -+ ) -+ .unwrap(); -+ assert_eq!(counts.indexed_query_execs, 1); - assert_eq!(counts.flat_match_query_execs, 1); - assert_eq!(prepared_fts_plan_shape(&rewritten), (1, 0, 0)); - -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 8805764..a3e4ef7 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -5762,6 +5762,203 @@ fn load_fts_segment_uuids(uri: &str, column: &str) -> Vec<[u8; 16]> { - }) - } - -+#[test] -+#[allow(deprecated)] -+fn test_prepared_fts_match_phrase_and_legacy_compatibility() { -+ let tmp = tempfile::tempdir().unwrap(); -+ let uri = tmp -+ .path() -+ .join("prepared_fts_queries") -+ .to_str() -+ .unwrap() -+ .to_string(); -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("text", DataType::Utf8, false), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from(vec![1, 2, 3, 4, 5])), -+ Arc::new(StringArray::from(vec![ -+ "quick brown fox", -+ "quick blue fox", -+ "slow brown fox", -+ "quik brown fox", -+ "quick red brown fox", -+ ])), -+ ], -+ ) -+ .unwrap(); -+ lance_c::runtime::block_on(async { -+ Dataset::write( -+ arrow::record_batch::RecordBatchIterator::new(vec![Ok(batch)], schema), -+ &uri, -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ -+ let uri_c = c_str(&uri); -+ let column = c_str("text"); -+ let index_params = -+ c_str(r#"{"base_tokenizer":"simple","language":"English","with_position":true}"#); -+ let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert_eq!( -+ unsafe { -+ lance_dataset_create_scalar_index( -+ dataset, -+ column.as_ptr(), -+ ptr::null(), -+ LanceScalarIndexType::Inverted as i32, -+ index_params.as_ptr(), -+ false, -+ ) -+ }, -+ 0 -+ ); -+ -+ let query = c_str("quick brown"); -+ let exact_or = unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!exact_or.is_null()); -+ assert_eq!(collect_context_fts_scores(dataset, exact_or, None).len(), 5); -+ unsafe { lance_fts_query_context_close(exact_or) }; -+ -+ let legacy_or = unsafe { -+ lance_dataset_prepare_fts_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!legacy_or.is_null()); -+ assert_eq!( -+ collect_context_fts_scores(dataset, legacy_or, None).len(), -+ 5 -+ ); -+ unsafe { lance_fts_query_context_close(legacy_or) }; -+ -+ let exact_and = unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::And as i32, -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!exact_and.is_null()); -+ let exact_and_scores = collect_context_fts_scores(dataset, exact_and, None); -+ let mut exact_and_ids = exact_and_scores.keys().copied().collect::>(); -+ exact_and_ids.sort_unstable(); -+ assert_eq!(exact_and_ids, vec![1, 5]); -+ unsafe { lance_fts_query_context_close(exact_and) }; -+ -+ let fuzzy_and = unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::And as i32, -+ 1, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(fuzzy_and.is_null()); -+ let message = take_last_error_message(); -+ assert!( -+ message.contains("max_fuzzy_distance must be 0"), -+ "{message}" -+ ); -+ -+ let phrase = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!phrase.is_null(), "{}", take_last_error_message()); -+ let phrase_scores = collect_context_fts_scores(dataset, phrase, None); -+ assert_eq!(phrase_scores.keys().copied().collect::>(), vec![1]); -+ unsafe { lance_fts_query_context_close(phrase) }; -+ -+ let phrase_with_slop = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 1, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!phrase_with_slop.is_null(), "{}", take_last_error_message()); -+ let phrase_with_slop_scores = collect_context_fts_scores(dataset, phrase_with_slop, None); -+ let mut phrase_with_slop_ids = phrase_with_slop_scores.keys().copied().collect::>(); -+ phrase_with_slop_ids.sort_unstable(); -+ assert_eq!(phrase_with_slop_ids, vec![1, 5]); -+ unsafe { lance_fts_query_context_close(phrase_with_slop) }; -+ -+ unsafe { lance_dataset_close(dataset) }; -+} -+ -+#[test] -+fn test_prepared_fts_phrase_requires_positions() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let column = c_str("name"); -+ let query = c_str("alice smith"); -+ let index_params = c_str(r#"{"base_tokenizer":"simple","language":"English"}"#); -+ let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert_eq!( -+ unsafe { -+ lance_dataset_create_scalar_index( -+ dataset, -+ column.as_ptr(), -+ ptr::null(), -+ LanceScalarIndexType::Inverted as i32, -+ index_params.as_ptr(), -+ false, -+ ) -+ }, -+ 0 -+ ); -+ -+ let context = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(context.is_null()); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ let message = take_last_error_message(); -+ assert!( -+ message.contains("does not store token positions"), -+ "{message}" -+ ); -+ -+ unsafe { lance_dataset_close(dataset) }; -+} -+ - #[test] - fn test_prepared_fts_row_id_output_is_explicit() { - let (_tmp, uri) = create_test_dataset(); -@@ -5785,10 +5982,11 @@ fn test_prepared_fts_row_id_output_is_explicit() { - 0 - ); - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -5880,10 +6078,11 @@ fn test_prepare_fts_query_index_only_allows_unindexed_fragment() { - - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; - let strict = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -5898,10 +6097,11 @@ fn test_prepare_fts_query_index_only_allows_unindexed_fragment() { - assert!(message.contains("unindexed fragments"), "{message}"); - - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::IndexOnly as i32, - ) -@@ -5976,10 +6176,11 @@ fn test_prepared_fts_index_only_empty_segment_returns_empty_shard() { - let query = c_str("alice"); - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::IndexOnly as i32, - ) -@@ -6075,10 +6276,11 @@ fn test_prepared_fts_global_scorer_is_shared_across_segment_splits() { - - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -6186,7 +6388,7 @@ fn test_prepared_fts_global_scorer_is_shared_across_segment_splits() { - } - - #[test] --fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { -+fn test_prepare_fts_queries_reject_invalid_inputs() { - let (_tmp, uri) = create_test_dataset(); - let uri_c = c_str(&uri); - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -@@ -6196,10 +6398,11 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - - assert!( - unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - ptr::null(), - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -6208,10 +6411,11 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - ); - assert!( - unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - empty.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -6219,20 +6423,39 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - .is_null() - ); - assert!( -- unsafe { lance_dataset_prepare_fts_query(dataset, column.as_ptr(), empty.as_ptr(), 0, 0) } -- .is_null() -+ unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ empty.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ } -+ .is_null() - ); - assert!( -- unsafe { lance_dataset_prepare_fts_query(dataset, column.as_ptr(), query.as_ptr(), 0, 99) } -- .is_null() -+ unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, -+ 0, -+ 99, -+ ) -+ } -+ .is_null() - ); - assert!( - unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -- 1, -+ 99, -+ 0, - LanceFtsCoverageMode::Strict as i32, - ) - } -@@ -6244,9 +6467,18 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - .to_string_lossy() - .into_owned() - }; -+ assert!(message.contains("invalid match_operator"), "{message}"); - assert!( -- message.contains("max_fuzzy_distance must be 0"), -- "{message}" -+ unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ ptr::null(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ } -+ .is_null() - ); - let scanner = unsafe { lance_scanner_new(dataset, ptr::null(), ptr::null()) }; - assert_eq!( - -From 9bb38749c9de185664511def3cbeb699ee60a38b Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 3 Sep 2026 17:49:02 +0800 -Subject: [PATCH 2/2] update slop i32 - ---- - include/lance/lance.h | 6 +++--- - include/lance/lance.hpp | 5 +++-- - src/fts_query.rs | 6 ++++-- - tests/c_api_test.rs | 14 ++++++++++++++ - 4 files changed, 24 insertions(+), 7 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 0c76edc..25c0430 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1761,8 +1761,8 @@ LanceFtsQueryContext* lance_dataset_prepare_fts_match_query( - * Dataset identity, coverage, sharing, and segment-scoped execution follow the - * same contract as lance_dataset_prepare_fts_match_query(). - * -- * @param slop Maximum number of intervening token positions permitted between -- * adjacent phrase terms. -+ * @param slop Maximum non-negative number of intervening token positions -+ * permitted between adjacent phrase terms. - * @param coverage_mode Fixed-width LanceFtsCoverageMode discriminant. - * @return Context handle on success, or NULL on error. - */ -@@ -1770,7 +1770,7 @@ LanceFtsQueryContext* lance_dataset_prepare_fts_phrase_query( - const LanceDataset* dataset, - const char* column, - const char* query, -- uint32_t slop, -+ int32_t slop, - int32_t coverage_mode - ); - -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 973216a..1e69a06 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -798,11 +798,12 @@ class Dataset { - return FtsQueryContext(context); - } - -- /// Prepare a Phrase query. Its FTS index must store token positions. -+ /// Prepare a Phrase query. Its FTS index must store token positions and -+ /// slop must be non-negative. - FtsQueryContext prepare_fts_phrase_query( - const std::string& column, - const std::string& query, -- uint32_t slop = 0, -+ int32_t slop = 0, - FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { - auto* context = lance_dataset_prepare_fts_phrase_query( - handle_.get(), column.c_str(), query.c_str(), slop, -diff --git a/src/fts_query.rs b/src/fts_query.rs -index 3bf7f3e..c9d3a43 100644 ---- a/src/fts_query.rs -+++ b/src/fts_query.rs -@@ -401,7 +401,7 @@ pub unsafe extern "C" fn lance_dataset_prepare_fts_phrase_query( - dataset: *const LanceDataset, - column: *const c_char, - query: *const c_char, -- slop: u32, -+ slop: i32, - coverage_mode: i32, - ) -> *mut LanceFtsQueryContext { - ffi_try!( -@@ -414,9 +414,11 @@ unsafe fn prepare_fts_phrase_query_inner( - dataset: *const LanceDataset, - column: *const c_char, - query: *const c_char, -- slop: u32, -+ slop: i32, - coverage_mode: i32, - ) -> Result<*mut LanceFtsQueryContext> { -+ let slop = u32::try_from(slop) -+ .map_err(|_| invalid_input(format!("slop must be non-negative, got {slop}")))?; - let (snapshot, column, query_text, coverage_mode) = - unsafe { parse_query_inputs(dataset, column, query, coverage_mode)? }; - let query = FullTextSearchQuery::new_query( -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index a3e4ef7..9411e15 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -5914,6 +5914,20 @@ fn test_prepared_fts_match_phrase_and_legacy_compatibility() { - assert_eq!(phrase_with_slop_ids, vec![1, 5]); - unsafe { lance_fts_query_context_close(phrase_with_slop) }; - -+ let negative_phrase_slop = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ -1, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(negative_phrase_slop.is_null()); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ let message = take_last_error_message(); -+ assert!(message.contains("slop must be non-negative"), "{message}"); -+ - unsafe { lance_dataset_close(dataset) }; - } - diff --git a/thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch b/thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch deleted file mode 100644 index 08687627e7ea21..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch +++ /dev/null @@ -1,2441 +0,0 @@ -Lance-C v0.1.9 scanner options: PR #75 followed by PR #78. -Apply after lance-c-0.1.9-pr-74.patch, before lance-c-0.1.9-pr-77.patch. -The upstream mail patches below are concatenated without modification. - -PR #75: https://github.com/lance-format/lance-c/pull/75 -Head: 043a1f7eac253d8ac6be3f970b60fcbf615295aa -PR #78: https://github.com/lance-format/lance-c/pull/78 -Head: e894f591aef358cd36fdbf915c4d5b95fd0e8348 - -From 0752592cc4bbc69f8f9333362a00f3bceee73100 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Fri, 4 Sep 2026 23:35:29 +0800 -Subject: [PATCH 1/2] add some setter - ---- - include/lance/lance.h | 81 +++++++++++ - include/lance/lance.hpp | 47 +++++++ - src/scanner.rs | 279 +++++++++++++++++++++++++++++++++++++ - tests/c_api_test.rs | 230 ++++++++++++++++++++++++++++++ - tests/cpp/test_cpp_api.cpp | 7 + - 5 files changed, 644 insertions(+) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 3bf291f..00f415b 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -934,6 +934,74 @@ LanceScanner* lance_scanner_new( - int32_t lance_scanner_set_limit(LanceScanner* scanner, int64_t limit); - int32_t lance_scanner_set_offset(LanceScanner* scanner, int64_t offset); - int32_t lance_scanner_set_batch_size(LanceScanner* scanner, int64_t batch_size); -+ -+/** -+ * Set the target output batch size in bytes. -+ * -+ * When set, this takes precedence over the row-based batch size. The value -+ * must be greater than zero and must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_batch_size_bytes( -+ LanceScanner* scanner, -+ uint64_t batch_size_bytes -+); -+ -+/** -+ * Set the scanner I/O buffer size in bytes. -+ * -+ * The value must be greater than zero and must be set before scanning starts. -+ * This bounds buffered I/O received from storage, but is not a hard limit on -+ * all memory used by the scanner. -+ * -+ * @param scanner Scanner handle. Must not be NULL. -+ * @param io_buffer_size_bytes I/O buffer size in bytes. Must be greater than zero. -+ * @return 0 on success, -1 on error. -+ */ -+int32_t lance_scanner_set_io_buffer_size( -+ LanceScanner* scanner, -+ uint64_t io_buffer_size_bytes -+); -+ -+/** -+ * Set the maximum number of batches decoded concurrently. -+ * -+ * @param batch_readahead Number of in-flight batch decode tasks. Must be greater than zero. -+ */ -+int32_t lance_scanner_set_batch_readahead( -+ LanceScanner* scanner, -+ size_t batch_readahead -+); -+ -+/** -+ * Set fragment readahead for unordered scans. -+ * -+ * This setting is only used when scan-in-order is disabled. The value must be -+ * greater than zero. -+ */ -+int32_t lance_scanner_set_fragment_readahead( -+ LanceScanner* scanner, -+ size_t fragment_readahead -+); -+ -+/** -+ * Set the target number of physical execution partitions. -+ * -+ * This controls the partition count used by the physical optimizer and can be -+ * used to bound scan CPU parallelism. The value must be greater than zero and -+ * must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_target_parallelism( -+ LanceScanner* scanner, -+ size_t target_parallelism -+); -+ -+/** -+ * Configure whether batches are returned in storage order (default: true). -+ * -+ * Disabling ordering can improve throughput by returning batches as soon as -+ * they are ready. -+ */ -+int32_t lance_scanner_set_scan_in_order(LanceScanner* scanner, bool scan_in_order); - int32_t lance_scanner_with_row_id(LanceScanner* scanner, bool enable); - - /** -@@ -1656,6 +1724,19 @@ int32_t lance_scanner_nearest( - ); - - int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t n); -+ -+/** -+ * Set vector index partition-search concurrency for each query. -+ * -+ * A value of -1 uses the CPU pool size, 0 selects Lance's automatic policy, -+ * 1 uses the sequential path, and values greater than 1 request parallel -+ * partition search. The effective value is capped by available parallelism. -+ * Values below -1 are rejected. Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_query_parallelism( -+ LanceScanner* scanner, -+ int32_t query_parallelism -+); - int32_t lance_scanner_set_refine_factor(LanceScanner* scanner, uint32_t f); - int32_t lance_scanner_set_ef(LanceScanner* scanner, uint32_t e); - int32_t lance_scanner_set_metric(LanceScanner* scanner, LanceMetricType metric); -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 6cf245f..3c03f86 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1196,6 +1196,48 @@ class Scanner { - return *this; - } - -+ /// Set the target output batch size in bytes. -+ Scanner& batch_size_bytes(uint64_t bytes) { -+ if (lance_scanner_set_batch_size_bytes(handle_.get(), bytes) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set the scanner I/O buffer size in bytes. -+ Scanner& io_buffer_size(uint64_t bytes) { -+ if (lance_scanner_set_io_buffer_size(handle_.get(), bytes) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set the number of batches decoded concurrently. -+ Scanner& batch_readahead(size_t batches) { -+ if (lance_scanner_set_batch_readahead(handle_.get(), batches) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set fragment readahead for unordered scans. -+ Scanner& fragment_readahead(size_t fragments) { -+ if (lance_scanner_set_fragment_readahead(handle_.get(), fragments) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set the target number of physical execution partitions. -+ Scanner& target_parallelism(size_t partitions) { -+ if (lance_scanner_set_target_parallelism(handle_.get(), partitions) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether batches are returned in storage order. -+ Scanner& scan_in_order(bool ordered = true) { -+ if (lance_scanner_set_scan_in_order(handle_.get(), ordered) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Enable/disable row ID in output. - Scanner& with_row_id(bool enable = true) { - if (lance_scanner_with_row_id(handle_.get(), enable) != 0) -@@ -1313,6 +1355,11 @@ class Scanner { - if (lance_scanner_set_nprobes(handle_.get(), n) != 0) check_error(); - return *this; - } -+ Scanner& query_parallelism(int32_t parallelism) { -+ if (lance_scanner_set_query_parallelism(handle_.get(), parallelism) != 0) -+ check_error(); -+ return *this; -+ } - Scanner& refine_factor(uint32_t f) { - if (lance_scanner_set_refine_factor(handle_.get(), f) != 0) check_error(); - return *this; -diff --git a/src/scanner.rs b/src/scanner.rs -index 0c29b17..ebedafc 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -60,11 +60,18 @@ pub struct LanceScanner { - limit: Option, - offset: Option, - batch_size: Option, -+ batch_size_bytes: Option, -+ io_buffer_size: Option, -+ batch_readahead: Option, -+ fragment_readahead: Option, -+ target_parallelism: Option, -+ scan_in_order: Option, - with_row_id: bool, - fragment_ids: Option>, - index_segments: Option>, - nearest: Option, - nprobes: Option, -+ query_parallelism: Option, - refine_factor: Option, - ef: Option, - metric_override: Option, -@@ -129,11 +136,18 @@ impl LanceScanner { - limit: None, - offset: None, - batch_size: None, -+ batch_size_bytes: None, -+ io_buffer_size: None, -+ batch_readahead: None, -+ fragment_readahead: None, -+ target_parallelism: None, -+ scan_in_order: None, - with_row_id: false, - fragment_ids: None, - index_segments: None, - nearest: None, - nprobes: None, -+ query_parallelism: None, - refine_factor: None, - ef: None, - metric_override: None, -@@ -164,6 +178,15 @@ impl LanceScanner { - Arc::clone(&self.poisoned) - } - -+ fn ensure_scan_not_started(&self, setting_name: &str) -> Result<()> { -+ if self.scan_started.load(Ordering::Acquire) { -+ return Err(lance_core::Error::invalid_input_source( -+ format!("{setting_name} must be set before the scan starts").into(), -+ )); -+ } -+ Ok(()) -+ } -+ - /// Apply fragment selection to a scanner builder if fragment_ids is set. - fn apply_fragment_filter(&self, scanner: &mut lance::dataset::scanner::Scanner) -> Result<()> { - if let Some(ids) = &self.fragment_ids { -@@ -231,6 +254,24 @@ impl LanceScanner { - if let Some(bs) = self.batch_size { - scanner.batch_size(bs); - } -+ if let Some(batch_size_bytes) = self.batch_size_bytes { -+ scanner.batch_size_bytes(batch_size_bytes); -+ } -+ if let Some(io_buffer_size) = self.io_buffer_size { -+ scanner.io_buffer_size(io_buffer_size); -+ } -+ if let Some(batch_readahead) = self.batch_readahead { -+ scanner.batch_readahead(batch_readahead); -+ } -+ if let Some(fragment_readahead) = self.fragment_readahead { -+ scanner.fragment_readahead(fragment_readahead); -+ } -+ if let Some(target_parallelism) = self.target_parallelism { -+ scanner.target_parallelism(target_parallelism); -+ } -+ if let Some(scan_in_order) = self.scan_in_order { -+ scanner.scan_in_order(scan_in_order); -+ } - if self.with_row_id { - scanner.with_row_id(); - } -@@ -260,6 +301,9 @@ impl LanceScanner { - if let Some(np) = self.nprobes { - scanner.nprobes(np as usize); - } -+ if let Some(query_parallelism) = self.query_parallelism { -+ scanner.query_parallelism(query_parallelism); -+ } - if let Some(rf) = self.refine_factor { - scanner.refine(rf); - } -@@ -757,6 +801,205 @@ unsafe fn scanner_set_batch_size_inner(scanner: *mut LanceScanner, batch_size: i - Ok(0) - } - -+/// Set the target output batch size in bytes. Returns 0 on success. -+/// -+/// The size must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_batch_size_bytes( -+ scanner: *mut LanceScanner, -+ batch_size_bytes: u64, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_batch_size_bytes_inner(scanner, batch_size_bytes) -+ }) -+} -+ -+unsafe fn scanner_set_batch_size_bytes_inner( -+ scanner: *mut LanceScanner, -+ batch_size_bytes: u64, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if batch_size_bytes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "batch_size_bytes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("batch_size_bytes")?; -+ scanner.batch_size_bytes = Some(batch_size_bytes); -+ Ok(0) -+} -+ -+/// Set the scanner I/O buffer size in bytes. Returns 0 on success. -+/// -+/// The size must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_io_buffer_size( -+ scanner: *mut LanceScanner, -+ io_buffer_size_bytes: u64, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_io_buffer_size_inner(scanner, io_buffer_size_bytes) -+ }) -+} -+ -+unsafe fn scanner_set_io_buffer_size_inner( -+ scanner: *mut LanceScanner, -+ io_buffer_size_bytes: u64, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if io_buffer_size_bytes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "io_buffer_size_bytes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("io_buffer_size_bytes")?; -+ scanner.io_buffer_size = Some(io_buffer_size_bytes); -+ Ok(0) -+} -+ -+/// Set the number of batches to decode concurrently. Returns 0 on success. -+/// -+/// The value must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_batch_readahead( -+ scanner: *mut LanceScanner, -+ batch_readahead: usize, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_batch_readahead_inner(scanner, batch_readahead) -+ }) -+} -+ -+unsafe fn scanner_set_batch_readahead_inner( -+ scanner: *mut LanceScanner, -+ batch_readahead: usize, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if batch_readahead == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "batch_readahead must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("batch_readahead")?; -+ scanner.batch_readahead = Some(batch_readahead); -+ Ok(0) -+} -+ -+/// Set the number of fragments to read ahead for unordered scans. -+/// -+/// The value must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_fragment_readahead( -+ scanner: *mut LanceScanner, -+ fragment_readahead: usize, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_fragment_readahead_inner(scanner, fragment_readahead) -+ }) -+} -+ -+unsafe fn scanner_set_fragment_readahead_inner( -+ scanner: *mut LanceScanner, -+ fragment_readahead: usize, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if fragment_readahead == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "fragment_readahead must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("fragment_readahead")?; -+ scanner.fragment_readahead = Some(fragment_readahead); -+ Ok(0) -+} -+ -+/// Set the target number of physical execution partitions. -+/// -+/// The value must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_target_parallelism( -+ scanner: *mut LanceScanner, -+ target_parallelism: usize, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_target_parallelism_inner(scanner, target_parallelism) -+ }) -+} -+ -+unsafe fn scanner_set_target_parallelism_inner( -+ scanner: *mut LanceScanner, -+ target_parallelism: usize, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if target_parallelism == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "target_parallelism must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("target_parallelism")?; -+ scanner.target_parallelism = Some(target_parallelism); -+ Ok(0) -+} -+ -+/// Configure whether scan results are returned in storage order. -+/// -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_scan_in_order( -+ scanner: *mut LanceScanner, -+ scan_in_order: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_scan_in_order_inner(scanner, scan_in_order) -+ }) -+} -+ -+unsafe fn scanner_set_scan_in_order_inner( -+ scanner: *mut LanceScanner, -+ scan_in_order: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("scan_in_order")?; -+ scanner.scan_in_order = Some(scan_in_order); -+ Ok(0) -+} -+ - /// Enable or disable row ID in scan output. Returns 0. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_with_row_id( -@@ -1766,6 +2009,42 @@ scanner_set_u32!(lance_scanner_set_nprobes, nprobes); - scanner_set_u32!(lance_scanner_set_refine_factor, refine_factor); - scanner_set_u32!(lance_scanner_set_ef, ef); - -+/// Set vector index partition-search concurrency for each query. -+/// -+/// `-1` uses the CPU pool size, `0` selects Lance's automatic policy, and -+/// positive values request that many workers. Values below `-1` are invalid. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_query_parallelism( -+ scanner: *mut LanceScanner, -+ query_parallelism: i32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_query_parallelism_inner(scanner, query_parallelism) -+ }) -+} -+ -+unsafe fn scanner_set_query_parallelism_inner( -+ scanner: *mut LanceScanner, -+ query_parallelism: i32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if query_parallelism < -1 { -+ return Err(lance_core::Error::invalid_input_source( -+ format!("query_parallelism must be -1, 0, or greater than 0, got {query_parallelism}") -+ .into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("query_parallelism")?; -+ scanner.query_parallelism = Some(query_parallelism); -+ Ok(0) -+} -+ - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_metric(scanner: *mut LanceScanner, metric: i32) -> i32 { - scanner_poison_check!(scanner, -1); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 8805764..42f842d 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1207,6 +1207,154 @@ fn test_scanner_batch_size() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_execution_tuning_options() { -+ let (_tmp, uri) = create_multi_fragment_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(scanner, 1024) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, 64 * 1024) }, -+ 0 -+ ); -+ assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 1) }, 0); -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(scanner, 1) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(scanner, 1) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_scan_in_order(scanner, false) }, -+ 0 -+ ); -+ -+ let mut ffi_stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut ffi_stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); -+ let total_rows: usize = reader.map(|batch| batch.unwrap().num_rows()).sum(); -+ assert_eq!(total_rows, 10); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ -+#[test] -+fn test_scanner_execution_tuning_options_reject_zero() { -+ let (_tmp, uri) = create_test_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(scanner, 0) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("batch_size_bytes must be greater than 0, got 0")); -+ -+ assert_eq!(unsafe { lance_scanner_set_io_buffer_size(scanner, 0) }, -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!( -+ take_last_error_message().contains("io_buffer_size_bytes must be greater than 0, got 0") -+ ); -+ -+ assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 0) }, -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("batch_readahead must be greater than 0, got 0")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(scanner, 0) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("fragment_readahead must be greater than 0, got 0")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(scanner, 0) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("target_parallelism must be greater than 0, got 0")); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ -+#[test] -+fn test_scanner_execution_tuning_options_reject_after_scan_start() { -+ let (_tmp, uri) = create_test_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ let mut ffi_stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut ffi_stream) }, -+ 0 -+ ); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(scanner, 1024) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("batch_size_bytes must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, 64 * 1024) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("io_buffer_size_bytes must be set before")); -+ -+ assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("batch_readahead must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(scanner, 1) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("fragment_readahead must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(scanner, 1) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("target_parallelism must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_scan_in_order(scanner, false) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("scan_in_order must be set before")); -+ -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - // --------------------------------------------------------------------------- - // Combined filter + projection + limit - // --------------------------------------------------------------------------- -@@ -1441,6 +1589,34 @@ fn test_null_safety_comprehensive() { - unsafe { lance_scanner_set_batch_size(ptr::null_mut(), 10) }, - -1 - ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(ptr::null_mut(), 1024) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(ptr::null_mut(), 64 * 1024) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_readahead(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_scan_in_order(ptr::null_mut(), true) }, -+ -1 -+ ); - assert_eq!( - unsafe { lance_scanner_with_row_id(ptr::null_mut(), true) }, - -1 -@@ -5216,6 +5392,7 @@ fn test_scanner_nearest_with_ivf_pq_index() { - 10, - ); - lance_scanner_set_nprobes(scanner, 4); -+ assert_eq!(lance_scanner_set_query_parallelism(scanner, 4), 0); - } - - let mut stream = FFI_ArrowArrayStream::empty(); -@@ -5234,6 +5411,59 @@ fn test_scanner_nearest_with_ivf_pq_index() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_query_parallelism_validation_and_lifecycle() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, -1) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, 0) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, 2) }, -+ 0 -+ ); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, -2) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!( -+ take_last_error_message() -+ .contains("query_parallelism must be -1, 0, or greater than 0, got -2") -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, 1) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("query_parallelism must be set before")); -+ -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scanner_nearest_dim_mismatch() { - let (_tmp, uri) = create_vector_dataset(64, 8); -diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index 17b1ab6..28762b8 100644 ---- a/tests/cpp/test_cpp_api.cpp -+++ b/tests/cpp/test_cpp_api.cpp -@@ -134,6 +134,12 @@ static void test_scanner_fluent(const std::string& uri) { - scanner.limit(5) - .offset(0) - .batch_size(2) -+ .batch_size_bytes(1024) -+ .io_buffer_size(64 * 1024) -+ .batch_readahead(1) -+ .fragment_readahead(1) -+ .target_parallelism(1) -+ .scan_in_order(false) - .statistics_callback(capture_scan_statistics, &captured); - - ArrowArrayStream stream; -@@ -370,6 +376,7 @@ static void test_nearest_smoke(const std::string& uri) { - try { - scanner.nearest("embedding", q, 8, 5) - .nprobes(2) -+ .query_parallelism(2) - .refine_factor(1) - .ef(50) - .metric(LANCE_METRIC_L2) - -From 043a1f7eac253d8ac6be3f970b60fcbf615295aa Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Sat, 5 Sep 2026 00:07:06 +0800 -Subject: [PATCH 2/2] add check - ---- - include/lance/lance.h | 8 ++++---- - include/lance/lance.hpp | 2 +- - src/scanner.rs | 11 ++++++++++- - tests/c_api_test.rs | 17 ++++++++++++++++- - 4 files changed, 31 insertions(+), 7 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 00f415b..541134f 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -949,12 +949,12 @@ int32_t lance_scanner_set_batch_size_bytes( - /** - * Set the scanner I/O buffer size in bytes. - * -- * The value must be greater than zero and must be set before scanning starts. -- * This bounds buffered I/O received from storage, but is not a hard limit on -- * all memory used by the scanner. -+ * The value must be between 1 and INT64_MAX, inclusive, and must be set before -+ * scanning starts. This bounds buffered I/O received from storage, but is not -+ * a hard limit on all memory used by the scanner. - * - * @param scanner Scanner handle. Must not be NULL. -- * @param io_buffer_size_bytes I/O buffer size in bytes. Must be greater than zero. -+ * @param io_buffer_size_bytes I/O buffer size in bytes, in the range [1, INT64_MAX]. - * @return 0 on success, -1 on error. - */ - int32_t lance_scanner_set_io_buffer_size( -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 3c03f86..a60dcd4 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1203,7 +1203,7 @@ class Scanner { - return *this; - } - -- /// Set the scanner I/O buffer size in bytes. -+ /// Set the scanner I/O buffer size in bytes, in the range [1, INT64_MAX]. - Scanner& io_buffer_size(uint64_t bytes) { - if (lance_scanner_set_io_buffer_size(handle_.get(), bytes) != 0) - check_error(); -diff --git a/src/scanner.rs b/src/scanner.rs -index ebedafc..6414cc0 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -837,7 +837,7 @@ unsafe fn scanner_set_batch_size_bytes_inner( - - /// Set the scanner I/O buffer size in bytes. Returns 0 on success. - /// --/// The size must be greater than zero and must be set before the scan starts. -+/// The size must be between 1 and [`i64::MAX`] and must be set before the scan starts. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_io_buffer_size( - scanner: *mut LanceScanner, -@@ -863,6 +863,15 @@ unsafe fn scanner_set_io_buffer_size_inner( - "io_buffer_size_bytes must be greater than 0, got 0".into(), - )); - } -+ if io_buffer_size_bytes > i64::MAX as u64 { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "io_buffer_size_bytes must be at most {}, got {io_buffer_size_bytes}", -+ i64::MAX -+ ) -+ .into(), -+ )); -+ } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("io_buffer_size_bytes")?; - scanner.io_buffer_size = Some(io_buffer_size_bytes); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 42f842d..ffdd915 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1252,7 +1252,7 @@ fn test_scanner_execution_tuning_options() { - } - - #[test] --fn test_scanner_execution_tuning_options_reject_zero() { -+fn test_scanner_execution_tuning_options_reject_invalid_values() { - let (_tmp, uri) = create_test_dataset(); - let c_uri = c_str(&uri); - let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -@@ -1274,6 +1274,21 @@ fn test_scanner_execution_tuning_options_reject_zero() { - take_last_error_message().contains("io_buffer_size_bytes must be greater than 0, got 0") - ); - -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, i64::MAX as u64) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, u64::MAX) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains(&format!( -+ "io_buffer_size_bytes must be at most {}, got {}", -+ i64::MAX, -+ u64::MAX -+ ))); -+ - assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 0) }, -1); - assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); - assert!(take_last_error_message().contains("batch_readahead must be greater than 0, got 0")); -From 057135cdd4ac6ac7b5348a1832caf7081a8541fb Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Mon, 7 Sep 2026 12:35:45 +0800 -Subject: [PATCH 1/2] feat: expose scanner use_scalar_index options - ---- - include/lance/lance.h | 83 +++++++++ - include/lance/lance.hpp | 47 ++++++ - src/scanner.rs | 335 +++++++++++++++++++++++++++++++++++++ - tests/c_api_test.rs | 296 ++++++++++++++++++++++++++++++++ - tests/cpp/test_cpp_api.cpp | 8 + - 5 files changed, 769 insertions(+) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 31213da..6ace2cb 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -147,6 +147,13 @@ typedef enum { - LANCE_METRIC_HAMMING = 3, - } LanceMetricType; - -+/** Speed / accuracy tradeoff for approximate vector search. */ -+typedef enum { -+ LANCE_APPROX_MODE_FAST = 0, -+ LANCE_APPROX_MODE_NORMAL = 1, -+ LANCE_APPROX_MODE_ACCURATE = 2, -+} LanceApproxMode; -+ - typedef enum { - LANCE_DTYPE_FLOAT32 = 0, - LANCE_DTYPE_FLOAT16 = 1, -@@ -1002,8 +1009,54 @@ int32_t lance_scanner_set_target_parallelism( - * they are ready. - */ - int32_t lance_scanner_set_scan_in_order(LanceScanner* scanner, bool scan_in_order); -+ -+/** -+ * Configure whether scalar indices may be used to optimize filters. -+ * -+ * Scalar indices are enabled by default. Disable this to force filter -+ * evaluation without scalar indices. This setting is independent of -+ * `lance_scanner_set_use_index`, which controls vector ANN index usage. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_use_scalar_index( -+ LanceScanner* scanner, -+ bool use_scalar_index -+); -+ -+/** -+ * Configure whether row-based output batches are strict. -+ * -+ * When enabled, every batch except the last has exactly the configured row -+ * batch size. This may require copying and cannot be combined with a byte-based -+ * batch-size limit. Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_strict_batch_size( -+ LanceScanner* scanner, -+ bool strict_batch_size -+); -+ -+/** -+ * Configure whether file statistics may optimize the scan (default: true). -+ * Intended primarily for debugging and benchmarking. Must be set before -+ * scanning starts. -+ */ -+int32_t lance_scanner_set_use_stats(LanceScanner* scanner, bool use_stats); -+ - int32_t lance_scanner_with_row_id(LanceScanner* scanner, bool enable); - -+/** Include or omit the `_rowaddr` metadata column. Must be set before scanning. */ -+int32_t lance_scanner_with_row_address(LanceScanner* scanner, bool enable); -+ -+/** -+ * Configure whether deleted rows still present in storage are returned. -+ * Deleted rows have a NULL `_rowid`; callers should also enable row IDs. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_include_deleted_rows( -+ LanceScanner* scanner, -+ bool include_deleted_rows -+); -+ - /** - * Restrict scan to the given fragment IDs. Must be called before iteration. - * @param ids Array of fragment IDs -@@ -1725,6 +1778,36 @@ int32_t lance_scanner_nearest( - - int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t n); - -+/** -+ * Set the minimum number of vector-index partitions to search. -+ * Must be greater than zero and no greater than `maximum_nprobes` when set. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_minimum_nprobes( -+ LanceScanner* scanner, -+ uint32_t minimum_nprobes -+); -+ -+/** -+ * Set the maximum number of vector-index partitions to search. -+ * Must be greater than zero and no less than `minimum_nprobes` when set. -+ * This only affects prefiltered searches that need more candidates. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_maximum_nprobes( -+ LanceScanner* scanner, -+ uint32_t maximum_nprobes -+); -+ -+/** -+ * Configure the speed / accuracy tradeoff for approximate vector search. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_approx_mode( -+ LanceScanner* scanner, -+ LanceApproxMode approx_mode -+); -+ - /** - * Set vector index partition-search concurrency for each query. - * -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 404d2df..e08f76a 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1269,6 +1269,27 @@ class Scanner { - return *this; - } - -+ /// Configure whether scalar indices may be used to optimize filters. -+ Scanner& use_scalar_index(bool enable = true) { -+ if (lance_scanner_set_use_scalar_index(handle_.get(), enable) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether row-based output batches are strict. -+ Scanner& strict_batch_size(bool strict_batch_size = true) { -+ if (lance_scanner_set_strict_batch_size(handle_.get(), strict_batch_size) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether file statistics may optimize the scan. -+ Scanner& use_stats(bool use_stats = true) { -+ if (lance_scanner_set_use_stats(handle_.get(), use_stats) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Enable/disable row ID in output. - Scanner& with_row_id(bool enable = true) { - if (lance_scanner_with_row_id(handle_.get(), enable) != 0) -@@ -1276,6 +1297,20 @@ class Scanner { - return *this; - } - -+ /// Include or omit the `_rowaddr` metadata column. -+ Scanner& with_row_address(bool enable = true) { -+ if (lance_scanner_with_row_address(handle_.get(), enable) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether deleted rows still present in storage are returned. -+ Scanner& include_deleted_rows(bool include_deleted_rows = true) { -+ if (lance_scanner_set_include_deleted_rows(handle_.get(), include_deleted_rows) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Restrict scan to specific fragment IDs. - Scanner& fragment_ids(const uint64_t* ids, size_t len) { - if (lance_scanner_set_fragment_ids(handle_.get(), ids, len) != 0) -@@ -1386,6 +1421,18 @@ class Scanner { - if (lance_scanner_set_nprobes(handle_.get(), n) != 0) check_error(); - return *this; - } -+ Scanner& minimum_nprobes(uint32_t minimum_nprobes) { -+ if (lance_scanner_set_minimum_nprobes(handle_.get(), minimum_nprobes) != 0) check_error(); -+ return *this; -+ } -+ Scanner& maximum_nprobes(uint32_t maximum_nprobes) { -+ if (lance_scanner_set_maximum_nprobes(handle_.get(), maximum_nprobes) != 0) check_error(); -+ return *this; -+ } -+ Scanner& approx_mode(LanceApproxMode approx_mode) { -+ if (lance_scanner_set_approx_mode(handle_.get(), approx_mode) != 0) check_error(); -+ return *this; -+ } - Scanner& query_parallelism(int32_t parallelism) { - if (lance_scanner_set_query_parallelism(handle_.get(), parallelism) != 0) - check_error(); -diff --git a/src/scanner.rs b/src/scanner.rs -index d3ef3be..53cd5b6 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -21,6 +21,7 @@ use lance::dataset::scanner::{ - use lance::io::exec::fts::{FlatMatchQueryExec, MatchQueryExec, PhraseQueryExec}; - use lance_core::Result; - use lance_index::scalar::FullTextSearchQuery; -+use lance_index::vector::ApproxMode; - use lance_io::stream::RecordBatchStream; - use lance_table::format::IndexMetadata; - use uuid::Uuid; -@@ -51,6 +52,38 @@ pub enum LanceDataType { - Int8 = 4, - } - -+/// Speed / accuracy tradeoff for approximate vector search, mirroring the C -+/// enum `LanceApproxMode`. -+#[repr(i32)] -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub enum LanceApproxMode { -+ Fast = 0, -+ Normal = 1, -+ Accurate = 2, -+} -+ -+impl LanceApproxMode { -+ fn from_i32(value: i32) -> Result { -+ match value { -+ 0 => Ok(Self::Fast), -+ 1 => Ok(Self::Normal), -+ 2 => Ok(Self::Accurate), -+ _ => Err(lance_core::Error::invalid_input_source( -+ format!("approx_mode must be 0 (FAST), 1 (NORMAL), or 2 (ACCURATE), got {value}") -+ .into(), -+ )), -+ } -+ } -+ -+ fn to_approx_mode(self) -> ApproxMode { -+ match self { -+ Self::Fast => ApproxMode::Fast, -+ Self::Normal => ApproxMode::Normal, -+ Self::Accurate => ApproxMode::Accurate, -+ } -+ } -+} -+ - /// Opaque scanner handle. Stores configuration until stream materialization. - pub struct LanceScanner { - dataset: Arc, -@@ -62,16 +95,24 @@ pub struct LanceScanner { - offset: Option, - batch_size: Option, - batch_size_bytes: Option, -+ strict_batch_size: Option, - io_buffer_size: Option, - batch_readahead: Option, - fragment_readahead: Option, - target_parallelism: Option, - scan_in_order: Option, -+ use_scalar_index: Option, -+ use_stats: Option, - with_row_id: bool, -+ with_row_address: bool, -+ include_deleted_rows: bool, - fragment_ids: Option>, - index_segments: Option>, - nearest: Option, - nprobes: Option, -+ minimum_nprobes: Option, -+ maximum_nprobes: Option, -+ approx_mode: Option, - query_parallelism: Option, - refine_factor: Option, - ef: Option, -@@ -138,16 +179,24 @@ impl LanceScanner { - offset: None, - batch_size: None, - batch_size_bytes: None, -+ strict_batch_size: None, - io_buffer_size: None, - batch_readahead: None, - fragment_readahead: None, - target_parallelism: None, - scan_in_order: None, -+ use_scalar_index: None, -+ use_stats: None, - with_row_id: false, -+ with_row_address: false, -+ include_deleted_rows: false, - fragment_ids: None, - index_segments: None, - nearest: None, - nprobes: None, -+ minimum_nprobes: None, -+ maximum_nprobes: None, -+ approx_mode: None, - query_parallelism: None, - refine_factor: None, - ef: None, -@@ -258,6 +307,9 @@ impl LanceScanner { - if let Some(batch_size_bytes) = self.batch_size_bytes { - scanner.batch_size_bytes(batch_size_bytes); - } -+ if let Some(strict_batch_size) = self.strict_batch_size { -+ scanner.strict_batch_size(strict_batch_size); -+ } - if let Some(io_buffer_size) = self.io_buffer_size { - scanner.io_buffer_size(io_buffer_size); - } -@@ -273,9 +325,21 @@ impl LanceScanner { - if let Some(scan_in_order) = self.scan_in_order { - scanner.scan_in_order(scan_in_order); - } -+ if let Some(use_scalar_index) = self.use_scalar_index { -+ scanner.use_scalar_index(use_scalar_index); -+ } -+ if let Some(use_stats) = self.use_stats { -+ scanner.use_stats(use_stats); -+ } - if self.with_row_id { - scanner.with_row_id(); - } -+ if self.with_row_address { -+ scanner.with_row_address(); -+ } -+ if self.include_deleted_rows { -+ scanner.include_deleted_rows(); -+ } - self.apply_fragment_filter(&mut scanner)?; - if self.index_segments.is_some() && self.nearest.is_none() { - return Err(lance_core::Error::invalid_input_source( -@@ -302,6 +366,15 @@ impl LanceScanner { - if let Some(np) = self.nprobes { - scanner.nprobes(np as usize); - } -+ if let Some(minimum_nprobes) = self.minimum_nprobes { -+ scanner.minimum_nprobes(minimum_nprobes as usize); -+ } -+ if let Some(maximum_nprobes) = self.maximum_nprobes { -+ scanner.maximum_nprobes(maximum_nprobes as usize); -+ } -+ if let Some(approx_mode) = self.approx_mode { -+ scanner.approx_mode(approx_mode.to_approx_mode()); -+ } - if let Some(query_parallelism) = self.query_parallelism { - scanner.query_parallelism(query_parallelism); - } -@@ -1035,6 +1108,92 @@ unsafe fn scanner_set_scan_in_order_inner( - Ok(0) - } - -+/// Configure whether scalar indices may be used to optimize filters. -+/// -+/// Scalar indices are enabled by default in Lance. Must be set before the scan -+/// starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_use_scalar_index( -+ scanner: *mut LanceScanner, -+ use_scalar_index: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_use_scalar_index_inner(scanner, use_scalar_index) -+ }) -+} -+ -+unsafe fn scanner_set_use_scalar_index_inner( -+ scanner: *mut LanceScanner, -+ use_scalar_index: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("use_scalar_index")?; -+ scanner.use_scalar_index = Some(use_scalar_index); -+ Ok(0) -+} -+ -+/// Configure whether output batches use the exact row-based batch size. -+/// -+/// Must be set before the scan starts. Lance rejects enabling this together -+/// with a byte-based batch-size limit when the scan is materialized. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_strict_batch_size( -+ scanner: *mut LanceScanner, -+ strict_batch_size: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_strict_batch_size_inner(scanner, strict_batch_size) -+ }) -+} -+ -+unsafe fn scanner_set_strict_batch_size_inner( -+ scanner: *mut LanceScanner, -+ strict_batch_size: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("strict_batch_size")?; -+ scanner.strict_batch_size = Some(strict_batch_size); -+ Ok(0) -+} -+ -+/// Configure whether file statistics may be used to optimize the scan. -+/// -+/// Statistics are enabled by default. Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_use_stats( -+ scanner: *mut LanceScanner, -+ use_stats: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_use_stats_inner(scanner, use_stats) -+ }) -+} -+ -+unsafe fn scanner_set_use_stats_inner(scanner: *mut LanceScanner, use_stats: bool) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("use_stats")?; -+ scanner.use_stats = Some(use_stats); -+ Ok(0) -+} -+ - /// Enable or disable row ID in scan output. Returns 0. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_with_row_id( -@@ -1058,6 +1217,62 @@ unsafe fn scanner_with_row_id_inner(scanner: *mut LanceScanner, enable: bool) -> - Ok(0) - } - -+/// Enable or disable the `_rowaddr` metadata column in scan output. -+/// -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_with_row_address( -+ scanner: *mut LanceScanner, -+ enable: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_with_row_address_inner(scanner, enable) -+ }) -+} -+ -+unsafe fn scanner_with_row_address_inner(scanner: *mut LanceScanner, enable: bool) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("with_row_address")?; -+ scanner.with_row_address = enable; -+ Ok(0) -+} -+ -+/// Configure whether deleted rows still present in storage are returned. -+/// -+/// Deleted rows have a NULL `_rowid`, so callers should also enable row IDs. -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_include_deleted_rows( -+ scanner: *mut LanceScanner, -+ include_deleted_rows: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_include_deleted_rows_inner(scanner, include_deleted_rows) -+ }) -+} -+ -+unsafe fn scanner_set_include_deleted_rows_inner( -+ scanner: *mut LanceScanner, -+ include_deleted_rows: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("include_deleted_rows")?; -+ scanner.include_deleted_rows = include_deleted_rows; -+ Ok(0) -+} -+ - /// Restrict the scan to the given fragment IDs. - /// Must be called before any iteration method. - /// -@@ -2044,6 +2259,126 @@ scanner_set_u32!(lance_scanner_set_nprobes, nprobes); - scanner_set_u32!(lance_scanner_set_refine_factor, refine_factor); - scanner_set_u32!(lance_scanner_set_ef, ef); - -+/// Set the minimum number of vector-index partitions to search. -+/// -+/// The value must be greater than zero, no greater than a configured -+/// `maximum_nprobes`, and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_minimum_nprobes( -+ scanner: *mut LanceScanner, -+ minimum_nprobes: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_minimum_nprobes_inner(scanner, minimum_nprobes) -+ }) -+} -+ -+unsafe fn scanner_set_minimum_nprobes_inner( -+ scanner: *mut LanceScanner, -+ minimum_nprobes: u32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if minimum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "minimum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("minimum_nprobes")?; -+ if let Some(maximum_nprobes) = scanner.maximum_nprobes -+ && minimum_nprobes > maximum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "minimum_nprobes ({minimum_nprobes}) must not exceed maximum_nprobes ({maximum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ scanner.minimum_nprobes = Some(minimum_nprobes); -+ Ok(0) -+} -+ -+/// Set the maximum number of vector-index partitions to search. -+/// -+/// The value must be greater than zero, no less than a configured -+/// `minimum_nprobes`, and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_maximum_nprobes( -+ scanner: *mut LanceScanner, -+ maximum_nprobes: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_maximum_nprobes_inner(scanner, maximum_nprobes) -+ }) -+} -+ -+unsafe fn scanner_set_maximum_nprobes_inner( -+ scanner: *mut LanceScanner, -+ maximum_nprobes: u32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if maximum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "maximum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("maximum_nprobes")?; -+ if let Some(minimum_nprobes) = scanner.minimum_nprobes -+ && maximum_nprobes < minimum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "maximum_nprobes ({maximum_nprobes}) must not be less than minimum_nprobes ({minimum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ scanner.maximum_nprobes = Some(maximum_nprobes); -+ Ok(0) -+} -+ -+/// Configure the speed / accuracy tradeoff for approximate vector search. -+/// -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_approx_mode( -+ scanner: *mut LanceScanner, -+ approx_mode: i32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_approx_mode_inner(scanner, approx_mode) -+ }) -+} -+ -+unsafe fn scanner_set_approx_mode_inner( -+ scanner: *mut LanceScanner, -+ approx_mode: i32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let approx_mode = LanceApproxMode::from_i32(approx_mode)?; -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("approx_mode")?; -+ scanner.approx_mode = Some(approx_mode); -+ Ok(0) -+} -+ - /// Set vector index partition-search concurrency for each query. - /// - /// `-1` uses the CPU pool size, `0` selects Lance's automatic policy, and -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index bde742d..a550a93 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1237,6 +1237,12 @@ fn test_scanner_execution_tuning_options() { - unsafe { lance_scanner_set_scan_in_order(scanner, false) }, - 0 - ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(scanner, false) }, -+ 0 -+ ); -+ assert_eq!(unsafe { lance_scanner_set_use_stats(scanner, false) }, 0); -+ assert_eq!(unsafe { lance_scanner_with_row_address(scanner, true) }, 0); - - let mut ffi_stream = FFI_ArrowArrayStream::empty(); - assert_eq!( -@@ -1244,6 +1250,7 @@ fn test_scanner_execution_tuning_options() { - 0 - ); - let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); -+ assert!(reader.schema().field_with_name("_rowaddr").is_ok()); - let total_rows: usize = reader.map(|batch| batch.unwrap().num_rows()).sum(); - assert_eq!(total_rows, 10); - -@@ -1360,6 +1367,30 @@ fn test_scanner_execution_tuning_options_reject_after_scan_start() { - ); - assert!(take_last_error_message().contains("scan_in_order must be set before")); - -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(scanner, false) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("use_scalar_index must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(scanner, true) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("strict_batch_size must be set before")); -+ -+ assert_eq!(unsafe { lance_scanner_set_use_stats(scanner, false) }, -1); -+ assert!(take_last_error_message().contains("use_stats must be set before")); -+ -+ assert_eq!(unsafe { lance_scanner_with_row_address(scanner, true) }, -1); -+ assert!(take_last_error_message().contains("with_row_address must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_include_deleted_rows(scanner, true) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("include_deleted_rows must be set before")); -+ - let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); - assert_eq!( - reader.map(|batch| batch.unwrap().num_rows()).sum::(), -@@ -1370,6 +1401,79 @@ fn test_scanner_execution_tuning_options_reject_after_scan_start() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_strict_batch_size_across_fragments() { -+ let (_tmp, uri) = create_multi_fragment_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!(unsafe { lance_scanner_set_batch_size(scanner, 3) }, 0); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(scanner, true) }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ let batch_sizes = reader -+ .map(|batch| batch.unwrap().num_rows()) -+ .collect::>(); -+ assert_eq!(batch_sizes, vec![3, 3, 3, 1]); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ -+#[test] -+fn test_scanner_include_deleted_rows() { -+ let (_tmp, uri) = create_multi_fragment_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let predicate = c_str("id >= 8"); -+ let mut num_deleted = 0; -+ assert_eq!( -+ unsafe { lance_dataset_delete(ds, predicate.as_ptr(), &mut num_deleted) }, -+ 0 -+ ); -+ assert_eq!(num_deleted, 2); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!(unsafe { lance_scanner_with_row_id(scanner, true) }, 0); -+ assert_eq!( -+ unsafe { lance_scanner_set_include_deleted_rows(scanner, true) }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ let batches = reader.map(|batch| batch.unwrap()).collect::>(); -+ assert_eq!(batches.iter().map(RecordBatch::num_rows).sum::(), 10); -+ assert_eq!( -+ batches -+ .iter() -+ .map(|batch| batch.column_by_name("_rowid").unwrap().null_count()) -+ .sum::(), -+ 2 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - // --------------------------------------------------------------------------- - // Combined filter + projection + limit - // --------------------------------------------------------------------------- -@@ -1632,10 +1736,42 @@ fn test_null_safety_comprehensive() { - unsafe { lance_scanner_set_scan_in_order(ptr::null_mut(), true) }, - -1 - ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(ptr::null_mut(), false) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(ptr::null_mut(), true) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_stats(ptr::null_mut(), false) }, -+ -1 -+ ); - assert_eq!( - unsafe { lance_scanner_with_row_id(ptr::null_mut(), true) }, - -1 - ); -+ assert_eq!( -+ unsafe { lance_scanner_with_row_address(ptr::null_mut(), true) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_include_deleted_rows(ptr::null_mut(), true) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_minimum_nprobes(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_maximum_nprobes(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_approx_mode(ptr::null_mut(), LanceApproxMode::Normal as i32,) }, -+ -1 -+ ); - - // Scanner iteration with NULL. - let mut ffi_stream2 = FFI_ArrowArrayStream::empty(); -@@ -3077,6 +3213,90 @@ fn test_create_scalar_index_btree() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_set_use_scalar_index_controls_filter_planning() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let column = c_str("id"); -+ assert_eq!( -+ unsafe { -+ lance_dataset_create_scalar_index( -+ ds, -+ column.as_ptr(), -+ ptr::null(), -+ LanceScalarIndexType::BTree as i32, -+ ptr::null(), -+ false, -+ ) -+ }, -+ 0 -+ ); -+ -+ let run_scan = |use_scalar_index: bool| { -+ let filter = c_str("id = 3"); -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), filter.as_ptr()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(scanner, use_scalar_index) }, -+ 0 -+ ); -+ -+ let mut captured = CapturedScanStatistics::default(); -+ assert_eq!( -+ unsafe { -+ lance_scanner_set_statistics_callback( -+ scanner, -+ Some(capture_scan_statistics), -+ (&mut captured as *mut CapturedScanStatistics).cast(), -+ ) -+ }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ let ids = reader -+ .flat_map(|batch| { -+ let batch = batch.unwrap(); -+ batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .to_vec() -+ }) -+ .collect::>(); -+ -+ assert_eq!(captured.calls, 1); -+ unsafe { lance_scanner_close(scanner) }; -+ (ids, captured) -+ }; -+ -+ let (indexed_ids, indexed_statistics) = run_scan(true); -+ let (unindexed_ids, unindexed_statistics) = run_scan(false); -+ assert_eq!(indexed_ids, vec![3]); -+ assert_eq!(unindexed_ids, indexed_ids); -+ assert!( -+ indexed_statistics.indices_loaded > 0, -+ "enabled scan should load the scalar index" -+ ); -+ assert_eq!( -+ unindexed_statistics.indices_loaded, 0, -+ "disabled scan should bypass the scalar index" -+ ); -+ -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scalar_index_segment_build_is_fragment_scoped_and_uncommitted() { - let (_tmp, uri) = create_many_small_fragments(2); -@@ -5409,6 +5629,12 @@ fn test_scanner_nearest_with_ivf_pq_index() { - 10, - ); - lance_scanner_set_nprobes(scanner, 4); -+ assert_eq!(lance_scanner_set_minimum_nprobes(scanner, 2), 0); -+ assert_eq!(lance_scanner_set_maximum_nprobes(scanner, 6), 0); -+ assert_eq!( -+ lance_scanner_set_approx_mode(scanner, LanceApproxMode::Accurate as i32), -+ 0 -+ ); - assert_eq!(lance_scanner_set_query_parallelism(scanner, 4), 0); - } - -@@ -5428,6 +5654,76 @@ fn test_scanner_nearest_with_ivf_pq_index() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_adaptive_nprobes_and_approx_mode_validation_and_lifecycle() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 0) }, -1); -+ assert!(take_last_error_message().contains("minimum_nprobes must be greater than 0, got 0")); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 0) }, -1); -+ assert!(take_last_error_message().contains("maximum_nprobes must be greater than 0, got 0")); -+ assert_eq!(unsafe { lance_scanner_set_approx_mode(scanner, 3) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("approx_mode must be 0 (FAST), 1 (NORMAL), or 2 (ACCURATE), got 3") -+ ); -+ -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 2) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 3) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("minimum_nprobes (3) must not exceed maximum_nprobes (2)") -+ ); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 1) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 2) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("minimum_nprobes (2) must not exceed maximum_nprobes (1)") -+ ); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 2) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 2) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("maximum_nprobes (1) must not be less than minimum_nprobes (2)") -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_approx_mode(scanner, LanceApproxMode::Fast as i32) }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("minimum_nprobes must be set before")); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("maximum_nprobes must be set before")); -+ assert_eq!( -+ unsafe { lance_scanner_set_approx_mode(scanner, LanceApproxMode::Normal as i32) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("approx_mode must be set before")); -+ -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scanner_query_parallelism_validation_and_lifecycle() { - let (_tmp, uri) = create_test_dataset(); -diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index 28762b8..60b8fc4 100644 ---- a/tests/cpp/test_cpp_api.cpp -+++ b/tests/cpp/test_cpp_api.cpp -@@ -140,6 +140,11 @@ static void test_scanner_fluent(const std::string& uri) { - .fragment_readahead(1) - .target_parallelism(1) - .scan_in_order(false) -+ .use_scalar_index(false) -+ .strict_batch_size(false) -+ .use_stats(false) -+ .with_row_address(true) -+ .include_deleted_rows(false) - .statistics_callback(capture_scan_statistics, &captured); - - ArrowArrayStream stream; -@@ -376,6 +381,9 @@ static void test_nearest_smoke(const std::string& uri) { - try { - scanner.nearest("embedding", q, 8, 5) - .nprobes(2) -+ .minimum_nprobes(1) -+ .maximum_nprobes(2) -+ .approx_mode(LANCE_APPROX_MODE_NORMAL) - .query_parallelism(2) - .refine_factor(1) - .ef(50) - -From e894f591aef358cd36fdbf915c4d5b95fd0e8348 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Mon, 7 Sep 2026 14:27:10 +0800 -Subject: [PATCH 2/2] update - ---- - include/lance/lance.h | 18 +++- - include/lance/lance.hpp | 7 +- - src/scanner.rs | 210 +++++++++++++++++++++++++++++++--------- - tests/c_api_test.rs | 77 +++++++++++++++ - 4 files changed, 261 insertions(+), 51 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 6ace2cb..8173ae5 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -946,7 +946,8 @@ int32_t lance_scanner_set_batch_size(LanceScanner* scanner, int64_t batch_size); - * Set the target output batch size in bytes. - * - * When set, this takes precedence over the row-based batch size. The value -- * must be greater than zero and must be set before scanning starts. -+ * must be greater than zero and must be set before scanning starts. The call -+ * is rejected without changing scanner state if strict batch sizing is enabled. - */ - int32_t lance_scanner_set_batch_size_bytes( - LanceScanner* scanner, -@@ -1028,7 +1029,8 @@ int32_t lance_scanner_set_use_scalar_index( - * - * When enabled, every batch except the last has exactly the configured row - * batch size. This may require copying and cannot be combined with a byte-based -- * batch-size limit. Must be set before scanning starts. -+ * batch-size limit. The call is rejected without changing scanner state if a -+ * byte limit is already set. Must be set before scanning starts. - */ - int32_t lance_scanner_set_strict_batch_size( - LanceScanner* scanner, -@@ -1776,11 +1778,19 @@ int32_t lance_scanner_nearest( - uint32_t k - ); - --int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t n); -+/** -+ * Set both the minimum and maximum vector-index partition-search bounds. -+ * -+ * This replaces both bounds configured by earlier calls to any nprobes -+ * setter. The value must be greater than zero. Must be set before scanning. -+ */ -+int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t nprobes); - - /** - * Set the minimum number of vector-index partitions to search. -+ * This replaces only the minimum bound; the current maximum is preserved. - * Must be greater than zero and no greater than `maximum_nprobes` when set. -+ * An invalid resulting range is rejected without changing either bound. - * Must be set before scanning starts. - */ - int32_t lance_scanner_set_minimum_nprobes( -@@ -1790,8 +1800,10 @@ int32_t lance_scanner_set_minimum_nprobes( - - /** - * Set the maximum number of vector-index partitions to search. -+ * This replaces only the maximum bound; the current minimum is preserved. - * Must be greater than zero and no less than `minimum_nprobes` when set. - * This only affects prefiltered searches that need more candidates. -+ * An invalid resulting range is rejected without changing either bound. - * Must be set before scanning starts. - */ - int32_t lance_scanner_set_maximum_nprobes( -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index e08f76a..c12c0c6 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1417,14 +1417,17 @@ class Scanner { - return *this; - } - -- Scanner& nprobes(uint32_t n) { -- if (lance_scanner_set_nprobes(handle_.get(), n) != 0) check_error(); -+ /// Replace both minimum and maximum partition-search bounds. -+ Scanner& nprobes(uint32_t nprobes) { -+ if (lance_scanner_set_nprobes(handle_.get(), nprobes) != 0) check_error(); - return *this; - } -+ /// Replace only the minimum partition-search bound. - Scanner& minimum_nprobes(uint32_t minimum_nprobes) { - if (lance_scanner_set_minimum_nprobes(handle_.get(), minimum_nprobes) != 0) check_error(); - return *this; - } -+ /// Replace only the maximum partition-search bound. - Scanner& maximum_nprobes(uint32_t maximum_nprobes) { - if (lance_scanner_set_maximum_nprobes(handle_.get(), maximum_nprobes) != 0) check_error(); - return *this; -diff --git a/src/scanner.rs b/src/scanner.rs -index 53cd5b6..4ceeb0e 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -109,9 +109,7 @@ pub struct LanceScanner { - fragment_ids: Option>, - index_segments: Option>, - nearest: Option, -- nprobes: Option, -- minimum_nprobes: Option, -- maximum_nprobes: Option, -+ nprobes: NprobesRange, - approx_mode: Option, - query_parallelism: Option, - refine_factor: Option, -@@ -148,6 +146,72 @@ struct NearestQuery { - k: u32, - } - -+/// The effective adaptive partition-search range shared by all three nprobes -+/// setters. Updates are computed and validated before replacing this state. -+#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -+struct NprobesRange { -+ minimum: Option, -+ maximum: Option, -+} -+ -+impl NprobesRange { -+ fn exact(nprobes: u32) -> Result { -+ if nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ Ok(Self { -+ minimum: Some(nprobes), -+ maximum: Some(nprobes), -+ }) -+ } -+ -+ fn with_minimum(self, minimum_nprobes: u32) -> Result { -+ if minimum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "minimum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ if let Some(maximum_nprobes) = self.maximum -+ && minimum_nprobes > maximum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "minimum_nprobes ({minimum_nprobes}) must not exceed maximum_nprobes ({maximum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ Ok(Self { -+ minimum: Some(minimum_nprobes), -+ ..self -+ }) -+ } -+ -+ fn with_maximum(self, maximum_nprobes: u32) -> Result { -+ if maximum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "maximum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ if let Some(minimum_nprobes) = self.minimum -+ && maximum_nprobes < minimum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "maximum_nprobes ({maximum_nprobes}) must not be less than minimum_nprobes ({minimum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ Ok(Self { -+ maximum: Some(maximum_nprobes), -+ ..self -+ }) -+ } -+} -+ - /// Poll status for `lance_scanner_poll_next`. - #[repr(C)] - #[derive(Debug, PartialEq, Eq)] -@@ -193,9 +257,7 @@ impl LanceScanner { - fragment_ids: None, - index_segments: None, - nearest: None, -- nprobes: None, -- minimum_nprobes: None, -- maximum_nprobes: None, -+ nprobes: NprobesRange::default(), - approx_mode: None, - query_parallelism: None, - refine_factor: None, -@@ -363,13 +425,10 @@ impl LanceScanner { - } - if let Some(n) = &self.nearest { - scanner.nearest(&n.column, n.query.as_ref(), n.k as usize)?; -- if let Some(np) = self.nprobes { -- scanner.nprobes(np as usize); -- } -- if let Some(minimum_nprobes) = self.minimum_nprobes { -+ if let Some(minimum_nprobes) = self.nprobes.minimum { - scanner.minimum_nprobes(minimum_nprobes as usize); - } -- if let Some(maximum_nprobes) = self.maximum_nprobes { -+ if let Some(maximum_nprobes) = self.nprobes.maximum { - scanner.maximum_nprobes(maximum_nprobes as usize); - } - if let Some(approx_mode) = self.approx_mode { -@@ -930,6 +989,14 @@ unsafe fn scanner_set_batch_size_bytes_inner( - } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("batch_size_bytes")?; -+ if scanner.strict_batch_size == Some(true) { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "strict_batch_size=true cannot be combined with batch_size_bytes={batch_size_bytes}" -+ ) -+ .into(), -+ )); -+ } - scanner.batch_size_bytes = Some(batch_size_bytes); - Ok(0) - } -@@ -1140,8 +1207,8 @@ unsafe fn scanner_set_use_scalar_index_inner( - - /// Configure whether output batches use the exact row-based batch size. - /// --/// Must be set before the scan starts. Lance rejects enabling this together --/// with a byte-based batch-size limit when the scan is materialized. -+/// Must be set before the scan starts. Enabling this together with a -+/// byte-based batch-size limit is rejected without changing scanner state. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_strict_batch_size( - scanner: *mut LanceScanner, -@@ -1164,6 +1231,14 @@ unsafe fn scanner_set_strict_batch_size_inner( - } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("strict_batch_size")?; -+ if strict_batch_size && let Some(batch_size_bytes) = scanner.batch_size_bytes { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "strict_batch_size=true cannot be combined with batch_size_bytes={batch_size_bytes}" -+ ) -+ .into(), -+ )); -+ } - scanner.strict_batch_size = Some(strict_batch_size); - Ok(0) - } -@@ -2255,10 +2330,38 @@ macro_rules! scanner_set_u32 { - }; - } - --scanner_set_u32!(lance_scanner_set_nprobes, nprobes); - scanner_set_u32!(lance_scanner_set_refine_factor, refine_factor); - scanner_set_u32!(lance_scanner_set_ef, ef); - -+/// Set both vector-index partition-search bounds to the same value. -+/// -+/// This replaces any values previously configured through -+/// `minimum_nprobes` or `maximum_nprobes`. The value must be greater than zero -+/// and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_nprobes( -+ scanner: *mut LanceScanner, -+ nprobes: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_nprobes_inner(scanner, nprobes) -+ }) -+} -+ -+unsafe fn scanner_set_nprobes_inner(scanner: *mut LanceScanner, nprobes: u32) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("nprobes")?; -+ let next = NprobesRange::exact(nprobes)?; -+ scanner.nprobes = next; -+ Ok(0) -+} -+ - /// Set the minimum number of vector-index partitions to search. - /// - /// The value must be greater than zero, no greater than a configured -@@ -2283,24 +2386,10 @@ unsafe fn scanner_set_minimum_nprobes_inner( - "scanner is NULL".into(), - )); - } -- if minimum_nprobes == 0 { -- return Err(lance_core::Error::invalid_input_source( -- "minimum_nprobes must be greater than 0, got 0".into(), -- )); -- } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("minimum_nprobes")?; -- if let Some(maximum_nprobes) = scanner.maximum_nprobes -- && minimum_nprobes > maximum_nprobes -- { -- return Err(lance_core::Error::invalid_input_source( -- format!( -- "minimum_nprobes ({minimum_nprobes}) must not exceed maximum_nprobes ({maximum_nprobes})" -- ) -- .into(), -- )); -- } -- scanner.minimum_nprobes = Some(minimum_nprobes); -+ let next = scanner.nprobes.with_minimum(minimum_nprobes)?; -+ scanner.nprobes = next; - Ok(0) - } - -@@ -2328,24 +2417,10 @@ unsafe fn scanner_set_maximum_nprobes_inner( - "scanner is NULL".into(), - )); - } -- if maximum_nprobes == 0 { -- return Err(lance_core::Error::invalid_input_source( -- "maximum_nprobes must be greater than 0, got 0".into(), -- )); -- } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("maximum_nprobes")?; -- if let Some(minimum_nprobes) = scanner.minimum_nprobes -- && maximum_nprobes < minimum_nprobes -- { -- return Err(lance_core::Error::invalid_input_source( -- format!( -- "maximum_nprobes ({maximum_nprobes}) must not be less than minimum_nprobes ({minimum_nprobes})" -- ) -- .into(), -- )); -- } -- scanner.maximum_nprobes = Some(maximum_nprobes); -+ let next = scanner.nprobes.with_maximum(maximum_nprobes)?; -+ scanner.nprobes = next; - Ok(0) - } - -@@ -2885,6 +2960,49 @@ mod tests { - ) - } - -+ #[test] -+ fn nprobes_setters_share_one_validated_range() { -+ let (_tmp, uri) = create_test_dataset(); -+ let (dataset, scanner) = open_dataset_and_scanner(&uri); -+ let assert_range = |minimum, maximum| { -+ assert_eq!( -+ unsafe { &*scanner }.nprobes, -+ NprobesRange { minimum, maximum } -+ ); -+ }; -+ -+ assert_range(None, None); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 2) }, 0); -+ assert_range(Some(2), None); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 5) }, 0); -+ assert_range(Some(2), Some(5)); -+ -+ // The combined setter replaces both bounds. -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 4) }, 0); -+ assert_range(Some(4), Some(4)); -+ -+ // A failed partial update leaves both bounds unchanged. -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 5) }, -1); -+ assert_range(Some(4), Some(4)); -+ -+ // Widening the maximum first makes the new minimum valid. -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 6) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 5) }, 0); -+ assert_range(Some(5), Some(6)); -+ -+ // A later combined call deterministically replaces the widened range. -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 3) }, 0); -+ assert_range(Some(3), Some(3)); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 2) }, -1); -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 0) }, -1); -+ assert_range(Some(3), Some(3)); -+ -+ unsafe { -+ lance_scanner_close(scanner); -+ lance_dataset_close(dataset); -+ } -+ } -+ - #[test] - fn prepared_fts_index_only_plan_does_not_scan_indexed_fragment_row_ids() { - let (_tmp, uri) = create_test_dataset(); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index a550a93..3b3424b 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1318,6 +1318,78 @@ fn test_scanner_execution_tuning_options_reject_invalid_values() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_strict_batch_size_and_bytes_conflict_is_recoverable() { -+ let (_tmp, uri) = create_test_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let consume = |scanner| { -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ }; -+ -+ // A byte limit already exists: strict=true is rejected without starting -+ // the scan or replacing the prior strict setting. -+ let bytes_first = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(bytes_first, 1024) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(bytes_first, true) }, -+ -1 -+ ); -+ assert!( -+ take_last_error_message() -+ .contains("strict_batch_size=true cannot be combined with batch_size_bytes=1024") -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_stats(bytes_first, false) }, -+ 0, -+ "the rejected setter must not mark the scan as started" -+ ); -+ consume(bytes_first); -+ -+ // Strict sizing already exists: the byte limit is rejected without -+ // mutation. The caller can disable strict sizing and retry on this handle. -+ let strict_first = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(strict_first, true) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(strict_first, 1024) }, -+ -1 -+ ); -+ assert!( -+ take_last_error_message() -+ .contains("strict_batch_size=true cannot be combined with batch_size_bytes=1024") -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(strict_first, false) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(strict_first, 1024) }, -+ 0 -+ ); -+ consume(strict_first); -+ -+ unsafe { lance_scanner_close(bytes_first) }; -+ unsafe { lance_scanner_close(strict_first) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scanner_execution_tuning_options_reject_after_scan_start() { - let (_tmp, uri) = create_test_dataset(); -@@ -1760,6 +1832,7 @@ fn test_null_safety_comprehensive() { - unsafe { lance_scanner_set_include_deleted_rows(ptr::null_mut(), true) }, - -1 - ); -+ assert_eq!(unsafe { lance_scanner_set_nprobes(ptr::null_mut(), 1) }, -1); - assert_eq!( - unsafe { lance_scanner_set_minimum_nprobes(ptr::null_mut(), 1) }, - -1 -@@ -5663,6 +5736,8 @@ fn test_scanner_adaptive_nprobes_and_approx_mode_validation_and_lifecycle() { - let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; - assert!(!scanner.is_null()); - -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 0) }, -1); -+ assert!(take_last_error_message().contains("nprobes must be greater than 0, got 0")); - assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 0) }, -1); - assert!(take_last_error_message().contains("minimum_nprobes must be greater than 0, got 0")); - assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 0) }, -1); -@@ -5704,6 +5779,8 @@ fn test_scanner_adaptive_nprobes_and_approx_mode_validation_and_lifecycle() { - 0 - ); - -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("nprobes must be set before")); - assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 1) }, -1); - assert!(take_last_error_message().contains("minimum_nprobes must be set before")); - assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, -1); diff --git a/thirdparty/patches/lance-c-0.1.9-pr-77.patch b/thirdparty/patches/lance-c-0.1.9-pr-77.patch deleted file mode 100644 index 341a1c692dcebd..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-77.patch +++ /dev/null @@ -1,1863 +0,0 @@ -From eaf06c0374e62de6b519f55efe17de0c58e88c0a Mon Sep 17 00:00:00 2001 -From: "jianjian.xie" -Date: Fri, 4 Sep 2026 22:04:55 -0700 -Subject: [PATCH] build: bump lance to v11.0.0 - -Summary: -Intent: -- Move the lance git pins from e934cc2c to ab6b5bbe (lance v11.0.0 release tag) - so lance-c tracks a released upstream version instead of an arbitrary commit. -- Pick up the v11 blob APIs (read_blob_ranges, Option-based take_blobs results) - needed to answer #76 without a second pin bump. - -Changes: -- Point all lance, lance-core, lance-file, lance-index, lance-io, lance-linalg, - lance-table, lance-datafusion, and lance-datagen dependencies at ab6b5bbe. -- Re-resolve Cargo.lock; blake3, jiff, and reqwest 0.13 were unlocked explicitly - because v11 raised their minimum versions, the rest follows from lance v11 - (opendal 0.58, lance-namespace-reqwest-client 0.11, etc.). -- Adapt to upstream signature changes: build_global_bm25_scorer takes an optional - metrics collector, MatchQueryExec/PhraseQueryExec::new_with_segments are now - fallible, DataFile::new takes a ConcreteFileVersion, and pb::IndexMetadata - gained covering_fields. -- Keep the DOT PQ strict-subset guard; upstream make_global_pq is unchanged at - v11, so only the referenced revision in the comment and error text moved. - -Test Plan: -- cargo fmt, cargo check --all-targets, cargo clippy --all-targets -D warnings. -- cargo test: 367 passed, 0 failed, 2 ignored. -- cargo test --test compile_and_run_test -- --ignored: 2 passed (C and C++ - compile-and-run against the rebuilt library). - -Co-Authored-By: Claude Fable 5.1 ---- - Cargo.lock | 653 ++++++++++++++++++++++++------------------- - Cargo.toml | 22 +- - src/fts_query.rs | 4 +- - src/index_segment.rs | 4 +- - src/scanner.rs | 4 +- - tests/c_api_test.rs | 8 +- - 6 files changed, 391 insertions(+), 304 deletions(-) - -diff --git a/Cargo.lock b/Cargo.lock -index 60c1caf..bc37cb9 100644 ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -222,7 +222,7 @@ dependencies = [ - "arrow-schema", - "arrow-select", - "atoi", -- "base64", -+ "base64 0.22.1", - "chrono", - "comfy-table", - "half", -@@ -332,7 +332,7 @@ version = "58.3.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "f633dbfdf39c039ada1bf9e34c694816eb71fbb7dc78f613993b7245e078a1ed" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "serde_core", - "serde_json", - ] -@@ -434,6 +434,16 @@ dependencies = [ - "loom", - ] - -+[[package]] -+name = "asyncband" -+version = "0.6.7" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "94a214ba60d6231afd0e805e3c27c45a1626d9debaa5a5061c45a1ea1b2f1ed0" -+dependencies = [ -+ "hashbrown 0.17.1", -+ "slab", -+] -+ - [[package]] - name = "atoi" - version = "2.0.0" -@@ -504,7 +514,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "a054912289d18629dc78375ba2c3726a3afe3ff71b4edba9dedfca0e3446d1fc" - dependencies = [ - "aws-lc-sys", -- "untrusted 0.7.1", - "zeroize", - ] - -@@ -631,7 +640,7 @@ dependencies = [ - "bytes", - "form_urlencoded", - "hex", -- "hmac", -+ "hmac 0.12.1", - "http 0.2.12", - "http 1.4.0", - "percent-encoding", -@@ -829,6 +838,12 @@ version = "0.22.1" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" - -+[[package]] -+name = "base64" -+version = "0.23.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" -+ - [[package]] - name = "base64-simd" - version = "0.8.0" -@@ -858,6 +873,12 @@ dependencies = [ - "num-traits", - ] - -+[[package]] -+name = "bitflags" -+version = "1.3.2" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" -+ - [[package]] - name = "bitflags" - version = "2.11.0" -@@ -887,16 +908,15 @@ dependencies = [ - - [[package]] - name = "blake3" --version = "1.8.3" -+version = "1.8.7" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "2468ef7d57b3fb7e16b576e8377cdbde2320c60e1491e961d11da40fc4f02a2d" -+checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" - dependencies = [ -- "arrayref", - "arrayvec", - "cc", - "cfg-if 1.0.4", - "constant_time_eq", -- "cpufeatures 0.2.17", -+ "cpufeatures 0.3.0", - ] - - [[package]] -@@ -1127,6 +1147,12 @@ dependencies = [ - "cc", - ] - -+[[package]] -+name = "cmov" -+version = "0.5.4" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "0c9ea0ac24bc397ab3c98583a3c9ba74fa56b09a4449bbe172b9b1ddb016027a" -+ - [[package]] - name = "colorchoice" - version = "1.0.5" -@@ -1295,12 +1321,13 @@ dependencies = [ - ] - - [[package]] --name = "crc32c" --version = "0.6.8" -+name = "crc-fast" -+version = "1.10.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "3a47af21622d091a8f0fb295b88bc886ac74efcc613efc19f5d0b21de5c89e47" -+checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" - dependencies = [ -- "rustc_version", -+ "digest 0.10.7", -+ "spin 0.10.1", - ] - - [[package]] -@@ -1437,6 +1464,15 @@ version = "0.0.7" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "52560adf09603e58c9a7ee1fe1dcb95a16927b17c127f0ac02d6e768a0e25bc1" - -+[[package]] -+name = "ctutils" -+version = "0.4.2" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "7d5515a3834141de9eafb9717ad39eea8247b5674e6066c404e8c4b365d2a29e" -+dependencies = [ -+ "cmov", -+] -+ - [[package]] - name = "darling" - version = "0.23.0" -@@ -1785,7 +1821,7 @@ checksum = "5f64c983bbbdcb729d921a2b2ac3375598719b5cc0c30345ad664936f3176fc7" - dependencies = [ - "arrow", - "arrow-buffer", -- "base64", -+ "base64 0.22.1", - "blake2", - "blake3", - "chrono", -@@ -2112,6 +2148,37 @@ dependencies = [ - "url", - ] - -+[[package]] -+name = "defmt" -+version = "1.1.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" -+dependencies = [ -+ "bitflags 1.3.2", -+ "defmt-macros", -+] -+ -+[[package]] -+name = "defmt-macros" -+version = "1.1.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" -+dependencies = [ -+ "defmt-parser", -+ "proc-macro2", -+ "quote", -+ "syn 2.0.117", -+] -+ -+[[package]] -+name = "defmt-parser" -+version = "1.0.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" -+dependencies = [ -+ "thiserror 2.0.18", -+] -+ - [[package]] - name = "der" - version = "0.7.10" -@@ -2154,6 +2221,7 @@ dependencies = [ - "block-buffer 0.12.1", - "const-oid 0.10.2", - "crypto-common 0.2.2", -+ "ctutils", - ] - - [[package]] -@@ -2322,7 +2390,7 @@ version = "25.12.19" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "rustc_version", - ] - -@@ -2369,6 +2437,12 @@ dependencies = [ - "percent-encoding", - ] - -+[[package]] -+name = "frostem" -+version = "1.20260821.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "36a80a7406da302e04bfd2ca987907590d3a1f3c69958947c43890abd7426b2f" -+ - [[package]] - name = "fs_extra" - version = "1.3.0" -@@ -2377,8 +2451,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" - - [[package]] - name = "fsst" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "rand 0.9.2", -@@ -2727,14 +2801,23 @@ dependencies = [ - - [[package]] - name = "goosefs-sdk" --version = "0.1.5" -+version = "0.1.9" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "9ae079b88ffe7772d12cfc5c40a5a324babb357893d95b5e3a22ae857f236c5f" -+checksum = "e1ea4eee6dcbc31b25ab4fd577adc55b677d2bed3aa3016c44c58fbe1b2298a5" - dependencies = [ -+ "arc-swap", - "async-trait", - "bytes", - "dashmap", -+ "fastrand", -+ "futures", - "hostname", -+ "io-uring", -+ "itoa", -+ "libc", -+ "lru", -+ "memmap2", -+ "moka", - "prost", - "prost-types", - "rand 0.9.2", -@@ -2747,6 +2830,7 @@ dependencies = [ - "tonic-prost", - "tracing", - "uuid", -+ "xxhash-rust", - ] - - [[package]] -@@ -2900,6 +2984,15 @@ dependencies = [ - "digest 0.10.7", - ] - -+[[package]] -+name = "hmac" -+version = "0.13.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "6303bc9732ae41b04cb554b844a762b4115a61bfaa81e3e83050991eeb56863f" -+dependencies = [ -+ "digest 0.11.3", -+] -+ - [[package]] - name = "hostname" - version = "0.4.2" -@@ -3053,7 +3146,7 @@ version = "0.1.20" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "96547c2556ec9d12fb1578c4eaf448b04993e7fb79cbaad930a656880a6bdfa0" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bytes", - "futures-channel", - "futures-util", -@@ -3347,7 +3440,7 @@ version = "0.7.12" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "4d09b98f7eace8982db770e4408e7470b028ce513ac28fecdc6bf4c30fe92b62" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "cfg-if 1.0.4", - "libc", - ] -@@ -3400,10 +3493,12 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - - [[package]] - name = "jiff" --version = "0.2.23" -+version = "0.2.35" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "1a3546dc96b6d42c5f24902af9e2538e82e39ad350b0c766eb3fbf2d8f3d8359" -+checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" - dependencies = [ -+ "defmt", -+ "jiff-core", - "jiff-static", - "jiff-tzdb-platform", - "js-sys", -@@ -3412,15 +3507,25 @@ dependencies = [ - "portable-atomic-util", - "serde_core", - "wasm-bindgen", -- "windows-sys 0.61.2", -+ "windows-link", -+] -+ -+[[package]] -+name = "jiff-core" -+version = "0.1.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" -+dependencies = [ -+ "defmt", - ] - - [[package]] - name = "jiff-static" --version = "0.2.23" -+version = "0.2.35" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "2a8c8b344124222efd714b73bb41f8b5120b27a7cc1c75593a6ff768d9d05aa4" -+checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" - dependencies = [ -+ "jiff-core", - "proc-macro2", - "quote", - "syn 2.0.117", -@@ -3530,24 +3635,6 @@ dependencies = [ - "serde_json", - ] - --[[package]] --name = "jsonwebtoken" --version = "10.4.0" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "eba32bfb4ffdeaca3e34431072faf01745c9b26d25504aa7a6cf5684334fc4fc" --dependencies = [ -- "aws-lc-rs", -- "base64", -- "getrandom 0.2.17", -- "js-sys", -- "pem", -- "serde", -- "serde_json", -- "signature", -- "simple_asn1", -- "zeroize", --] -- - [[package]] - name = "konst" - version = "0.4.3" -@@ -3567,8 +3654,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a" - - [[package]] - name = "lance" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arc-swap", - "arrow", -@@ -3584,7 +3671,6 @@ dependencies = [ - "async-recursion", - "async-trait", - "async_cell", -- "aws-credential-types", - "byteorder", - "bytes", - "chrono", -@@ -3599,7 +3685,6 @@ dependencies = [ - "either", - "fst", - "futures", -- "half", - "humantime", - "itertools 0.14.0", - "lance-arrow", -@@ -3641,8 +3726,8 @@ dependencies = [ - - [[package]] - name = "lance-arrow" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", -@@ -3664,7 +3749,7 @@ dependencies = [ - [[package]] - name = "lance-arrow-scalar" - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", -@@ -3678,7 +3763,7 @@ dependencies = [ - [[package]] - name = "lance-arrow-stats" - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-schema", -@@ -3687,8 +3772,8 @@ dependencies = [ - - [[package]] - name = "lance-bitpacking" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrayref", - "crunchy", -@@ -3728,20 +3813,19 @@ dependencies = [ - - [[package]] - name = "lance-core" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "async-trait", -- "byteorder", -+ "blake3", - "bytes", - "datafusion-common", - "datafusion-sql", - "futures", -- "itertools 0.14.0", - "lance-arrow", - "lance-derive", - "libc", -@@ -3752,13 +3836,13 @@ dependencies = [ - "object_store", - "pin-project", - "prost", -+ "quick_cache", - "rand 0.9.2", - "roaring", - "serde_json", - "snafu", - "tempfile", - "tokio", -- "tokio-stream", - "tokio-util", - "tracing", - "twox-hash", -@@ -3767,8 +3851,8 @@ dependencies = [ - - [[package]] - name = "lance-datafusion" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "arrow-array", -@@ -3788,7 +3872,6 @@ dependencies = [ - "jsonb", - "lance-arrow", - "lance-core", -- "lance-datagen", - "lance-geo", - "log", - "pin-project", -@@ -3800,8 +3883,8 @@ dependencies = [ - - [[package]] - name = "lance-datagen" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "arrow-array", -@@ -3818,8 +3901,8 @@ dependencies = [ - - [[package]] - name = "lance-derive" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "proc-macro2", - "quote", -@@ -3828,8 +3911,8 @@ dependencies = [ - - [[package]] - name = "lance-encoding" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-arith", - "arrow-array", -@@ -3854,8 +3937,6 @@ dependencies = [ - "num-traits", - "prost", - "prost-build", -- "rand 0.9.2", -- "strum", - "tokio", - "tracing", - "xxhash-rust", -@@ -3864,12 +3945,13 @@ dependencies = [ - - [[package]] - name = "lance-file" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-arith", - "arrow-array", - "arrow-buffer", -+ "arrow-cast", - "arrow-data", - "arrow-schema", - "arrow-select", -@@ -3895,8 +3977,8 @@ dependencies = [ - - [[package]] - name = "lance-geo" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "datafusion", - "geo-traits", -@@ -3910,13 +3992,14 @@ dependencies = [ - - [[package]] - name = "lance-index" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arc-swap", - "arrow", - "arrow-arith", - "arrow-array", -+ "arrow-ipc", - "arrow-ord", - "arrow-schema", - "arrow-select", -@@ -3925,7 +4008,6 @@ dependencies = [ - "async-trait", - "bitvec", - "bytes", -- "chrono", - "crossbeam-queue", - "datafusion", - "datafusion-common", -@@ -3945,7 +4027,6 @@ dependencies = [ - "lance-bitpacking", - "lance-core", - "lance-datafusion", -- "lance-datagen", - "lance-encoding", - "lance-file", - "lance-geo", -@@ -3975,13 +4056,12 @@ dependencies = [ - "tempfile", - "tokio", - "tracing", -- "uuid", - ] - - [[package]] - name = "lance-index-core" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-schema", -@@ -4003,18 +4083,12 @@ dependencies = [ - - [[package]] - name = "lance-io" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", -- "arrow-arith", - "arrow-array", -- "arrow-buffer", -- "arrow-cast", -- "arrow-data", - "arrow-schema", -- "arrow-select", -- "async-recursion", - "async-trait", - "aws-config", - "aws-credential-types", -@@ -4022,10 +4096,8 @@ dependencies = [ - "bytes", - "chrono", - "futures", -- "goosefs-sdk", - "http 1.4.0", - "io-uring", -- "lance-arrow", - "lance-core", - "lance-namespace", - "log", -@@ -4037,34 +4109,37 @@ dependencies = [ - "pin-project", - "prost", - "rand 0.9.2", -+ "reqsign-core", -+ "reqsign-file-read-tokio", -+ "reqsign-google", - "serde", -+ "serde_json", - "tempfile", - "tokio", - "tracing", - "url", -+ "uuid", - ] - - [[package]] - name = "lance-linalg" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", -- "arrow-buffer", - "arrow-schema", - "cc", - "half", - "lance-arrow", - "lance-core", - "num-traits", -- "rand 0.9.2", - "rayon", - ] - - [[package]] - name = "lance-namespace" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "async-trait", -@@ -4076,9 +4151,9 @@ dependencies = [ - - [[package]] - name = "lance-namespace-reqwest-client" --version = "0.8.6" -+version = "0.11.1" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "ba3f0a235e3ed5f8805205649ccc7d7d0f3df23ce1294242c9265ad488d7f19d" -+checksum = "1d06b1fbb5d41f93bc652b61e2872af92e8a6c5f6b4ce8839a8ecfa05365d359" - dependencies = [ - "reqwest 0.12.28", - "serde", -@@ -4090,14 +4165,13 @@ dependencies = [ - - [[package]] - name = "lance-select" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-schema", - "byteorder", -- "bytes", - "itertools 0.14.0", - "lance-core", - "roaring", -@@ -4106,8 +4180,8 @@ dependencies = [ - - [[package]] - name = "lance-table" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "arrow-array", -@@ -4115,6 +4189,7 @@ dependencies = [ - "arrow-ipc", - "arrow-schema", - "async-trait", -+ "blake3", - "byteorder", - "bytes", - "chrono", -@@ -4144,11 +4219,11 @@ dependencies = [ - - [[package]] - name = "lance-tokenizer" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ -+ "frostem", - "icu_segmenter", -- "rust-stemmers", - "serde", - "stop-words", - "unicode-normalization", -@@ -4160,7 +4235,7 @@ version = "1.5.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" - dependencies = [ -- "spin", -+ "spin 0.9.8", - ] - - [[package]] -@@ -4291,9 +4366,9 @@ dependencies = [ - - [[package]] - name = "log" --version = "0.4.29" -+version = "0.4.34" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897" -+checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" - - [[package]] - name = "loom" -@@ -4308,6 +4383,15 @@ dependencies = [ - "tracing-subscriber", - ] - -+[[package]] -+name = "lru" -+version = "0.18.4" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "ff9840bcc50b71349309900da0ce7279aa336ae71d73250b07998932c7d97c25" -+dependencies = [ -+ "hashbrown 0.17.1", -+] -+ - [[package]] - name = "lru-slab" - version = "0.1.2" -@@ -4399,6 +4483,15 @@ version = "2.8.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" - -+[[package]] -+name = "memmap2" -+version = "0.9.11" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" -+dependencies = [ -+ "libc", -+] -+ - [[package]] - name = "mime" - version = "0.3.17" -@@ -4619,7 +4712,7 @@ version = "0.3.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - ] - - [[package]] -@@ -4648,7 +4741,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "622acbc9100d3c10e2ee15804b0caa40e55c933d5aa53814cd520805b7958a49" - dependencies = [ - "async-trait", -- "base64", -+ "base64 0.22.1", - "bytes", - "chrono", - "form_urlencoded", -@@ -4664,7 +4757,7 @@ dependencies = [ - "md-5 0.10.6", - "parking_lot", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.39.4", - "rand 0.10.1", - "reqwest 0.12.28", - "ring", -@@ -4683,9 +4776,9 @@ dependencies = [ - - [[package]] - name = "object_store_opendal" --version = "0.57.0" -+version = "0.58.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0eb12a624a41fce745838d0ef3701ff6c47797c13cd18ad3612fd2a3134fdbd8" -+checksum = "88f165780495c17aa3ce86846600504198c3fffd99073521552751c2430fa6ac" - dependencies = [ - "async-trait", - "bytes", -@@ -4718,12 +4811,13 @@ checksum = "269bca4c2591a28585d6bf10d9ed0332b7d76900a1b02bec41bdc3a2cdcda107" - - [[package]] - name = "opendal" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "96c9c85ce253ff87225e7669979d877a20c98a06604ec9d6dd5f4473e08f1ae1" -+checksum = "33dbff14cc9bb085224256d6a81289d2f3202e85b06f408d42534b42162a4231" - dependencies = [ - "ctor 1.0.13", - "opendal-core", -+ "opendal-http-transport-reqwest", - "opendal-layer-concurrent-limit", - "opendal-layer-logging", - "opendal-layer-retry", -@@ -4741,24 +4835,22 @@ dependencies = [ - - [[package]] - name = "opendal-core" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "c4f8607c90e2c963a91467f50fb49fbc7fb3d573f88cea219ca59ccd3740b309" -+checksum = "48dbcef97d3eb7591db2c18d5cae95c836bcce07359b98d98dd6f4e861eb77b7" - dependencies = [ - "anyhow", -- "base64", -+ "asyncband", -+ "base64 0.23.1", - "bytes", - "futures", - "http 1.4.0", -- "http-body 1.0.1", - "jiff", - "log", - "md-5 0.11.0", -- "mea", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", -- "reqwest 0.13.3", - "serde", - "serde_json", - "tokio", -@@ -4767,23 +4859,37 @@ dependencies = [ - "web-time", - ] - -+[[package]] -+name = "opendal-http-transport-reqwest" -+version = "0.58.2" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "85663452ea32bbc17e8f79ab29788c846d116ec7de31451be9c787e462dcb36c" -+dependencies = [ -+ "bytes", -+ "futures", -+ "http 1.4.0", -+ "http-body 1.0.1", -+ "opendal-core", -+ "reqwest 0.13.4", -+] -+ - [[package]] - name = "opendal-layer-concurrent-limit" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0d6f81ba6960e3fae1882f253b114b21d7e444e1534f209c7737a79f6243eb6f" -+checksum = "03f9e144b5228d741c3763ade8711d9b72e5fb6d998e779f2d7a09da0b5a3eba" - dependencies = [ -+ "asyncband", - "futures", - "http 1.4.0", -- "mea", - "opendal-core", - ] - - [[package]] - name = "opendal-layer-logging" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "58ada45c6d81d1aa4c9305d0c7d4bc317c59c85866a0908a2d75a7a978aa5ee2" -+checksum = "c2de17c61cd32e9d8d7d8efb91795e714dbccbafbc3c4e219e1542f4d6324161" - dependencies = [ - "log", - "opendal-core", -@@ -4791,9 +4897,9 @@ dependencies = [ - - [[package]] - name = "opendal-layer-retry" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "7b2a25a718afb81fad81cb9a0580a1cb989221fa2317f888c6a37f8dad408eb7" -+checksum = "e94db301964a25366090484d61e6da16d5979cc8faf02c3e12210dc74fafed38" - dependencies = [ - "backon", - "log", -@@ -4802,9 +4908,9 @@ dependencies = [ - - [[package]] - name = "opendal-layer-timeout" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "1e91f731724c213af81e9d03517859c8fc47b4578e64ad61ae4f099f10fe36e3" -+checksum = "08956ddda07465449bfd48825f4f0f25e0351278ac974eaa659895d9d74f2c80" - dependencies = [ - "opendal-core", - "tokio", -@@ -4812,17 +4918,17 @@ dependencies = [ - - [[package]] - name = "opendal-service-azblob" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0030644366ef5d8cbe3a4a5822bf99a4aafddc1666e9d24b44d158d9062fc76a" -+checksum = "6d278d2fb57661947d1c9fb44047e432b2782c48dc085b7abc99fe6e18c26cf8" - dependencies = [ -- "base64", -+ "base64 0.23.1", - "bytes", - "http 1.4.0", - "log", - "opendal-core", - "opendal-service-azure-common", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-azure-storage", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4833,17 +4939,18 @@ dependencies = [ - - [[package]] - name = "opendal-service-azdls" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "6dea4908d490143a9b0b7f7a790e139ff829b06a023f670455ed3d44f664b361" -+checksum = "2d564484a8f7d091827e825cfc91ed45bd48e64d262451ee041fe843db81bd8a" - dependencies = [ -- "base64", -+ "asyncband", -+ "base64 0.23.1", - "bytes", - "http 1.4.0", - "log", - "opendal-core", - "opendal-service-azure-common", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-azure-storage", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4853,9 +4960,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-azure-common" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "9b489f13c42e69d69bdd72952b634356ec43a7881a20259b38b540fcecdf4051" -+checksum = "cfcc1bfdac4f54d9018c462dd32ad9e2f68fdf584f825c811550c8ffc87894cd" - dependencies = [ - "http 1.4.0", - "opendal-core", -@@ -4863,15 +4970,15 @@ dependencies = [ - - [[package]] - name = "opendal-service-cos" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "aa8cafe9729213375c7331019b0cb756ad3e1aff7f45cd32c45eae91ebde8901" -+checksum = "bb021c128ebde42e6f3e719d4cfed27e994017aa503a7a1e5818bdb61fd67dc9" - dependencies = [ - "bytes", - "http 1.4.0", - "log", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "reqsign-file-read-tokio", - "reqsign-tencent-cos", -@@ -4880,9 +4987,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-gcs" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "48de101aac565ed06af4b47903c24eafd249075553ec1fb18256751c45148d47" -+checksum = "da1f2a8c975fd22fea01f0409bcb8774f1ac02ad79863a3490098203121555d3" - dependencies = [ - "async-trait", - "bytes", -@@ -4890,7 +4997,7 @@ dependencies = [ - "log", - "opendal-core", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "reqsign-file-read-tokio", - "reqsign-google", -@@ -4901,9 +5008,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-goosefs" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "69e43048bde419947ba826fbdc2f134d6c03f44ebf48bd33a03b72f9fc45fcb4" -+checksum = "89fd71b80078f2983bd363e322fbebfa6a46d5fc56f17e4f23d76d5edb31226b" - dependencies = [ - "bytes", - "goosefs-sdk", -@@ -4915,9 +5022,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-hf" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "c4922661976a1d40794a2adfbdb888cc3c23097690f825a92f773af38908a848" -+checksum = "c8a3c8ec0c2918fa23f258fa28822f222ec8c1e3a501ded665b78bc65a9d7734" - dependencies = [ - "bytes", - "hf-xet", -@@ -4925,22 +5032,21 @@ dependencies = [ - "log", - "opendal-core", - "percent-encoding", -- "reqwest 0.13.3", - "serde", - "serde_json", - ] - - [[package]] - name = "opendal-service-oss" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "328fa55e8888cbdfe00826bfea2a79042422b720e8369e9e021e46121dea5ace" -+checksum = "8e3ce7a2ceb925e0f28f545b169eb57d04cec6116aae94b8584d2bdbe7d456c6" - dependencies = [ - "bytes", - "http 1.4.0", - "log", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-aliyun-oss", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4949,18 +5055,18 @@ dependencies = [ - - [[package]] - name = "opendal-service-s3" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "313d46c9f5ae70bca26b7c3e3fbb9b639292625f28af73aa016f47e788af9deb" -+checksum = "c64335f9f24ccb62ac36f1d976342b48611a75ba61979813a4f78a4ebd94de42" - dependencies = [ -- "base64", -+ "base64 0.23.1", - "bytes", -- "crc32c", -+ "crc-fast", - "http 1.4.0", - "log", - "md-5 0.11.0", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-aws-v4", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4970,14 +5076,14 @@ dependencies = [ - - [[package]] - name = "opendal-service-tos" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "6f2f7a4c32e5202eb4ac72e76c4b5e30c86ab60762811172f4111103b9d673a1" -+checksum = "70c3c507c3a565b2feb4c7b5f653436acd34ca59ffc842a7671b5498b13b76bf" - dependencies = [ - "bytes", - "http 1.4.0", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "reqsign-file-read-tokio", - "reqsign-volcengine-tos", -@@ -5084,7 +5190,7 @@ version = "0.8.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "898bac3fa00d0ba57a4e8289837e965baa2dee8c3749f3b11d45a64b4223d9c3" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "serde", - ] - -@@ -5122,16 +5228,16 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "f8ed6a7761f76e3b9f92dfb0a60a6a6477c61024b775147ff0973a02653abaf2" - dependencies = [ - "digest 0.10.7", -- "hmac", -+ "hmac 0.12.1", - ] - - [[package]] - name = "pem" --version = "3.0.6" -+version = "4.0.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "1d30c53c26bc5b31a98cd02d20f25a7c8567146caf63ed593a9d87b2775291be" -+checksum = "d354a98a3d1251555de99e8fdd8afda05573c31b82f59063a7b0a29b5527f120" - dependencies = [ -- "base64", -+ "base64 0.23.1", - "serde_core", - ] - -@@ -5392,6 +5498,28 @@ dependencies = [ - "serde", - ] - -+[[package]] -+name = "quick-xml" -+version = "0.41.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" -+dependencies = [ -+ "memchr", -+ "serde", -+] -+ -+[[package]] -+name = "quick_cache" -+version = "0.6.24" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "b9c6658afe513a3b484e3abfdaa0d03ef3c0bbf017542c178dd55f94eb3051f9" -+dependencies = [ -+ "ahash", -+ "equivalent", -+ "hashbrown 0.16.1", -+ "parking_lot", -+] -+ - [[package]] - name = "quinn" - version = "0.11.9" -@@ -5616,7 +5744,7 @@ version = "0.5.18" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - ] - - [[package]] -@@ -5697,9 +5825,9 @@ dependencies = [ - - [[package]] - name = "reqsign-aliyun-oss" --version = "3.0.0" -+version = "3.1.4" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "57ac2757f3140aa2e213b554148ae0b52733e624fc6723f0cc6bb3d440176c95" -+checksum = "68d24d281f734a463093b7b93aae8b16f5f8496a54fbeec4d2d10b0488295296" - dependencies = [ - "anyhow", - "form_urlencoded", -@@ -5713,18 +5841,18 @@ dependencies = [ - ] - - [[package]] --name = "reqsign-aws-v4" --version = "3.0.0" -+name = "reqsign-aws-core" -+version = "3.1.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "44eaca382e94505a49f1a4849658d153aebf79d9c1a58e5dd3b10361511e9f43" -+checksum = "4d63b56638bb3cc7bd376a7cdce1ba3089777a08f47e4097888f2d784cc3f46c" - dependencies = [ -- "anyhow", - "bytes", - "form_urlencoded", -+ "hex", - "http 1.4.0", - "log", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "rust-ini", - "serde", -@@ -5733,18 +5861,32 @@ dependencies = [ - "sha1", - ] - -+[[package]] -+name = "reqsign-aws-v4" -+version = "3.2.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "4a0c499f4ed12d04c3d4c78fe4cb01aee22c9dae22848c14db2c6313d9df9f43" -+dependencies = [ -+ "bytes", -+ "http 1.4.0", -+ "log", -+ "quick-xml 0.41.0", -+ "reqsign-aws-core", -+ "reqsign-core", -+ "serde", -+] -+ - [[package]] - name = "reqsign-azure-storage" --version = "3.0.0" -+version = "3.2.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "7a321980405d596bd34aaf95c4722a3de4128a67fd19e74a81a83aa3fdf082e6" -+checksum = "e8177b4f08620ab7f2e9cab7d7ccb9da1b61c66b889fed46cc0880b0fe75eb6b" - dependencies = [ - "anyhow", -- "base64", -+ "base64 0.23.1", - "bytes", - "form_urlencoded", - "http 1.4.0", -- "jsonwebtoken", - "log", - "pem", - "percent-encoding", -@@ -5757,31 +5899,33 @@ dependencies = [ - - [[package]] - name = "reqsign-core" --version = "3.0.0" -+version = "3.3.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "b10302cf0a7d7e7352ba211fc92c3c5bebf1286153e49cc5aa87348078a8e102" -+checksum = "f4ac1510872d9481205975d264deb39c109797e5068cc882ed9064270eaae5fa" - dependencies = [ - "anyhow", -- "base64", -+ "base64 0.23.1", - "bytes", -- "form_urlencoded", - "futures", - "hex", -- "hmac", -+ "hmac 0.13.0", - "http 1.4.0", - "jiff", - "log", - "percent-encoding", -+ "rsa", -+ "serde", -+ "serde_json", - "sha1", -- "sha2 0.10.9", -+ "sha2 0.11.0", - "windows-sys 0.61.2", - ] - - [[package]] - name = "reqsign-file-read-tokio" --version = "3.0.0" -+version = "3.0.5" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e2d89295b3d17abea31851cc8de55d843d89c52132c864963c38d41920613dc5" -+checksum = "95c3371bfc7e5c7f9627a04133af3583fd6c28715e7c83f79db38f3b384f535f" - dependencies = [ - "anyhow", - "reqsign-core", -@@ -5790,13 +5934,13 @@ dependencies = [ - - [[package]] - name = "reqsign-google" --version = "3.0.0" -+version = "3.1.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "35cc609b49c69e76ecaceb775a03f792d1ed3e7755ab3548d4534fd801e3242e" -+checksum = "f81a9d38870892443489c0abb5332edfa81d5a14c437af9caef7c194897c92c1" - dependencies = [ -+ "bytes", - "form_urlencoded", - "http 1.4.0", -- "jsonwebtoken", - "log", - "percent-encoding", - "reqsign-aws-v4", -@@ -5804,15 +5948,14 @@ dependencies = [ - "rsa", - "serde", - "serde_json", -- "sha2 0.10.9", - "tokio", - ] - - [[package]] - name = "reqsign-tencent-cos" --version = "3.0.0" -+version = "3.0.5" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e128f19525861dbded59e1e7c17653a8ed63d573ca04aed708d552dbef5bb32a" -+checksum = "b15c5a4df7c3f16823ae242675c5ebfb52d640cc9a50d1fcf943263247fa1730" - dependencies = [ - "anyhow", - "http 1.4.0", -@@ -5825,9 +5968,9 @@ dependencies = [ - - [[package]] - name = "reqsign-volcengine-tos" --version = "3.0.0" -+version = "3.1.1" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "f9d757602a7ef2b6025c0da77e6d2e23fbdef35930fa466b15ffbf0a3f13acf7" -+checksum = "173387eb5ae4cf6a0a7098665ebcc6729862819dc3d95a81aee840767803e3d9" - dependencies = [ - "anyhow", - "http 1.4.0", -@@ -5842,7 +5985,7 @@ version = "0.12.28" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bytes", - "encoding_rs", - "futures-core", -@@ -5884,11 +6027,11 @@ dependencies = [ - - [[package]] - name = "reqwest" --version = "0.13.3" -+version = "0.13.4" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "62e0021ea2c22aed41653bc7e1419abb2c97e038ff2c33d0e1309e49a97deec0" -+checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bytes", - "futures-core", - "futures-util", -@@ -5931,7 +6074,7 @@ dependencies = [ - "anyhow", - "async-trait", - "http 1.4.0", -- "reqwest 0.13.3", -+ "reqwest 0.13.4", - "thiserror 2.0.18", - "tower-service", - ] -@@ -5946,7 +6089,7 @@ dependencies = [ - "cfg-if 1.0.4", - "getrandom 0.2.17", - "libc", -- "untrusted 0.9.0", -+ "untrusted", - "windows-sys 0.52.0", - ] - -@@ -6008,16 +6151,6 @@ dependencies = [ - "ordered-multimap", - ] - --[[package]] --name = "rust-stemmers" --version = "1.2.0" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e46a2036019fdb888131db7a4c847a1063a7493f971ed94ea82c67eada63ca54" --dependencies = [ -- "serde", -- "serde_derive", --] -- - [[package]] - name = "rustc-hash" - version = "2.1.1" -@@ -6039,7 +6172,7 @@ version = "1.1.4" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "errno", - "libc", - "linux-raw-sys", -@@ -6119,7 +6252,7 @@ dependencies = [ - "aws-lc-rs", - "ring", - "rustls-pki-types", -- "untrusted 0.9.0", -+ "untrusted", - ] - - [[package]] -@@ -6244,7 +6377,7 @@ version = "3.7.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "core-foundation 0.10.1", - "core-foundation-sys", - "libc", -@@ -6373,7 +6506,7 @@ version = "3.20.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "e72c1c2cb7b223fafb600a619537a871c2818583d619401b785e7c0b746ccde2" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bs58", - "chrono", - "hex", -@@ -6414,13 +6547,13 @@ dependencies = [ - - [[package]] - name = "sha1" --version = "0.10.6" -+version = "0.11.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e3bf829a2d51ab4a5ddf1352d8470c140cadc8301b2ae1789db023f01cedd6ba" -+checksum = "aacc4cc499359472b4abe1bf11d0b12e688af9a805fa5e3016f9a386dc2d0214" - dependencies = [ - "cfg-if 1.0.4", -- "cpufeatures 0.2.17", -- "digest 0.10.7", -+ "cpufeatures 0.3.0", -+ "digest 0.11.3", - ] - - [[package]] -@@ -6523,18 +6656,6 @@ version = "0.1.5" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" - --[[package]] --name = "simple_asn1" --version = "0.6.4" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0d585997b0ac10be3c5ee635f1bab02d512760d14b7c468801ac8a01d9ae5f1d" --dependencies = [ -- "num-bigint", -- "num-traits", -- "thiserror 2.0.18", -- "time", --] -- - [[package]] - name = "siphasher" - version = "1.0.2" -@@ -6602,6 +6723,12 @@ version = "0.9.8" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67" - -+[[package]] -+name = "spin" -+version = "0.10.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" -+ - [[package]] - name = "spki" - version = "0.7.3" -@@ -6682,28 +6809,6 @@ version = "0.11.1" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" - --[[package]] --name = "strum" --version = "0.26.3" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "8fec0f0aef304996cf250b31b5a10dee7980c85da9d759361292b8bca5a18f06" --dependencies = [ -- "strum_macros", --] -- --[[package]] --name = "strum_macros" --version = "0.26.4" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "4c6bee85a5a24955dc440386795aa378cd9cf82acd5f764469152d2270e581be" --dependencies = [ -- "heck", -- "proc-macro2", -- "quote", -- "rustversion", -- "syn 2.0.117", --] -- - [[package]] - name = "substrait" - version = "0.63.0" -@@ -6804,7 +6909,7 @@ version = "0.7.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "a13f3d0daba03132c0aa9767f98351b3488edc2c100cda2d2ec2b04f3d8d3c8b" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "core-foundation 0.9.4", - "system-configuration-sys", - ] -@@ -7078,7 +7183,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" - dependencies = [ - "async-trait", -- "base64", -+ "base64 0.22.1", - "bytes", - "h2", - "http 1.4.0", -@@ -7136,7 +7241,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "d4e6559d53cc268e5031cd8429d05415bc4cb4aefc4aa5d6cc35fbf5b924a1f8" - dependencies = [ - "async-compression", -- "bitflags", -+ "bitflags 2.11.0", - "bytes", - "futures-core", - "futures-util", -@@ -7370,12 +7475,6 @@ version = "0.2.11" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "673aac59facbab8a9007c7f6108d11f63b603f7cabff99fabf650fea5c32b861" - --[[package]] --name = "untrusted" --version = "0.7.1" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "a156c684c91ea7d62626509bce3cb4e1d9ed5c4d978f7b4352658f96a4c26b4a" -- - [[package]] - name = "untrusted" - version = "0.9.0" -@@ -7622,7 +7721,7 @@ version = "0.244.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "hashbrown 0.15.5", - "indexmap 2.14.0", - "semver", -@@ -8054,7 +8153,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" - dependencies = [ - "anyhow", -- "bitflags", -+ "bitflags 2.11.0", - "indexmap 2.14.0", - "log", - "serde", -@@ -8132,7 +8231,7 @@ checksum = "3e1e496dcbe6a09017acdfaf48e1a646735e7ff5b2a49e2c7e081cca77a59bc8" - dependencies = [ - "anyhow", - "async-trait", -- "base64", -+ "base64 0.22.1", - "bytes", - "clap", - "crc32fast", -@@ -8143,7 +8242,7 @@ dependencies = [ - "more-asserts", - "rand 0.10.1", - "redb", -- "reqwest 0.13.3", -+ "reqwest 0.13.4", - "reqwest-middleware", - "serde", - "serde_json", -@@ -8169,7 +8268,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "cb838aa8eb67d730af301584cf003caad407487606058292a6750711b603fbee" - dependencies = [ - "async-trait", -- "base64", -+ "base64 0.22.1", - "blake3", - "bytemuck", - "bytes", -@@ -8256,7 +8355,7 @@ dependencies = [ - "oneshot", - "pin-project", - "rand 0.10.1", -- "reqwest 0.13.3", -+ "reqwest 0.13.4", - "serde", - "serde_json", - "shellexpand", -@@ -8352,20 +8451,6 @@ name = "zeroize" - version = "1.8.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" --dependencies = [ -- "zeroize_derive", --] -- --[[package]] --name = "zeroize_derive" --version = "1.4.3" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "85a5b4158499876c763cb03bc4e49185d3cccbabb15b33c627f7884f43db852e" --dependencies = [ -- "proc-macro2", -- "quote", -- "syn 2.0.117", --] - - [[package]] - name = "zerotrie" -diff --git a/Cargo.toml b/Cargo.toml -index d072a5d..3920a65 100644 ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -18,14 +18,14 @@ rust-version = "1.91.0" - crate-type = ["cdylib", "staticlib", "rlib"] - - [dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c", features = ["substrait"] } --lance-core = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-index = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-io = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-table = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c", features = ["substrait"] } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } -+lance-core = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-index = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-io = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-table = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } - datafusion = { version = "54.0.0", default-features = false } - arrow = { version = "58.0.0", features = ["prettyprint", "ffi"] } - arrow-array = "58.0.0" -@@ -45,9 +45,9 @@ snafu = "0.9" - uuid = { version = "1", features = ["v4"] } - - [dev-dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c", features = ["substrait"] } --lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } -+lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } - tokio = { version = "1", features = ["rt-multi-thread", "macros"] } - arrow-array = "58.0.0" - arrow-schema = "58.0.0" -diff --git a/src/fts_query.rs b/src/fts_query.rs -index c9d3a43..71874e0 100644 ---- a/src/fts_query.rs -+++ b/src/fts_query.rs -@@ -246,7 +246,7 @@ async fn prepare_fts_query_context( - .with_max_expansions(match_query.max_expansions) - .with_prefix_length(match_query.prefix_length); - PreparedFtsQuery::Match(Arc::new( -- build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms, None).await?, - )) - } - FtsQuery::Phrase(phrase_query) => { -@@ -260,7 +260,7 @@ async fn prepare_fts_query_context( - let query_tokens = collect_query_tokens(&phrase_query.terms, &mut tokenizer); - let params = query.params().with_phrase_slop(Some(phrase_query.slop)); - PreparedFtsQuery::Phrase(Arc::new( -- build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms, None).await?, - )) - } - _ => { -diff --git a/src/index_segment.rs b/src/index_segment.rs -index a46c4f3..d4e143c 100644 ---- a/src/index_segment.rs -+++ b/src/index_segment.rs -@@ -889,7 +889,7 @@ unsafe fn new_vector_builder_inner( - // TODO(upstream-lance): Remove this fail-fast once Lance's distributed - // vector-index path reconstructs a supplied PQ codebook with an L2 - // ProductQuantizer, matching the ordinary full-dataset path. Pinned Lance -- // revision e934cc2c rewraps supplied codebooks with DistanceType::Dot in -+ // revision ab6b5bbe rewraps supplied codebooks with DistanceType::Dot in - // `make_global_pq`, which silently switches PQ code assignment away from - // the L2 contract shared by full-dataset builds and index readers. - if matches!( -@@ -910,7 +910,7 @@ unsafe fn new_vector_builder_inner( - let selected_fragment_ids: HashSet = fragment_ids.iter().copied().collect(); - if selected_fragment_ids != all_fragment_ids { - return Err(invalid_input(format!( -- "pq_codebook is supplied for metric=DOT, index_type={:?}, mode={:?}, and an effective strict fragment subset ({} of {} fragments): pinned Lance revision e934cc2c reconstructs the supplied codebook with a DOT ProductQuantizer in the distributed build path (make_global_pq), silently breaking the L2 PQ-assignment contract; cover the full dataset in one segment (pass NULL fragment_ids or list every fragment) or wait for upstream Lance DOT support", -+ "pq_codebook is supplied for metric=DOT, index_type={:?}, mode={:?}, and an effective strict fragment subset ({} of {} fragments): pinned Lance revision ab6b5bbe reconstructs the supplied codebook with a DOT ProductQuantizer in the distributed build path (make_global_pq), silently breaking the L2 PQ-assignment contract; cover the full dataset in one segment (pass NULL fragment_ids or list every fragment) or wait for upstream Lance DOT support", - params.index_type, - parsed.mode, - selected_fragment_ids.len(), -diff --git a/src/scanner.rs b/src/scanner.rs -index 7e898ce..d3ef3be 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -529,7 +529,7 @@ fn rewrite_prepared_fts_plan( - exec.params().clone(), - exec.prefilter_source().clone(), - segments.to_vec(), -- ) -+ )? - .with_base_scorer(Arc::clone(scorer)); - return Ok((Arc::new(replacement), rewritten)); - } -@@ -549,7 +549,7 @@ fn rewrite_prepared_fts_plan( - exec.params().clone(), - exec.prefilter_source().clone(), - segments.to_vec(), -- ) -+ )? - .with_base_scorer(Arc::clone(scorer)); - return Ok((Arc::new(replacement), rewritten)); - } -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index b763cef..bde742d 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -2449,8 +2449,7 @@ fn test_robotics_e2e_write_then_finalize() { - format!("data/{}", filename), - field_ids, - column_indices, -- meta.major_version as u32, -- meta.minor_version as u32, -+ meta.version, - None, // file_size_bytes - None, // base_id - ); -@@ -3405,6 +3404,7 @@ fn test_index_segment_metadata_parse_rejects_malformed_and_dangerous_input() { - created_at: Some(u64::MAX), - base_id: None, - files: Vec::new(), -+ covering_fields: Vec::new(), - } - .encode_to_vec(); - assert_eq!( -@@ -3429,6 +3429,7 @@ fn test_index_segment_metadata_parse_rejects_malformed_and_dangerous_input() { - created_at: None, - base_id: None, - files: Vec::new(), -+ covering_fields: Vec::new(), - } - .encode_to_vec(); - assert_eq!( -@@ -3456,6 +3457,7 @@ fn test_index_segment_metadata_parse_rejects_malformed_and_dangerous_input() { - created_at: None, - base_id: None, - files: Vec::new(), -+ covering_fields: Vec::new(), - } - .encode_to_vec(); - assert_eq!( -@@ -4033,7 +4035,7 @@ fn test_vector_index_segment_rejects_strict_subset_dot_pq() { - let message = take_last_error_message(); - assert!(message.contains("metric=DOT"), "{message}"); - assert!(message.contains("strict fragment subset"), "{message}"); -- assert!(message.contains("e934cc2c"), "{message}"); -+ assert!(message.contains("ab6b5bbe"), "{message}"); - assert!(message.contains("1 of 2 fragments"), "{message}"); - assert!(!centroids.is_released()); - assert!(!codebook.is_released()); diff --git a/thirdparty/patches/lance-c-0.1.9-pr-79.patch b/thirdparty/patches/lance-c-0.1.9-pr-79.patch deleted file mode 100644 index ad32a48065f09d..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-79.patch +++ /dev/null @@ -1,1782 +0,0 @@ -From d819fbdfa52031d84d1fa01f2d06c51a6712c40b Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Tue, 8 Sep 2026 22:30:48 +0800 -Subject: [PATCH 1/5] scalar index segment - ---- - docs/scalar-segment-scans.md | 78 +++++++++++ - include/lance/lance.h | 24 ++++ - include/lance/lance.hpp | 14 ++ - src/lib.rs | 1 + - src/scalar_segment.rs | 224 +++++++++++++++++++++++++++++++ - src/scanner.rs | 61 +++++++++ - tests/c_api_test.rs | 249 +++++++++++++++++++++++++++++++++++ - 7 files changed, 651 insertions(+) - create mode 100644 docs/scalar-segment-scans.md - create mode 100644 src/scalar_segment.rs - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -new file mode 100644 -index 0000000..dc7c141 ---- /dev/null -+++ b/docs/scalar-segment-scans.md -@@ -0,0 +1,78 @@ -+# Scalar index segment scans -+ -+An ordinary scanner can use one physical BTree/Bitmap segment to generate -+candidates, then read those candidates with the complete scanner filter. This -+does not run a global search of the other segments of the logical index. It does -+not subdivide a physical segment or make its own index search incremental. -+ -+## Configuring a task -+ -+Open a fixed dataset version. Select the physical index UUID from that version's -+metadata and pass the task's complete fragment domain explicitly: -+ -+```c -+LanceScanner *scanner = lance_scanner_new(dataset, columns, full_filter_sql); -+/* Check every return value in production. */ -+lance_scanner_set_fragment_ids(scanner, fragment_ids, fragment_count); -+lance_scanner_set_scalar_index_segment(scanner, segment_uuid_16_bytes); -+lance_scanner_set_limit(scanner, 20000); -+/* The scanner-owning thread calls lance_scanner_next as usual. */ -+``` -+ -+SQL, Substrait and additional SQL filters keep their existing precedence and AND -+composition. The caller does not supply a separate driver predicate: Lance-C -+uses the typed filter planner and selects a necessary indexed leaf belonging to -+the requested logical index. It only descends through AND, never through OR or -+NOT. It then searches the selected UUID and applies the complete filter while -+reading candidates with automatic scalar-index planning disabled. -+ -+Each task's fragment IDs define its result domain, including on fallback. A -+distributed planner must assign disjoint domains whose union covers the intended -+scan. Unindexed fragments need their own tasks, or an explicit domain including -+them (which causes that task to use fallback). Merely listing indexed segments -+does not include appended, unindexed data automatically. -+ -+An unknown UUID, absent fragment or invalid option combination is an error. A -+known segment with incomplete/unknown coverage, no suitable driver, unsupported -+index type, nested key, overlays, fragment reuse, non-exact results or unsupported -+row-ID domain falls back to a non-indexed scan of the entire explicit domain. -+I/O and corruption errors are propagated, not converted to empty results or -+successful fallback. -+ -+The first implementation supports live-row ordinary scans and cannot be combined -+with vector/FTS queries. Physical row-address -+results on stable-row-ID datasets currently fall back; results already expressed -+in the correct row-ID domain use the candidate path. Deletes and all remaining -+predicates are handled by the ordinary reader. No candidate-count limit is -+applied: LIMIT/OFFSET remain after the scanner's complete filter. -+ -+If the host has additional predicates outside Lance, do not set a local limit -+before those predicates. Never divide the global limit by the number of tasks. -+Global OFFSET belongs to the coordinator, not independently to each task. -+ -+## Stopping after the host limit -+ -+This mode uses the existing scanner/stream lifecycle. Once the host has enough -+rows, it stops requesting further batches and closes the scanner after any active -+call has returned. Do not call `lance_scanner_close` concurrently with `next`. -+Exported Arrow streams remain owned by the caller and must also be released after -+their active consumers have finished. -+ -+A host stop flag does not interrupt an in-progress `lance_scanner_next`: current -+index evaluation or I/O may finish before the host observes stop and closes the -+stream. No separate cancellation signal or thread is introduced. The host remains -+responsible for enforcing the global LIMIT across concurrent tasks. -+ -+## Memory and statistics -+ -+Candidate masks stay in Rust, and record batches are streamed. Each active task -+can still hold a complete segment's candidate set; scanner I/O buffer size does -+not cap that allocation. Control task concurrency and physical segment size. -+ -+Successful exhaustion merges segment-search metrics into the existing statistics -+callback exactly once. New metrics include `scalar_segments_requested`, -+`scalar_segments_searched`, `scalar_segment_candidate_rows`, -+`scalar_segment_prepare_time`, `scalar_segment_search_time`, and -+`scalar_segment_fallback_*` reasons. `prepare_time` includes search time. Early -+release, cancellation and errors retain the existing callback contract: final -+statistics are not guaranteed. Metrics do not establish global task concurrency. -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 8173ae5..cbd8330 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1858,6 +1858,30 @@ int32_t lance_scanner_set_index_segments( - size_t len - ); - -+/** -+ * Accelerate an ordinary scalar-filtered scan with one physical index segment. -+ * segment_uuid points to 16 UUID bytes in RFC 4122 order; NULL clears the setting. -+ * Must be configured before scanning. Requires explicit nonempty fragment_ids, -+ * which define BOTH the read and fallback domain, independently of the segment. -+ * Missing snapshot UUIDs / fragment IDs are errors. Extra segment coverage is -+ * excluded by fragment_ids; incomplete coverage falls back to a full filtered -+ * scan of those fragment_ids. Callers distributing work must assign disjoint -+ * fragment domains and separately include any unindexed data they wish to read. -+ * -+ * BTree/Bitmap searches use a necessary AND-conjunct of the full scanner filter -+ * on the selected logical index. All predicates are reapplied during candidate -+ * reads; other scalar indices are disabled. OR/NOT-only filters, overlays, -+ * fragment reuse, unsupported index types / result domains -+ * and missing coverage use the same domain without an index. No filter also -+ * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the -+ * unfiltered candidate set. Vector/FTS queries are rejected. -+ * -+ * UUID bytes are copied. Metadata and final option compatibility are validated -+ * when creating the stream. Index corruption or I/O failures remain errors. -+ */ -+int32_t lance_scanner_set_scalar_index_segment( -+ LanceScanner* scanner, const uint8_t* segment_uuid); -+ - /* ─── Full-text search (Phase 2) ─── */ - - /** -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index c12c0c6..8e1c6cf 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1311,6 +1311,20 @@ class Scanner { - return *this; - } - -+ /// Restrict scalar candidate generation to one segment; fragment_ids is -+ /// required and defines the complete read/fallback domain. See lance.h. -+ Scanner& scalar_index_segment(const std::array& segment_uuid) { -+ if (lance_scanner_set_scalar_index_segment(handle_.get(), segment_uuid.data()) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ Scanner& clear_scalar_index_segment() { -+ if (lance_scanner_set_scalar_index_segment(handle_.get(), nullptr) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Restrict scan to specific fragment IDs. - Scanner& fragment_ids(const uint64_t* ids, size_t len) { - if (lance_scanner_set_fragment_ids(handle_.get(), ids, len) != 0) -diff --git a/src/lib.rs b/src/lib.rs -index 8b212f5..c7ca4cf 100644 ---- a/src/lib.rs -+++ b/src/lib.rs -@@ -39,6 +39,7 @@ mod index_segment; - mod merge_insert; - mod restore; - pub mod runtime; -+mod scalar_segment; - mod scanner; - mod session; - pub mod stream_guard; -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -new file mode 100644 -index 0000000..2ab1297 ---- /dev/null -+++ b/src/scalar_segment.rs -@@ -0,0 +1,224 @@ -+// SPDX-License-Identifier: Apache-2.0 -+// SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+//! Segment-scoped candidate generation for ordinary scans. The explicit fragment -+//! list is the read domain, including on fallback; a segment is only an accelerator. -+ -+use std::collections::HashSet; -+use std::sync::Arc; -+use std::time::Instant; -+ -+use datafusion::physical_plan::metrics::ExecutionPlanMetricsSet; -+use lance::Dataset; -+use lance::dataset::scanner::{ -+ ExecutionStatsCallback, ExecutionSummaryCounts, RowAddrMask, Scanner, -+}; -+use lance::index::{DatasetIndexExt, DatasetIndexInternalExt}; -+use lance::io::exec::utils::IndexMetrics; -+use lance_core::{Error, Result}; -+use lance_datafusion::planner::Planner; -+use lance_datafusion::utils::MetricsExt; -+use lance_index::IndexType; -+use lance_index::scalar::SearchResult; -+use lance_index::scalar::expression::{PlannerIndexExt, ScalarIndexExpr, ScalarIndexSearch}; -+use uuid::Uuid; -+ -+pub(crate) struct PreparedScalarSegment { -+ pub dataset: Arc, -+ pub segment_uuid: Uuid, -+ pub fragment_ids: Vec, -+ pub callback: Option, -+} -+ -+fn invalid(message: impl Into) -> Error { -+ Error::invalid_input_source(message.into().into()) -+} -+ -+// Only descend through AND: a leaf below OR or NOT need not contain all matches -+// of the full expression. The original expression is always reapplied by reader. -+fn driver<'a>(expr: &'a ScalarIndexExpr, index_name: &str) -> Option<&'a ScalarIndexSearch> { -+ match expr { -+ ScalarIndexExpr::Query(search) if search.index_name == index_name => Some(search), -+ ScalarIndexExpr::And(lhs, rhs) => { -+ driver(lhs, index_name).or_else(|| driver(rhs, index_name)) -+ } -+ _ => None, -+ } -+} -+ -+impl PreparedScalarSegment { -+ pub async fn configure(self, mut reader: Scanner) -> Result { -+ // Never let either candidate reads or fallback re-enter a global index search. -+ reader.use_scalar_index(false); -+ let mut stats = ExecutionSummaryCounts::default(); -+ stats -+ .all_counts -+ .insert("scalar_segments_requested".into(), 1); -+ let plan_metrics = ExecutionPlanMetricsSet::new(); -+ let metrics = IndexMetrics::new(&plan_metrics, 0); -+ let started = Instant::now(); -+ let reason = self -+ .configure_candidates(&mut reader, &metrics, &mut stats) -+ .await?; -+ metrics.flush_io(); -+ stats.all_times.insert( -+ "scalar_segment_prepare_time".into(), -+ started.elapsed().as_nanos().min(usize::MAX as u128) as usize, -+ ); -+ if let Some(reason) = reason { -+ stats -+ .all_counts -+ .insert("scalar_segment_fallbacks".into(), 1); -+ stats -+ .all_counts -+ .insert(format!("scalar_segment_fallback_{reason}"), 1); -+ } -+ for (name, count) in plan_metrics.clone_inner().iter_counts() { -+ let name = name.as_ref(); -+ match name { -+ "iops" => stats.iops += count.value(), -+ "requests" => stats.requests += count.value(), -+ "bytes_read" => stats.bytes_read += count.value(), -+ "indices_loaded" => stats.indices_loaded += count.value(), -+ "parts_loaded" => stats.parts_loaded += count.value(), -+ "index_comparisons" => stats.index_comparisons += count.value(), -+ _ => *stats.all_counts.entry(name.to_string()).or_default() += count.value(), -+ } -+ } -+ if let Some(callback) = self.callback { -+ // Preserve the callback's once-per-successfully-exhausted-stream contract. -+ // Candidate work is not part of the underlying reader's plan metrics. -+ reader.scan_stats_callback(Arc::new(move |read| { -+ let mut combined = read.clone(); -+ combined.iops += stats.iops; -+ combined.requests += stats.requests; -+ combined.bytes_read += stats.bytes_read; -+ combined.indices_loaded += stats.indices_loaded; -+ combined.parts_loaded += stats.parts_loaded; -+ combined.index_comparisons += stats.index_comparisons; -+ for (name, value) in &stats.all_counts { -+ *combined.all_counts.entry(name.clone()).or_default() += value; -+ } -+ for (name, value) in &stats.all_times { -+ *combined.all_times.entry(name.clone()).or_default() += value; -+ } -+ callback(&combined); -+ })); -+ } -+ Ok(reader) -+ } -+ -+ async fn configure_candidates( -+ &self, -+ reader: &mut Scanner, -+ metrics: &IndexMetrics, -+ stats: &mut ExecutionSummaryCounts, -+ ) -> Result> { -+ let fragments = self.dataset.get_fragments(); -+ let visible: HashSet = fragments.iter().map(|f| f.id() as u64).collect(); -+ if self.fragment_ids.iter().any(|id| !visible.contains(id)) { -+ return Err(invalid( -+ "scalar segment fragment_ids contains a fragment absent from the dataset snapshot", -+ )); -+ } -+ let indices = self.dataset.load_indices().await?; -+ let index_meta = indices -+ .iter() -+ .find(|i| i.uuid == self.segment_uuid) -+ .ok_or_else(|| { -+ invalid(format!( -+ "scalar index segment {} is absent from the dataset snapshot", -+ self.segment_uuid -+ )) -+ })?; -+ let field_id = index_meta -+ .keyed_field() -+ .ok_or_else(|| invalid("scalar segment must index a single key field"))?; -+ let field = -+ self.dataset.schema().field_by_id(field_id).ok_or_else(|| { -+ invalid("scalar segment key field is absent from the dataset schema") -+ })?; -+ // Keep V1 to flat scalar fields. A dotted name is not sufficient to prove -+ // the field path of an evolved or nested schema. -+ if !self -+ .dataset -+ .schema() -+ .fields -+ .iter() -+ .any(|f| f.id == field.id) -+ { -+ return Ok(Some("nested_field")); -+ } -+ let scope: HashSet = self.fragment_ids.iter().copied().collect(); -+ let Some(coverage) = index_meta.fragment_bitmap.as_ref() else { -+ return Ok(Some("unknown_coverage")); -+ }; -+ if self -+ .fragment_ids -+ .iter() -+ .any(|id| u32::try_from(*id).map_or(true, |id| !coverage.contains(id))) -+ { -+ // Scan the ENTIRE explicit read domain, not just the covered part. -+ return Ok(Some("partial_coverage")); -+ } -+ if fragments -+ .iter() -+ .filter(|f| scope.contains(&(f.id() as u64))) -+ .any(|f| !f.metadata().overlays.is_empty() || f.metadata().physical_rows.is_none()) -+ { -+ return Ok(Some("fragment_state")); -+ } -+ // Fragment reuse can change the domain of an old segment. Until its -+ // coverage mapping is handled here, preserve correctness with a scoped scan. -+ if self.dataset.frag_reuse_index_uuid().await.is_some() { -+ return Ok(Some("fragment_reuse")); -+ } -+ let Some(filter) = reader.get_expr_filter()? else { -+ return Ok(Some("no_filter")); -+ }; -+ let planner = Planner::new(Arc::new(self.dataset.schema().into())); -+ let index_info = self.dataset.scalar_index_info().await?; -+ let filter_plan = planner.create_filter_plan(filter, &index_info, true)?; -+ let Some(search) = filter_plan -+ .index_query -+ .as_ref() -+ .and_then(|expr| driver(expr, &index_meta.name)) -+ else { -+ return Ok(Some("no_driver")); -+ }; -+ if search.column != field.name { -+ return Ok(Some("field_path")); -+ } -+ let index = self -+ .dataset -+ .open_scalar_index(&search.column, &self.segment_uuid, metrics) -+ .await?; -+ if !matches!(index.index_type(), IndexType::BTree | IndexType::Bitmap) { -+ return Ok(Some("index_type")); -+ } -+ // External masks use _rowid, not necessarily physical row addresses. -+ if index.results_are_row_addresses() && self.dataset.manifest.uses_stable_row_ids() { -+ return Ok(Some("row_id_domain")); -+ } -+ let started = Instant::now(); -+ let result = index.search(search.query.as_ref(), metrics).await?; -+ stats.all_times.insert( -+ "scalar_segment_search_time".into(), -+ started.elapsed().as_nanos().min(usize::MAX as u128) as usize, -+ ); -+ stats -+ .all_counts -+ .insert("scalar_segments_searched".into(), 1); -+ let SearchResult::Exact(rows) = result else { -+ return Ok(Some("inexact_result")); -+ }; -+ stats.all_counts.insert( -+ "scalar_segment_candidate_rows".into(), -+ rows.len().unwrap_or(0) as usize, -+ ); -+ // Do not truncate candidates at LIMIT. The reader evaluates the complete -+ // filter before applying its existing limit/offset operators. -+ reader.with_row_addr_prefilter(RowAddrMask::from_allowed(rows.selected_rows().clone())); -+ Ok(None) -+ } -+} -diff --git a/src/scanner.rs b/src/scanner.rs -index 4ceeb0e..ac3cff6 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -39,6 +39,7 @@ use crate::fts_query::{ - }; - use crate::helpers; - use crate::runtime::{RT, block_on}; -+use crate::scalar_segment::PreparedScalarSegment; - use crate::stream_guard::GuardedReader; - - /// Data type tag for query vectors, mirroring the C enum `LanceDataType`. -@@ -108,6 +109,7 @@ pub struct LanceScanner { - include_deleted_rows: bool, - fragment_ids: Option>, - index_segments: Option>, -+ scalar_index_segment: Option, - nearest: Option, - nprobes: NprobesRange, - approx_mode: Option, -@@ -256,6 +258,7 @@ impl LanceScanner { - include_deleted_rows: false, - fragment_ids: None, - index_segments: None, -+ scalar_index_segment: None, - nearest: None, - nprobes: NprobesRange::default(), - approx_mode: None, -@@ -470,12 +473,37 @@ impl LanceScanner { - None - }; - self.apply_filter(&mut scanner)?; -+ let scalar_segment = if let Some(segment_uuid) = self.scalar_index_segment { -+ if self.nearest.is_some() -+ || self.fts_query.is_some() -+ || self.fts_context.is_some() -+ || self.index_segments.is_some() -+ || self.fts_index_segments.is_some() -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ "scalar_index_segment requires an ordinary scan of live rows".into(), -+ )); -+ } -+ let fragment_ids = self.fragment_ids.as_ref().filter(|ids| !ids.is_empty()) -+ .ok_or_else(|| lance_core::Error::invalid_input_source( -+ "scalar_index_segment requires explicit nonempty fragment_ids for its read and fallback domain".into(), -+ ))?; -+ Some(PreparedScalarSegment { -+ dataset: Arc::clone(&self.dataset), -+ segment_uuid, -+ fragment_ids: fragment_ids.clone(), -+ callback: self.scan_statistics_callback.clone(), -+ }) -+ } else { -+ None -+ }; - if let Some(callback) = &self.scan_statistics_callback { - scanner.scan_stats_callback(callback.clone()); - } - Ok(PreparedScanner { - scanner, - distributed_fts, -+ scalar_segment, - }) - } - } -@@ -490,10 +518,18 @@ struct PreparedFtsExecution { - struct PreparedScanner { - scanner: lance::dataset::scanner::Scanner, - distributed_fts: Option, -+ scalar_segment: Option, - } - - impl PreparedScanner { - async fn try_into_stream(self) -> Result { -+ if let Some(scalar_segment) = self.scalar_segment { -+ return scalar_segment -+ .configure(self.scanner) -+ .await? -+ .try_into_stream() -+ .await; -+ } - let Some(distributed_fts) = self.distributed_fts else { - return self.scanner.try_into_stream().await; - }; -@@ -858,6 +894,31 @@ macro_rules! scanner_ffi_try { - }}; - } - -+/// Select one physical scalar index segment. NULL clears the selection. -+/// Requires explicit fragment_ids and an ordinary live-row scan. See the C header. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_scalar_index_segment( -+ scanner: *mut LanceScanner, -+ segment_uuid: *const u8, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, { -+ let scanner = unsafe { scanner.as_mut() } -+ .ok_or_else(|| lance_core::Error::invalid_input_source("scanner is NULL".into()))?; -+ scanner.ensure_scan_not_started("scalar_index_segment")?; -+ let segment = if segment_uuid.is_null() { -+ None -+ } else { -+ Some( -+ Uuid::from_slice(unsafe { std::slice::from_raw_parts(segment_uuid, 16) }) -+ .map_err(|e| lance_core::Error::invalid_input_source(e.into()))?, -+ ) -+ }; -+ scanner.scalar_index_segment = segment; -+ Ok(0) -+ }) -+} -+ - // --------------------------------------------------------------------------- - // Scanner lifecycle + builder - // --------------------------------------------------------------------------- -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 3b3424b..a8a7eab 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12513,3 +12513,252 @@ fn test_add_columns_stream_null_dataset_consumes_stream() { - assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); - assert_stream_consumed(&stream, &drop_count); - } -+ -+// Segment scans deliberately use an unprojected nullable key and a residual -+// predicate so a candidate LIMIT or loss of filter columns changes the answer. -+fn create_scalar_segment_fixture( -+ kind: lance_index::IndexType, -+ stable: bool, -+) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { -+ use lance::dataset::WriteParams; -+ use lance::index::DatasetIndexExt; -+ use lance_index::scalar::{BuiltinIndexType, ScalarIndexParams}; -+ let tmp = tempfile::tempdir().unwrap(); -+ let uri = tmp.path().join("segments").to_str().unwrap().to_owned(); -+ let uuids = lance_c::runtime::block_on(async { -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("key", DataType::Int32, true), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from_iter_values(0..12)), -+ Arc::new(Int32Array::from( -+ (0..12) -+ .map(|id| if id % 4 == 0 { None } else { Some(id % 3) }) -+ .collect::>(), -+ )), -+ ], -+ ) -+ .unwrap(); -+ let mut ds = Dataset::write( -+ arrow::record_batch::RecordBatchIterator::new(vec![Ok(batch)], schema), -+ &uri, -+ Some(WriteParams { -+ max_rows_per_file: 4, -+ enable_stable_row_ids: stable, -+ ..Default::default() -+ }), -+ ) -+ .await -+ .unwrap(); -+ let params = ScalarIndexParams::for_builtin(if kind == lance_index::IndexType::Bitmap { -+ BuiltinIndexType::Bitmap -+ } else { -+ BuiltinIndexType::BTree -+ }); -+ let fragments = ds.get_fragments(); -+ assert_eq!(fragments.len(), 3); -+ let mut segments = Vec::new(); -+ for fragment in fragments.iter().take(2) { -+ segments.push( -+ ds.create_index_builder(&["key"], kind, ¶ms) -+ .name("key_idx".into()) -+ .fragments(vec![fragment.id() as u32]) -+ .execute_uncommitted() -+ .await -+ .unwrap(), -+ ); -+ } -+ let uuids = segments.iter().map(|s| *s.uuid.as_bytes()).collect(); -+ ds.commit_existing_index_segments("key_idx", "key", segments) -+ .await -+ .unwrap(); -+ uuids -+ }); -+ (tmp, uri, uuids) -+} -+ -+fn scalar_segment_ids( -+ uri: &str, -+ uuid: &[u8; 16], -+ fragments: &[u64], -+ filter: &str, -+ limit: Option, -+ offset: i64, -+) -> (Vec, CapturedScanStatistics) { -+ let uri = c_str(uri); -+ let filter = c_str(filter); -+ let id = c_str("id"); -+ let columns = [id.as_ptr(), ptr::null()]; -+ let mut captured = CapturedScanStatistics::default(); -+ let mut ids = Vec::new(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ let scanner = lance_scanner_new(ds, columns.as_ptr(), filter.as_ptr()); -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, fragments.as_ptr(), fragments.len()), -+ 0 -+ ); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuid.as_ptr()), -+ 0 -+ ); -+ if let Some(limit) = limit { -+ assert_eq!(lance_scanner_set_limit(scanner, limit), 0); -+ } -+ assert_eq!(lance_scanner_set_offset(scanner, offset), 0); -+ assert_eq!( -+ lance_scanner_set_statistics_callback( -+ scanner, -+ Some(capture_scan_statistics), -+ (&mut captured as *mut CapturedScanStatistics).cast() -+ ), -+ 0 -+ ); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ let rc = lance_scanner_to_arrow_stream(scanner, &mut stream); -+ assert_eq!( -+ rc, -+ 0, -+ "{}", -+ if rc != 0 { -+ take_last_error_message() -+ } else { -+ String::new() -+ } -+ ); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, ptr::null()), -+ -1 -+ ); -+ { -+ let reader = ArrowArrayStreamReader::from_raw(&mut stream).unwrap(); -+ for batch in reader { -+ let batch = batch.unwrap(); -+ assert_eq!(batch.num_columns(), 1); -+ ids.extend( -+ batch -+ .column(0) -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .iter() -+ .copied(), -+ ); -+ } -+ } -+ lance_scanner_close(scanner); -+ lance_dataset_close(ds); -+ } -+ (ids, captured) -+} -+ -+#[test] -+fn test_scalar_segment_scope_residual_limit_and_unindexed_fallback() { -+ for kind in [ -+ lance_index::IndexType::BTree, -+ lance_index::IndexType::Bitmap, -+ ] { -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(kind, false); -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![2, 3]); -+ assert_eq!(stats.calls, 1); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| name == "scalar_segments_searched" && *value == 1) -+ ); -+ let (ids, _) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); -+ assert_eq!( -+ ids, -+ vec![3], -+ "offset and limit must apply after residual filtering" -+ ); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[1], &[1], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![5, 6, 7]); -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0, 2], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!( -+ ids, -+ vec![2, 3, 9, 10, 11], -+ "partial coverage must not omit unindexed rows" -+ ); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, _)| name == "scalar_segment_fallback_partial_coverage") -+ ); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], "key = 99 OR id = 0", None, 0); -+ assert_eq!( -+ ids, -+ vec![0], -+ "OR must not use just one branch as candidates" -+ ); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, _)| name == "scalar_segment_fallback_no_driver") -+ ); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0], "key = 99", None, 0); -+ assert!(ids.is_empty()); -+ } -+} -+ -+#[test] -+fn test_scalar_segment_stable_row_ids_and_deletes() { -+ use lance::index::DatasetIndexExt; -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, true); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(&uri).await.unwrap(); -+ ds.delete("id = 2").await.unwrap(); -+ assert_eq!(ds.load_indices().await.unwrap().len(), 2); -+ }); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![3]); -+} -+ -+#[test] -+fn test_scalar_segment_requires_explicit_domain_and_checks_uuid() { -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); -+ let uri = c_str(&uri); -+ let filter = c_str("key >= 0"); -+ unsafe { -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(ptr::null_mut(), ptr::null()), -+ -1 -+ ); -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuids[0].as_ptr()), -+ 0 -+ ); -+ let mut batch = ptr::null_mut(); -+ assert_eq!(lance_scanner_next(scanner, &mut batch), -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ lance_scanner_close(scanner); -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, [0u64].as_ptr(), 1), -+ 0 -+ ); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, [0u8; 16].as_ptr()), -+ 0 -+ ); -+ assert_eq!(lance_scanner_next(scanner, &mut batch), -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ lance_scanner_close(scanner); -+ lance_dataset_close(ds); -+ } -+} - -From 17240674d739293a494c07dfe5aec92a97185363 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Tue, 8 Sep 2026 23:17:23 +0800 -Subject: [PATCH 2/5] update - ---- - docs/scalar-segment-scans.md | 3 ++ - include/lance/lance.h | 4 +-- - src/scalar_segment.rs | 7 +++++ - tests/c_api_test.rs | 60 ++++++++++++++++++++++++++++++++++-- - 4 files changed, 70 insertions(+), 4 deletions(-) - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -index dc7c141..f72dbfc 100644 ---- a/docs/scalar-segment-scans.md -+++ b/docs/scalar-segment-scans.md -@@ -36,6 +36,9 @@ An unknown UUID, absent fragment or invalid option combination is an error. A - known segment with incomplete/unknown coverage, no suitable driver, unsupported - index type, nested key, overlays, fragment reuse, non-exact results or unsupported - row-ID domain falls back to a non-indexed scan of the entire explicit domain. -+Legacy (v1) storage also takes this fallback because ordinary scans cannot consume -+external row masks; it reports `scalar_segment_fallback_legacy_storage` without -+searching the index. - I/O and corruption errors are propagated, not converted to empty results or - successful fallback. - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index cbd8330..ab15247 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1870,8 +1870,8 @@ int32_t lance_scanner_set_index_segments( - * - * BTree/Bitmap searches use a necessary AND-conjunct of the full scanner filter - * on the selected logical index. All predicates are reapplied during candidate -- * reads; other scalar indices are disabled. OR/NOT-only filters, overlays, -- * fragment reuse, unsupported index types / result domains -+ * reads; other scalar indices are disabled. Legacy storage, OR/NOT-only filters, -+ * overlays, fragment reuse, unsupported index types / result domains - * and missing coverage use the same domain without an index. No filter also - * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the - * unfiltered candidate set. Vector/FTS queries are rejected. -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 2ab1297..5e121c9 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -138,6 +138,13 @@ impl PreparedScalarSegment { - self.dataset.schema().field_by_id(field_id).ok_or_else(|| { - invalid("scalar segment key field is absent from the dataset schema") - })?; -+ // Match Lance's plain-scan external-mask restriction. Keep the scoped, -+ // full-filtered reader intact and avoid index work on legacy storage. -+ if self.dataset.manifest().data_storage_format.lance_file_format() -+ == lance_file::version::ConcreteFileVersion::V1 -+ { -+ return Ok(Some("legacy_storage")); -+ } - // Keep V1 to flat scalar fields. A dotted name is not sufficient to prove - // the field path of an evolved or nested schema. - if !self -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index a8a7eab..0cb998b 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12519,6 +12519,15 @@ fn test_add_columns_stream_null_dataset_consumes_stream() { - fn create_scalar_segment_fixture( - kind: lance_index::IndexType, - stable: bool, -+) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { -+ create_scalar_segment_fixture_with_options(kind, stable, None, &[&[0], &[1]]) -+} -+ -+fn create_scalar_segment_fixture_with_options( -+ kind: lance_index::IndexType, -+ stable: bool, -+ storage_version: Option, -+ segment_fragments: &[&[u32]], - ) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { - use lance::dataset::WriteParams; - use lance::index::DatasetIndexExt; -@@ -12548,6 +12557,7 @@ fn create_scalar_segment_fixture( - Some(WriteParams { - max_rows_per_file: 4, - enable_stable_row_ids: stable, -+ data_storage_version: storage_version, - ..Default::default() - }), - ) -@@ -12561,11 +12571,11 @@ fn create_scalar_segment_fixture( - let fragments = ds.get_fragments(); - assert_eq!(fragments.len(), 3); - let mut segments = Vec::new(); -- for fragment in fragments.iter().take(2) { -+ for fragment_ids in segment_fragments { - segments.push( - ds.create_index_builder(&["key"], kind, ¶ms) - .name("key_idx".into()) -- .fragments(vec![fragment.id() as u32]) -+ .fragments(fragment_ids.to_vec()) - .execute_uncommitted() - .await - .unwrap(), -@@ -12714,6 +12724,52 @@ fn test_scalar_segment_scope_residual_limit_and_unindexed_fallback() { - } - } - -+#[test] -+fn test_scalar_segment_legacy_storage_falls_back() { -+ use lance_file::version::{ConcreteFileVersion, LanceFileVersion}; -+ -+ // Three fragments, with one segment covering 0 and 1. Reading only fragment -+ // 0 must retain the full predicate and must not leak rows from fragment 1. -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture_with_options( -+ lance_index::IndexType::BTree, -+ false, -+ Some(LanceFileVersion::Legacy), -+ &[&[0, 1]], -+ ); -+ assert_eq!(uuids.len(), 1); -+ lance_c::runtime::block_on(async { -+ let ds = Dataset::open(&uri).await.unwrap(); -+ assert_eq!( -+ ds.manifest().data_storage_format.lance_file_format(), -+ ConcreteFileVersion::V1 -+ ); -+ }); -+ -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![2, 3]); -+ assert_eq!(stats.calls, 1); -+ assert_eq!(stats.indices_loaded, 0); -+ assert_eq!(stats.index_comparisons, 0); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_legacy_storage" && *value == 1 -+ })); -+ assert!( -+ !stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| name == "scalar_segments_searched" && *value != 0) -+ ); -+ -+ let (ids, _) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); -+ assert_eq!( -+ ids, -+ vec![3], -+ "fallback must retain LIMIT/OFFSET after filtering" -+ ); -+} -+ - #[test] - fn test_scalar_segment_stable_row_ids_and_deletes() { - use lance::index::DatasetIndexExt; - -From f9ce263e4e32540c26afa6ccd18021ca81c090f1 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Tue, 8 Sep 2026 23:21:02 +0800 -Subject: [PATCH 3/5] update - ---- - src/scalar_segment.rs | 6 +++++- - tests/c_api_test.rs | 6 ++---- - 2 files changed, 7 insertions(+), 5 deletions(-) - -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 5e121c9..01df58b 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -140,7 +140,11 @@ impl PreparedScalarSegment { - })?; - // Match Lance's plain-scan external-mask restriction. Keep the scoped, - // full-filtered reader intact and avoid index work on legacy storage. -- if self.dataset.manifest().data_storage_format.lance_file_format() -+ if self -+ .dataset -+ .manifest() -+ .data_storage_format -+ .lance_file_format() - == lance_file::version::ConcreteFileVersion::V1 - { - return Ok(Some("legacy_storage")); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 0cb998b..efd4284 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12745,8 +12745,7 @@ fn test_scalar_segment_legacy_storage_falls_back() { - ); - }); - -- let (ids, stats) = -- scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); - assert_eq!(ids, vec![2, 3]); - assert_eq!(stats.calls, 1); - assert_eq!(stats.indices_loaded, 0); -@@ -12761,8 +12760,7 @@ fn test_scalar_segment_legacy_storage_falls_back() { - .any(|(name, _, value)| name == "scalar_segments_searched" && *value != 0) - ); - -- let (ids, _) = -- scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); - assert_eq!( - ids, - vec![3], - -From 221d8800f56790ab8b48b12f3aaaf9da06359380 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Wed, 9 Sep 2026 12:59:08 +0800 -Subject: [PATCH 4/5] add LabelList - ---- - docs/scalar-segment-scans.md | 12 ++- - include/lance/lance.h | 8 +- - include/lance/lance.hpp | 4 +- - src/scalar_segment.rs | 7 +- - tests/c_api_test.rs | 190 ++++++++++++++++++++++++++++++++--- - 5 files changed, 200 insertions(+), 21 deletions(-) - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -index f72dbfc..72f6188 100644 ---- a/docs/scalar-segment-scans.md -+++ b/docs/scalar-segment-scans.md -@@ -1,10 +1,18 @@ - # Scalar index segment scans - --An ordinary scanner can use one physical BTree/Bitmap segment to generate --candidates, then read those candidates with the complete scanner filter. This -+An ordinary scanner can use one physical BTree, Bitmap, or LabelList segment to -+generate candidates, then read them with the complete scanner filter. This - does not run a global search of the other segments of the logical index. It does - not subdivide a physical segment or make its own index search incremental. - -+LabelList supports indexed array membership predicates. Every candidate search -+must return `SearchResult::Exact`. -+LabelList query values should match the array element type, for example -+`array_contains(int32_labels, CAST(42 AS INT))`; a cast on the indexed column -+can prevent the planner from finding an index driver and cause fallback. -+`AtMost` and `AtLeast` results still fall back; this mode does not enable FMIndex, -+NGram, BloomFilter, ZoneMap, or Inverted indices. -+ - ## Configuring a task - - Open a fixed dataset version. Select the physical index UUID from that version's -diff --git a/include/lance/lance.h b/include/lance/lance.h -index ab15247..b5c6902 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1868,9 +1868,11 @@ int32_t lance_scanner_set_index_segments( - * scan of those fragment_ids. Callers distributing work must assign disjoint - * fragment domains and separately include any unindexed data they wish to read. - * -- * BTree/Bitmap searches use a necessary AND-conjunct of the full scanner filter -- * on the selected logical index. All predicates are reapplied during candidate -- * reads; other scalar indices are disabled. Legacy storage, OR/NOT-only filters, -+ * BTree/Bitmap/LabelList searches use a necessary AND-conjunct of the -+ * full scanner filter on the selected logical index and require an Exact result. -+ * AtMost/AtLeast results fall back to a full filtered scan of fragment_ids. -+ * All predicates are reapplied during candidate reads; other scalar indices -+ * are disabled. Legacy storage, OR/NOT-only filters, - * overlays, fragment reuse, unsupported index types / result domains - * and missing coverage use the same domain without an index. No filter also - * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 8e1c6cf..ebb8141 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1311,8 +1311,8 @@ class Scanner { - return *this; - } - -- /// Restrict scalar candidate generation to one segment; fragment_ids is -- /// required and defines the complete read/fallback domain. See lance.h. -+ /// Generate exact candidates from one BTree/Bitmap/LabelList segment. -+ /// fragment_ids is required and defines the complete read/fallback domain. See lance.h. - Scanner& scalar_index_segment(const std::array& segment_uuid) { - if (lance_scanner_set_scalar_index_segment(handle_.get(), segment_uuid.data()) != 0) - check_error(); -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 01df58b..1faf500 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -204,7 +204,12 @@ impl PreparedScalarSegment { - .dataset - .open_scalar_index(&search.column, &self.segment_uuid, metrics) - .await?; -- if !matches!(index.index_type(), IndexType::BTree | IndexType::Bitmap) { -+ // These implementations can return exact candidates. Keep the runtime -+ // Exact check below: a type alone is not a guarantee for every query. -+ if !matches!( -+ index.index_type(), -+ IndexType::BTree | IndexType::Bitmap | IndexType::LabelList -+ ) { - return Ok(Some("index_type")); - } - // External masks use _rowid, not necessarily physical row addresses. -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index efd4284..3322d42 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12528,6 +12528,21 @@ fn create_scalar_segment_fixture_with_options( - stable: bool, - storage_version: Option, - segment_fragments: &[&[u32]], -+) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { -+ let key = Arc::new(Int32Array::from( -+ (0..12) -+ .map(|id| if id % 4 == 0 { None } else { Some(id % 3) }) -+ .collect::>(), -+ )); -+ create_scalar_segment_fixture_from_key(kind, stable, storage_version, segment_fragments, key) -+} -+ -+fn create_scalar_segment_fixture_from_key( -+ kind: lance_index::IndexType, -+ stable: bool, -+ storage_version: Option, -+ segment_fragments: &[&[u32]], -+ key: arrow_array::ArrayRef, - ) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { - use lance::dataset::WriteParams; - use lance::index::DatasetIndexExt; -@@ -12537,17 +12552,14 @@ fn create_scalar_segment_fixture_with_options( - let uuids = lance_c::runtime::block_on(async { - let schema = Arc::new(Schema::new(vec![ - Field::new("id", DataType::Int32, false), -- Field::new("key", DataType::Int32, true), -+ Field::new("key", key.data_type().clone(), true), - ])); -+ let row_count = key.len(); - let batch = RecordBatch::try_new( - schema.clone(), - vec![ -- Arc::new(Int32Array::from_iter_values(0..12)), -- Arc::new(Int32Array::from( -- (0..12) -- .map(|id| if id % 4 == 0 { None } else { Some(id % 3) }) -- .collect::>(), -- )), -+ Arc::new(Int32Array::from_iter_values(0..row_count as i32)), -+ key, - ], - ) - .unwrap(); -@@ -12563,13 +12575,9 @@ fn create_scalar_segment_fixture_with_options( - ) - .await - .unwrap(); -- let params = ScalarIndexParams::for_builtin(if kind == lance_index::IndexType::Bitmap { -- BuiltinIndexType::Bitmap -- } else { -- BuiltinIndexType::BTree -- }); -+ let params = ScalarIndexParams::for_builtin(BuiltinIndexType::try_from(kind).unwrap()); - let fragments = ds.get_fragments(); -- assert_eq!(fragments.len(), 3); -+ assert_eq!(fragments.len(), row_count.div_ceil(4)); - let mut segments = Vec::new(); - for fragment_ids in segment_fragments { - segments.push( -@@ -12724,6 +12732,162 @@ fn test_scalar_segment_scope_residual_limit_and_unindexed_fallback() { - } - } - -+#[test] -+fn test_scalar_segment_label_list_exact_candidates() { -+ use arrow_array::builder::{Int32Builder, ListBuilder}; -+ use lance::index::DatasetIndexExt; -+ use lance_index::IndexType; -+ -+ for stable in [false, true] { -+ let mut lists = ListBuilder::new(Int32Builder::new()); -+ for row in 0..16 { -+ match row { -+ 0 | 9 | 13 => lists.append(false), -+ 1 | 10 | 14 => lists.append(true), -+ 4 => { -+ lists.values().append_value(7); -+ lists.append(true); -+ } -+ _ => { -+ lists.values().append_value(42); -+ if row == 3 || row == 11 || row == 15 { -+ lists.values().append_value(7); -+ } -+ if row == 6 { -+ lists.values().append_null(); -+ } -+ lists.append(true); -+ } -+ } -+ } -+ // S0 covers fragments 0 and 1, S1 covers 2, and 3 is unindexed. -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture_from_key( -+ IndexType::LabelList, -+ stable, -+ None, -+ &[&[0, 1], &[2]], -+ Arc::new(lists.finish()), -+ ); -+ let predicate = "array_contains(key, CAST(42 AS INT))"; -+ let filter = format!("{predicate} AND id >= 3"); -+ for (fragments, expected) in [(vec![0, 1], vec![3, 5, 6, 7]), (vec![0], vec![3])] { -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &fragments, &filter, None, 0); -+ assert_eq!(ids, expected, "stable={stable}"); -+ assert_eq!(stats.calls, 1); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| { name == "scalar_segments_searched" && *value == 1 }), -+ "stable={stable}, metrics={:?}", -+ stats.metrics -+ ); -+ assert!( -+ !stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| { name == "scalar_segment_fallbacks" && *value != 0 }) -+ ); -+ } -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0, 1], &filter, Some(1), 1); -+ assert_eq!(ids, vec![5], "limit/offset must follow the residual filter"); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[1], &[2], &filter, None, 0); -+ assert_eq!(ids, vec![8, 11]); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0, 3], &filter, None, 0); -+ assert_eq!(ids, vec![3, 12, 15]); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_partial_coverage" && *value == 1 -+ })); -+ let (ids, stats) = scalar_segment_ids( -+ &uri, -+ &uuids[0], -+ &[0], -+ &format!("{predicate} OR id = 0"), -+ None, -+ 0, -+ ); -+ assert_eq!(ids, vec![0, 2, 3]); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_no_driver" && *value == 1 -+ })); -+ -+ for (predicate, expected) in [ -+ ( -+ "array_has_all(key, [CAST(42 AS INT), CAST(7 AS INT)])", -+ vec![3], -+ ), -+ ( -+ "array_has_any(key, [CAST(42 AS INT), CAST(99 AS INT)])", -+ vec![2, 3, 5, 6, 7], -+ ), -+ ("array_contains(key, CAST(99 AS INT))", vec![]), -+ ("array_contains(key, CAST(NULL AS INT))", vec![]), -+ ("array_has_any(key, [])", vec![]), -+ ] { -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0, 1], predicate, None, 0); -+ assert_eq!(ids, expected, "{predicate}, stable={stable}"); -+ } -+ // An untyped integer literal casts this Int32 list to Int64. Such -+ // a column expression must retain the scan fallback. -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0, 1], "array_contains(key, 42)", None, 0); -+ assert_eq!(ids, vec![2, 3, 5, 6, 7]); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_no_driver" && *value == 1 -+ })); -+ -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(&uri).await.unwrap(); -+ ds.delete("id = 3").await.unwrap(); -+ assert_eq!(ds.load_indices().await.unwrap().len(), 2); -+ }); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0, 1], &filter, None, 0); -+ assert_eq!(ids, vec![5, 6, 7]); -+ } -+} -+ -+#[test] -+fn test_scalar_segment_text_indices_still_fall_back() { -+ for kind in [lance_index::IndexType::Fm, lance_index::IndexType::NGram] { -+ for stable in [false, true] { -+ let key = Arc::new(StringArray::from(vec![ -+ Some("needle"), -+ None, -+ Some(""), -+ Some("other"), -+ Some("needle"), -+ Some("other"), -+ Some(""), -+ None, -+ ])); -+ let (_tmp, uri, uuids) = -+ create_scalar_segment_fixture_from_key(kind, stable, None, &[&[0, 1]], key); -+ for (predicate, expected) in [ -+ ("contains(key, 'needle')", vec![0]), -+ ("contains(key, '')", vec![0, 2, 3]), -+ ] { -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], predicate, None, 0); -+ assert_eq!(ids, expected, "{kind:?}, stable={stable}, {predicate}"); -+ assert!( -+ stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallbacks" && *value == 1 -+ }) -+ ); -+ if predicate == "contains(key, 'needle')" { -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_index_type" && *value == 1 -+ })); -+ } -+ assert!( -+ !stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segments_searched" && *value != 0 -+ }) -+ ); -+ } -+ } -+ } -+} -+ - #[test] - fn test_scalar_segment_legacy_storage_falls_back() { - use lance_file::version::{ConcreteFileVersion, LanceFileVersion}; - -From 30dda06a1cc6bad37d04ed207115409ccef91715 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Wed, 9 Sep 2026 13:27:54 +0800 -Subject: [PATCH 5/5] update - ---- - docs/scalar-segment-scans.md | 55 +++++++++-- - include/lance/lance.h | 16 ++- - include/lance/lance.hpp | 5 + - src/scalar_segment.rs | 12 ++- - src/scanner.rs | 13 ++- - tests/c_api_test.rs | 187 +++++++++++++++++++++++++++++++++++ - 6 files changed, 272 insertions(+), 16 deletions(-) - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -index 72f6188..efd0498 100644 ---- a/docs/scalar-segment-scans.md -+++ b/docs/scalar-segment-scans.md -@@ -27,11 +27,19 @@ lance_scanner_set_limit(scanner, 20000); - /* The scanner-owning thread calls lance_scanner_next as usual. */ - ``` - -+The segment setter copies the UUID; passing NULL clears it. Final option -+compatibility and snapshot metadata are checked when preparing the stream, not -+by the setter. Configure all options before the first `next`, Arrow stream -+export, or asynchronous scan. A failed preparation also freezes the options; -+create a new scanner to retry with different settings. -+ - SQL, Substrait and additional SQL filters keep their existing precedence and AND - composition. The caller does not supply a separate driver predicate: Lance-C - uses the typed filter planner and selects a necessary indexed leaf belonging to - the requested logical index. It only descends through AND, never through OR or --NOT. It then searches the selected UUID and applies the complete filter while -+NOT, and chooses the first matching leaf in the planner's expression tree; -+this is not a selectivity-based choice or a guarantee of SQL text order. -+It then searches the selected UUID and applies the complete filter while - reading candidates with automatic scalar-index planning disabled. - - Each task's fragment IDs define its result domain, including on fallback. A -@@ -40,9 +48,16 @@ scan. Unindexed fragments need their own tasks, or an explicit domain including - them (which causes that task to use fallback). Merely listing indexed segments - does not include appended, unindexed data automatically. - --An unknown UUID, absent fragment or invalid option combination is an error. A --known segment with incomplete/unknown coverage, no suitable driver, unsupported --index type, nested key, overlays, fragment reuse, non-exact results or unsupported -+`use_scalar_index=false` disables segment search regardless of setter order. -+The scanner validates the selected snapshot UUID and fragment domain, then scans -+that domain with the full filter and LIMIT/OFFSET without opening the index or -+generating candidates. It reports `scalar_segment_fallback_disabled`. -+ -+An unknown UUID, absent fragment, invalid option combination or segment metadata -+without one valid schema key field is an error. A known segment with -+incomplete/unknown coverage, no suitable driver, unsupported -+index type, nested key, overlays, unknown physical row counts, fragment reuse, -+non-exact results or unsupported - row-ID domain falls back to a non-indexed scan of the entire explicit domain. - Legacy (v1) storage also takes this fallback because ordinary scans cannot consume - external row masks; it reports `scalar_segment_fallback_legacy_storage` without -@@ -50,8 +65,15 @@ searching the index. - I/O and corruption errors are propagated, not converted to empty results or - successful fallback. - --The first implementation supports live-row ordinary scans and cannot be combined --with vector/FTS queries. Physical row-address -+The first implementation supports live-row ordinary scans and rejects vector/FTS -+queries and `include_deleted_rows=true` at stream creation, even when -+`use_scalar_index=false`. An index built after a delete does not contain the -+tombstoned rows, so even exact segment candidates -+cannot satisfy a scan that includes deleted rows. To read those rows, clear the -+segment setting and use an ordinary scan with `with_row_id=true`, -+`include_deleted_rows=true`, and `use_scalar_index=false`. -+Fragments removed from the current snapshot are not scanned by this option. -+Physical row-address - results on stable-row-ID datasets currently fall back; results already expressed - in the correct row-ID domain use the candidate path. Deletes and all remaining - predicates are handled by the ordinary reader. No candidate-count limit is -@@ -84,6 +106,23 @@ Successful exhaustion merges segment-search metrics into the existing statistics - callback exactly once. New metrics include `scalar_segments_requested`, - `scalar_segments_searched`, `scalar_segment_candidate_rows`, - `scalar_segment_prepare_time`, `scalar_segment_search_time`, and --`scalar_segment_fallback_*` reasons. `prepare_time` includes search time. Early --release, cancellation and errors retain the existing callback contract: final -+`scalar_segment_fallbacks` plus `scalar_segment_fallback_*` reasons. -+`scalar_segment_prepare_time` includes search time. Metrics describe each -+successfully exhausted stream, including separately exported streams: -+ -+- `scalar_segments_requested` is 1 even on fallback. -+- `scalar_segments_searched` is 1 after a completed index search, including one -+ whose inexact result causes fallback. A fallback can therefore include index work. -+- `scalar_segment_candidate_rows` counts TRUE rows in the exact segment result before fragment -+ restriction, residual filtering, deletion handling and LIMIT/OFFSET. It is not -+ the output or physical-read row count; a result without a known cardinality is -+ reported as 0. The reader's mask may also include NULL candidates that the full -+ filter subsequently discards. -+- A fallback records one reason, the first eligibility check that fails. Disabled -+ scalar indices and legacy storage bypass index opening and search after snapshot -+ validation. Other fallback reasons may be found after opening or searching an index. -+- Search and candidate metrics may be absent when their stage did not execute; -+ consumers should treat absent counts as 0. -+ -+Early release, cancellation and errors retain the existing callback contract: final - statistics are not guaranteed. Metrics do not establish global task concurrency. -diff --git a/include/lance/lance.h b/include/lance/lance.h -index b5c6902..ed5bb6c 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1015,7 +1015,9 @@ int32_t lance_scanner_set_scan_in_order(LanceScanner* scanner, bool scan_in_orde - * Configure whether scalar indices may be used to optimize filters. - * - * Scalar indices are enabled by default. Disable this to force filter -- * evaluation without scalar indices. This setting is independent of -+ * evaluation without scalar indices, including an explicitly selected scalar -+ * segment (which falls back to a scan of its explicit fragment_ids). -+ * This setting is independent of - * `lance_scanner_set_use_index`, which controls vector ANN index usage. - * Must be set before scanning starts. - */ -@@ -1051,7 +1053,11 @@ int32_t lance_scanner_with_row_address(LanceScanner* scanner, bool enable); - - /** - * Configure whether deleted rows still present in storage are returned. -- * Deleted rows have a NULL `_rowid`; callers should also enable row IDs. -+ * Requires with_row_id=true; deleted rows have a NULL `_rowid`. -+ * For filtered scans, also set use_scalar_index=false: indices built after a -+ * deletion may omit tombstoned rows. Incompatible with scalar_index_segment, -+ * even when scalar indices are disabled. -+ * Fragments removed from the current snapshot are not scanned. - * Must be set before scanning starts. - */ - int32_t lance_scanner_set_include_deleted_rows( -@@ -1867,16 +1873,20 @@ int32_t lance_scanner_set_index_segments( - * excluded by fragment_ids; incomplete coverage falls back to a full filtered - * scan of those fragment_ids. Callers distributing work must assign disjoint - * fragment domains and separately include any unindexed data they wish to read. -+ * The segment metadata must identify one key field present in the schema. - * - * BTree/Bitmap/LabelList searches use a necessary AND-conjunct of the - * full scanner filter on the selected logical index and require an Exact result. -+ * use_scalar_index=false skips segment search and uses the scoped fallback; -+ * snapshot UUID and fragment validation still applies. - * AtMost/AtLeast results fall back to a full filtered scan of fragment_ids. - * All predicates are reapplied during candidate reads; other scalar indices - * are disabled. Legacy storage, OR/NOT-only filters, - * overlays, fragment reuse, unsupported index types / result domains - * and missing coverage use the same domain without an index. No filter also - * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the -- * unfiltered candidate set. Vector/FTS queries are rejected. -+ * unfiltered candidate set. Vector/FTS queries and include_deleted_rows=true -+ * are rejected even when use_scalar_index=false; segment mode is live-row-only. - * - * UUID bytes are copied. Metadata and final option compatibility are validated - * when creating the stream. Index corruption or I/O failures remain errors. -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index ebb8141..d96cf9f 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1270,6 +1270,7 @@ class Scanner { - } - - /// Configure whether scalar indices may be used to optimize filters. -+ /// False also disables explicit scalar segment search, retaining its fragment domain. - Scanner& use_scalar_index(bool enable = true) { - if (lance_scanner_set_use_scalar_index(handle_.get(), enable) != 0) - check_error(); -@@ -1305,6 +1306,8 @@ class Scanner { - } - - /// Configure whether deleted rows still present in storage are returned. -+ /// Requires with_row_id(true); use_scalar_index(false) is needed for filtered scans. -+ /// Incompatible with scalar_index_segment. See lance.h. - Scanner& include_deleted_rows(bool include_deleted_rows = true) { - if (lance_scanner_set_include_deleted_rows(handle_.get(), include_deleted_rows) != 0) - check_error(); -@@ -1313,6 +1316,8 @@ class Scanner { - - /// Generate exact candidates from one BTree/Bitmap/LabelList segment. - /// fragment_ids is required and defines the complete read/fallback domain. See lance.h. -+ /// Requires live rows only: include_deleted_rows(true) is rejected at stream creation. -+ /// use_scalar_index(false) selects the scoped fallback without searching the segment. - Scanner& scalar_index_segment(const std::array& segment_uuid) { - if (lance_scanner_set_scalar_index_segment(handle_.get(), segment_uuid.data()) != 0) - check_error(); -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 1faf500..747dac6 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -27,6 +27,7 @@ pub(crate) struct PreparedScalarSegment { - pub dataset: Arc, - pub segment_uuid: Uuid, - pub fragment_ids: Vec, -+ pub use_scalar_index: bool, - pub callback: Option, - } - -@@ -138,6 +139,11 @@ impl PreparedScalarSegment { - self.dataset.schema().field_by_id(field_id).ok_or_else(|| { - invalid("scalar segment key field is absent from the dataset schema") - })?; -+ // Explicitly disabling scalar indices also disables this accelerator. -+ // Keep snapshot validation above, but do not plan, open or search an index. -+ if !self.use_scalar_index { -+ return Ok(Some("disabled")); -+ } - // Match Lance's plain-scan external-mask restriction. Keep the scoped, - // full-filtered reader intact and avoid index work on legacy storage. - if self -@@ -149,8 +155,8 @@ impl PreparedScalarSegment { - { - return Ok(Some("legacy_storage")); - } -- // Keep V1 to flat scalar fields. A dotted name is not sufficient to prove -- // the field path of an evolved or nested schema. -+ // Keep this implementation to flat scalar fields. A dotted name cannot -+ // prove the field path of an evolved or nested schema. - if !self - .dataset - .schema() -@@ -234,6 +240,8 @@ impl PreparedScalarSegment { - ); - // Do not truncate candidates at LIMIT. The reader evaluates the complete - // filter before applying its existing limit/offset operators. -+ // The raw selected bitmap can overlap NULL rows; the full filter removes -+ // those as well. The metric above counts semantic TRUE rows, not mask size. - reader.with_row_addr_prefilter(RowAddrMask::from_allowed(rows.selected_rows().clone())); - Ok(None) - } -diff --git a/src/scanner.rs b/src/scanner.rs -index ac3cff6..bda554d 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -479,9 +479,10 @@ impl LanceScanner { - || self.fts_context.is_some() - || self.index_segments.is_some() - || self.fts_index_segments.is_some() -+ || self.include_deleted_rows - { - return Err(lance_core::Error::invalid_input_source( -- "scalar_index_segment requires an ordinary scan of live rows".into(), -+ "scalar_index_segment requires an ordinary scan of live rows; vector/FTS queries and include_deleted_rows=true are unsupported".into(), - )); - } - let fragment_ids = self.fragment_ids.as_ref().filter(|ids| !ids.is_empty()) -@@ -492,6 +493,7 @@ impl LanceScanner { - dataset: Arc::clone(&self.dataset), - segment_uuid, - fragment_ids: fragment_ids.clone(), -+ use_scalar_index: self.use_scalar_index.unwrap_or(true), - callback: self.scan_statistics_callback.clone(), - }) - } else { -@@ -896,6 +898,8 @@ macro_rules! scanner_ffi_try { - - /// Select one physical scalar index segment. NULL clears the selection. - /// Requires explicit fragment_ids and an ordinary live-row scan. See the C header. -+/// include_deleted_rows=true is rejected when preparing the scan, even if -+/// use_scalar_index=false selects the scoped non-indexed fallback. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_scalar_index_segment( - scanner: *mut LanceScanner, -@@ -1239,7 +1243,8 @@ unsafe fn scanner_set_scan_in_order_inner( - /// Configure whether scalar indices may be used to optimize filters. - /// - /// Scalar indices are enabled by default in Lance. Must be set before the scan --/// starts. -+/// starts. False also disables explicit scalar segment search while preserving -+/// the configured fragment domain and snapshot validation. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_use_scalar_index( - scanner: *mut LanceScanner, -@@ -1381,7 +1386,9 @@ unsafe fn scanner_with_row_address_inner(scanner: *mut LanceScanner, enable: boo - - /// Configure whether deleted rows still present in storage are returned. - /// --/// Deleted rows have a NULL `_rowid`, so callers should also enable row IDs. -+/// Requires with_row_id=true; deleted rows have a NULL `_rowid`. -+/// Filtered scans also need use_scalar_index=false because indices may omit -+/// tombstoned rows. Incompatible with scalar_index_segment. - /// Must be set before the scan starts. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_include_deleted_rows( -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 3322d42..513f8d0 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12945,6 +12945,193 @@ fn test_scalar_segment_stable_row_ids_and_deletes() { - assert_eq!(ids, vec![3]); - } - -+#[test] -+fn test_scalar_segment_honors_use_scalar_index_false() { -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0", None, 0); -+ assert_eq!(ids, vec![1, 2, 3]); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| name == "scalar_segments_searched" && *value == 1) -+ ); -+ -+ let uri = c_str(&uri); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ for disable_first in [false, true] { -+ for (fragments, filter, limit, offset, expected) in [ -+ (vec![0u64], "key >= 0", None, 0, vec![1, 2, 3]), -+ (vec![0, 2], "key >= 0 AND id >= 2", Some(2), 1, vec![3, 9]), -+ ] { -+ let filter = c_str(filter); -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, fragments.as_ptr(), fragments.len()), -+ 0 -+ ); -+ if disable_first { -+ assert_eq!(lance_scanner_set_use_scalar_index(scanner, false), 0); -+ } -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuids[0].as_ptr()), -+ 0 -+ ); -+ if !disable_first { -+ assert_eq!(lance_scanner_set_use_scalar_index(scanner, false), 0); -+ } -+ if let Some(limit) = limit { -+ assert_eq!(lance_scanner_set_limit(scanner, limit), 0); -+ } -+ assert_eq!(lance_scanner_set_offset(scanner, offset), 0); -+ let mut captured = CapturedScanStatistics::default(); -+ assert_eq!( -+ lance_scanner_set_statistics_callback( -+ scanner, -+ Some(capture_scan_statistics), -+ (&mut captured as *mut CapturedScanStatistics).cast(), -+ ), -+ 0 -+ ); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scanner, &mut stream), 0); -+ let mut ids = Vec::new(); -+ for batch in ArrowArrayStreamReader::from_raw(&mut stream).unwrap() { -+ let batch = batch.unwrap(); -+ ids.extend_from_slice( -+ batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values(), -+ ); -+ } -+ assert_eq!(ids, expected); -+ assert_eq!(captured.calls, 1); -+ for metric in ["scalar_segments_searched", "scalar_segment_candidate_rows"] { -+ assert_eq!( -+ captured -+ .metrics -+ .iter() -+ .filter(|(name, _, _)| name == metric) -+ .map(|(_, _, value)| *value) -+ .sum::(), -+ 0, -+ "{metric}" -+ ); -+ } -+ assert_eq!(captured.indices_loaded, 0); -+ assert_eq!(captured.index_comparisons, 0); -+ assert!(captured.metrics.iter().any(|(name, _, value)| name -+ == "scalar_segment_fallback_disabled" -+ && *value == 1)); -+ lance_scanner_close(scanner); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn test_scalar_segment_rejects_include_deleted_rows_after_index_rebuild() { -+ use lance::index::DatasetIndexExt; -+ use lance_index::scalar::{BuiltinIndexType, ScalarIndexParams}; -+ -+ let (_tmp, uri, _) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); -+ let uuid = lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(&uri).await.unwrap(); -+ ds.delete("id = 2").await.unwrap(); -+ ds.drop_index("key_idx").await.unwrap(); -+ // A segment built after the delete cannot return the tombstoned row, -+ // even though its search result is Exact for the indexed live rows. -+ let params = ScalarIndexParams::for_builtin(BuiltinIndexType::BTree); -+ let segment = ds -+ .create_index_builder(&["key"], lance_index::IndexType::BTree, ¶ms) -+ .name("key_idx".into()) -+ .fragments(vec![0]) -+ .execute_uncommitted() -+ .await -+ .unwrap(); -+ let uuid = *segment.uuid.as_bytes(); -+ ds.commit_existing_index_segments("key_idx", "key", vec![segment]) -+ .await -+ .unwrap(); -+ uuid -+ }); -+ -+ let (ids, _) = scalar_segment_ids(&uri, &uuid, &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![3], "live-row segment scans remain supported"); -+ -+ let uri = c_str(&uri); -+ let filter = c_str("key >= 0 AND id >= 2"); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ // Check both setter orders: compatibility is validated at stream creation. -+ for segment_first in [None, Some(false), Some(true)] { -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, [0u64].as_ptr(), 1), -+ 0 -+ ); -+ assert_eq!(lance_scanner_with_row_id(scanner, true), 0); -+ if segment_first.is_none() { -+ assert_eq!(lance_scanner_set_use_scalar_index(scanner, false), 0); -+ } -+ if segment_first == Some(true) { -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuid.as_ptr()), -+ 0 -+ ); -+ } -+ assert_eq!(lance_scanner_set_include_deleted_rows(scanner, true), 0); -+ if segment_first == Some(false) { -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuid.as_ptr()), -+ 0 -+ ); -+ } -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ let rc = lance_scanner_to_arrow_stream(scanner, &mut stream); -+ if segment_first.is_some() { -+ assert_eq!(rc, -1, "segment scans must not silently omit deleted rows"); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("include_deleted_rows=true")); -+ } else { -+ assert_eq!(rc, 0); -+ let reader = ArrowArrayStreamReader::from_raw(&mut stream).unwrap(); -+ let mut ids = Vec::new(); -+ for batch in reader { -+ let batch = batch.unwrap(); -+ ids.extend_from_slice( -+ batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values(), -+ ); -+ } -+ ids.sort_unstable(); -+ assert_eq!( -+ ids, -+ vec![2, 3], -+ "ordinary scans can still read tombstoned rows" -+ ); -+ } -+ lance_scanner_close(scanner); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ - #[test] - fn test_scalar_segment_requires_explicit_domain_and_checks_uuid() { - let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); diff --git a/thirdparty/patches/lance-c-0.1.9-pr-80.patch b/thirdparty/patches/lance-c-0.1.9-pr-80.patch deleted file mode 100644 index edc42cad0a1271..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-80.patch +++ /dev/null @@ -1,225 +0,0 @@ -From 7fcd9c4ff7c03c10bdc9d8a600b3f7b0cf28a9ad Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 10 Sep 2026 15:50:53 +0800 -Subject: [PATCH] oss provider error - -Doris integration: rebase only Cargo.toml/Cargo.lock hunk context over PR #73. -All added and removed lines are identical to upstream PR #80 at 7fcd9c4. - ---- - Cargo.lock | 1 + - Cargo.toml | 2 + - src/runtime.rs | 5 ++ - tests/compile_and_run_test.rs | 16 ++++++ - tests/cpp/test_oss_transport.c | 40 +++++++++++++ - tests/static_oss_transport_test.py | 91 ++++++++++++++++++++++++++++++ - 6 files changed, 155 insertions(+) - create mode 100644 tests/cpp/test_oss_transport.c - create mode 100644 tests/static_oss_transport_test.py - -diff --git a/Cargo.lock b/Cargo.lock ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -3970,6 +3970,7 @@ - "libc", - "log", - "object_store", -+ "opendal", - "pin-project", - "prost", - "snafu", -diff --git a/Cargo.toml b/Cargo.toml ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -43,6 +43,8 @@ - log = "0.4" - libc = "0.2" - object_store = "0.13.2" -+# Explicitly install the HTTP transport when embedded in a static C/C++ executable. -+opendal = { version = "=0.58.2", default-features = false, features = ["http-transport-reqwest"] } - pin-project = "1.0" - prost = "0.14" - snafu = "0.9" -diff --git a/src/runtime.rs b/src/runtime.rs -index 0153d3f..3bd8964 100644 ---- a/src/runtime.rs -+++ b/src/runtime.rs -@@ -8,6 +8,11 @@ use std::sync::LazyLock; - /// Global multi-threaded Tokio runtime, shared across all FFI calls. - /// Initialized lazily on first access. - pub static RT: LazyLock = LazyLock::new(|| { -+ // A native linker can omit OpenDAL's automatic constructor from liblance_c.a. -+ // Keep initialization reachable from the FFI entry points, before any HTTP I/O. -+ // Installation is idempotent and preserves an already installed transport. -+ opendal::install_default(); -+ - tokio::runtime::Builder::new_multi_thread() - .enable_all() - .build() -diff --git a/tests/compile_and_run_test.rs b/tests/compile_and_run_test.rs -index b419ac9..8566d10 100644 ---- a/tests/compile_and_run_test.rs -+++ b/tests/compile_and_run_test.rs -@@ -249,3 +249,19 @@ fn test_cpp_compilation_and_execution() { - - run_test_binary(&binary, &dataset_uri, &write_uri); - } -+ -+/// A fresh C executable must initialize OpenDAL even when archive constructors are omitted. -+#[cfg(target_os = "linux")] -+#[test] -+#[ignore = "requires a C compiler, Python 3, and building the static library"] -+fn test_static_oss_transport() { -+ let (shared_library, _) = build_lance_c(); -+ let static_library = shared_library.with_file_name("liblance_c.a"); -+ assert!(static_library.exists(), "static library was not built"); -+ let status = Command::new("python3") -+ .arg(Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/static_oss_transport_test.py")) -+ .arg(static_library) -+ .status() -+ .expect("failed to run the static OSS transport test"); -+ assert!(status.success(), "static OSS HTTP transport test failed"); -+} -diff --git a/tests/cpp/test_oss_transport.c b/tests/cpp/test_oss_transport.c -new file mode 100644 -index 0000000..c96ee7e ---- /dev/null -+++ b/tests/cpp/test_oss_transport.c -@@ -0,0 +1,40 @@ -+/* SPDX-License-Identifier: Apache-2.0 */ -+/* SPDX-FileCopyrightText: Copyright The Lance Authors */ -+ -+#include "lance/lance.h" -+#include -+#include -+ -+/* The Python harness serves a missing manifest on a local HTTP endpoint. */ -+int main(int argc, char **argv) { -+ if (argc != 3) return 2; -+ const char *options[] = { -+ "oss_endpoint", argv[1], -+ "oss_region", "cn-test", -+ "oss_access_key_id", "test-key", -+ "oss_secret_access_key", "test-secret", -+ "addressing_style", "path", -+ NULL -+ }; -+ LanceSession *session = NULL; -+ LanceDataset *dataset = NULL; -+ const char *uri = "oss://test-bucket/missing.lance"; -+ if (strcmp(argv[2], "shared") == 0) { -+ session = lance_session_new(0, 0); -+ if (session == NULL) return 3; -+ dataset = lance_dataset_open_with_session(uri, options, 1, session); -+ } else { -+ dataset = lance_dataset_open(uri, options, 1); -+ } -+ /* The object does not exist, but the request must reach the HTTP server. */ -+ const char *error = lance_last_error_message(); -+ int failed = dataset != NULL || error == NULL; -+ if (error != NULL) { -+ fprintf(stderr, "%s\n", error); -+ failed |= strstr(error, "default HTTP transport is not installed") != NULL; -+ lance_free_string(error); -+ } -+ lance_dataset_close(dataset); -+ lance_session_close(session); -+ return failed ? 1 : 0; -+} -diff --git a/tests/static_oss_transport_test.py b/tests/static_oss_transport_test.py -new file mode 100644 -index 0000000..7d0e87c ---- /dev/null -+++ b/tests/static_oss_transport_test.py -@@ -0,0 +1,91 @@ -+# SPDX-License-Identifier: Apache-2.0 -+# SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+"""Exercise native OSS HTTP initialization from fresh, statically linked C processes. -+ -+Build lance-c first, then run on Linux: -+ python3 tests/static_oss_transport_test.py target/release/liblance_c.a -+ -+No OSS account is needed. A local HTTP server returns 404 for a missing manifest. -+The assertion is that an HTTP request reaches it, not merely that opening fails. -+Unlike a Rust test binary, the C executable must pull initialization from the archive. -+""" -+ -+import argparse -+from http.server import BaseHTTPRequestHandler, HTTPServer -+import os -+from pathlib import Path -+import shlex -+import subprocess -+import sys -+import tempfile -+import threading -+ -+ -+def main(): -+ parser = argparse.ArgumentParser(description=__doc__) -+ parser.add_argument("library", type=Path) -+ args = parser.parse_args() -+ if not sys.platform.startswith("linux"): -+ parser.error("this static-link regression test currently supports Linux") -+ library = args.library.resolve(strict=True) -+ root = Path(__file__).resolve().parents[1] -+ requests = [] -+ -+ class Handler(BaseHTTPRequestHandler): -+ def missing(self): -+ requests.append((self.command, self.path)) -+ self.send_response(404) -+ self.send_header("Content-Length", "0") -+ self.send_header("Connection", "close") -+ self.end_headers() -+ -+ do_HEAD = missing -+ do_GET = missing -+ -+ def log_message(self, *_args): -+ pass -+ -+ with tempfile.TemporaryDirectory(prefix="lance-static-oss-") as directory: -+ executable = Path(directory) / "test_oss_transport" -+ # Pass the archive explicitly; -llance_c could silently select the shared library. -+ # Do not use --whole-archive: ordinary native linking must retain initialization. -+ subprocess.run( -+ shlex.split(os.environ.get("CC", "cc")) -+ + ["-std=c11", "-Wall", "-Wextra", "-Werror", "-Wl,--gc-sections", -+ "-I", str(root / "include"), str(root / "tests/cpp/test_oss_transport.c"), -+ str(library), "-lgcc_s", "-lutil", "-lrt", "-lpthread", "-lm", "-ldl", -+ "-o", str(executable)], -+ check=True, -+ ) -+ environment = { -+ key: value for key, value in os.environ.items() -+ if not key.startswith(("AWS_", "OSS_", "ALIBABA_CLOUD_")) -+ and key.lower() not in ("http_proxy", "https_proxy", "all_proxy", "no_proxy") -+ } -+ environment["NO_PROXY"] = "127.0.0.1,localhost" -+ # Each mode starts a new process so an earlier call cannot hide missing initialization. -+ for mode in ("ordinary", "shared"): -+ requests.clear() -+ with HTTPServer(("127.0.0.1", 0), Handler) as server: -+ thread = threading.Thread( -+ target=server.serve_forever, kwargs={"poll_interval": 0.05}, daemon=True -+ ) -+ thread.start() -+ try: -+ result = subprocess.run( -+ [str(executable), f"http://127.0.0.1:{server.server_port}", mode], -+ env=environment, capture_output=True, text=True, timeout=30, -+ ) -+ finally: -+ server.shutdown() -+ thread.join() -+ assert result.returncode == 0, f"{mode}: {result.stderr}" -+ assert any("/_versions/" in path for _, path in requests), ( -+ f"{mode}: no manifest HTTP request reached the server: {result.stderr}" -+ ) -+ print(f"PASS: {mode} OSS open reached the local HTTP server") -+ -+ -+if __name__ == "__main__": -+ main() diff --git a/thirdparty/patches/lance-c-0.1.9-pr-83.patch b/thirdparty/patches/lance-c-0.1.9-pr-83.patch deleted file mode 100644 index 86c0c23e227d31..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-83.patch +++ /dev/null @@ -1,1914 +0,0 @@ -From 0a30ee6c5a9d1455ceb36f4745acb79e53a00461 Mon Sep 17 00:00:00 2001 -Subject: [PATCH] feat: support multi-vector queries through C and C++ APIs (#83) - -Upstream: https://github.com/lance-format/lance-c/pull/83 -Commit: 0a30ee6c5a9d1455ceb36f4745acb79e53a00461 - -Adapted to the v0.1.9 community patch chain used by Doris. Resolve context -conflicts with PR #73 and retain PR #79 scalar_segment fields, execution -branch, and tests alongside the upstream multi-vector additions. The -multi-vector implementation and its integration tests are unchanged. - -diff --git a/README.md b/README.md -index d9a5f2b..29c89c4 100644 ---- a/README.md -+++ b/README.md -@@ -70,6 +70,40 @@ Based on the [liblance RFC](https://github.com/lance-format/lance/discussions/60 - | [x] | Filter pushdown | `lance_scanner_set_substrait_filter()` accepts a serialized Substrait `ExtendedExpression`; `lance_scanner_additional_sql_filter()` adds SQL predicates with AND before scanning starts | - | [x] | Data-file cache | Optional Foyer memory/disk cache for immutable `data/*.lance` reads | - -+## Multi-vector search -+ -+Use `lance_scanner_nearest_multivector` or the C++ `Scanner::nearest_multivector` -+method for a `List>` column: -+ -+```cpp -+const float query[] = {1.0f, 0.0f, 0.0f, 1.0f}; -+auto scanner = dataset.scan(); -+scanner.nearest_multivector("embeddings", query, 2, 2, LANCE_DTYPE_FLOAT32, 10) -+ .metric(LANCE_METRIC_COSINE) -+ .prefilter(true); -+``` -+ -+The copied, row-major matrix is **one query** containing two subvectors. Results -+rank logical rows by the sum of each query subvector's minimum distance to a -+stored subvector. Empty or null outer rows do not rank. Inner vectors must be -+non-nullable; actual stored null or non-finite elements encountered during -+scoring fail the stream. Float types and dimensions must match the column. -+Cosine pairs with zero norm have undefined distance and are ignored. A row is -+excluded if any query subvector has no defined match; a zero-norm query subvector -+therefore produces no results. Column names use Lance field-path syntax, -+including nested paths such as `payload.embeddings` and backtick-quoted names. -+ -+L2 is the default on every fragment. Cosine multi-vector indexes are supported -+by the pinned Lance version; incompatible metrics use exact search. Indexed -+candidates are refined against stored values (`refine_factor` defaults to 1). -+ANN candidate selection remains approximate. Limit and offset apply after -+restoring distance order, including fragment-scoped searches. Strict row batching -+is applied after that final result window, preserving full batches except the last. -+ -+Queries accept at most 128 subvectors. Both `num_vectors * k` and -+`refine_factor * k` must be at most 100,000 to bound plan expansion and candidate -+allocation. The existing single-vector API and its defaults are unchanged. -+ - ## Building - - There are four supported entry points; pick whichever matches your toolchain. -diff --git a/include/lance/lance.h b/include/lance/lance.h -index a201bfb..e404612 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1847,6 +1847,21 @@ int32_t lance_scanner_nearest( - uint32_t k - ); - -+/** -+ * Set one multi-vector query on a List> column. -+ * Inner vectors must be non-nullable and contain no null elements; the outer list may be nullable. -+ * query_data contains dimension * num_vectors aligned elements in row-major order. -+ * Both sizes and k must be positive. At most 128 query subvectors are accepted; -+ * num_vectors * k and refine_factor * k must each be at most 100000. -+ * Values are copied before returning. The default metric is L2 on every fragment. -+ * Scores sum each query vector's minimum distance; refinement defaults to 1. -+ * Returns 0 on success, -1 on error. Stored invalid elements fail during execution. -+ */ -+int32_t lance_scanner_nearest_multivector( -+ LanceScanner* scanner, const char* column, const void* query_data, -+ size_t dimension, size_t num_vectors, LanceDataType element_type, uint32_t k -+); -+ - /** - * Set both the minimum and maximum vector-index partition-search bounds. - * -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index a231a00..146c6b1 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1465,6 +1465,16 @@ public: - return *this; - } - -+ /// One multi-vector query, copied from dimension * num_vectors row-major elements. -+ Scanner& nearest_multivector(const std::string& column, const void* query_data, -+ size_t dimension, size_t num_vectors, -+ LanceDataType element_type, uint32_t k) { -+ if (lance_scanner_nearest_multivector(handle_.get(), column.c_str(), query_data, -+ dimension, num_vectors, element_type, k) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Replace both minimum and maximum partition-search bounds. - Scanner& nprobes(uint32_t nprobes) { - if (lance_scanner_set_nprobes(handle_.get(), nprobes) != 0) check_error(); -diff --git a/src/lib.rs b/src/lib.rs -index 197da5f..55f6ecf 100644 ---- a/src/lib.rs -+++ b/src/lib.rs -@@ -39,6 +39,7 @@ mod index; - mod index_model; - mod index_segment; - mod merge_insert; -+mod multivector; - mod restore; - pub mod runtime; - mod scalar_segment; -diff --git a/src/multivector.rs b/src/multivector.rs -new file mode 100644 -index 0000000..c08df28 ---- /dev/null -+++ b/src/multivector.rs -@@ -0,0 +1,514 @@ -+// SPDX-License-Identifier: Apache-2.0 -+// SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+//! Correct multi-vector scoring before the pinned Lance plan's candidate limits. -+ -+use std::collections::HashMap; -+use std::sync::Arc; -+ -+use arrow_array::types::{Float16Type, Float32Type, Float64Type}; -+use arrow_array::{ -+ Array, ArrayRef, ArrowPrimitiveType, BooleanArray, FixedSizeListArray, Float32Array, ListArray, -+ RecordBatch, UInt64Array, -+}; -+use arrow_schema::{DataType, SchemaRef}; -+use datafusion::error::{DataFusionError, Result}; -+use datafusion::execution::context::TaskContext; -+use datafusion::physical_plan::{ -+ DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, SendableRecordBatchStream, -+ stream::RecordBatchStreamAdapter, -+}; -+use futures::{StreamExt, TryStreamExt, stream}; -+use lance::io::exec::KNNVectorDistanceExec; -+use lance_linalg::distance::{Cosine, DistanceType, Dot, L2}; -+ -+// Lance creates one ANN branch per query vector, -+// each overfetching 10 * k candidates before scoring; wire bytes alone cannot bound this work. -+pub(crate) const MAX_QUERY_VECTORS: usize = 128; -+pub(crate) const MAX_QUERY_VECTOR_CANDIDATES: usize = 100_000; -+ -+fn invalid(message: impl Into) -> DataFusionError { -+ DataFusionError::Execution(message.into()) -+} -+ -+/// Rewrite inside TopK/refinement, before any score can discard a candidate. -+pub(crate) fn rewrite(plan: Arc) -> Result> { -+ let children = plan -+ .children() -+ .into_iter() -+ .map(|child| rewrite(child.clone())) -+ .collect::>>()?; -+ let plan = if children.is_empty() { -+ plan -+ } else { -+ plan.with_new_children(children)? -+ }; -+ let mode = if let Some(exact) = plan.downcast_ref::() { -+ if exact.is_batch { -+ return Err(invalid( -+ "expected one logical multi-vector query, not batch queries", -+ )); -+ } -+ Some(Scoring::Exact { -+ query: exact.query.clone(), -+ column: exact.column.clone(), -+ metric: exact.distance_type, -+ }) -+ // This pinned Lance node is not publicly re-exported, so match its stable plan name. -+ } else if plan.name() == "MultivectorScoringExec" { -+ Some(Scoring::Indexed) -+ } else { -+ None -+ }; -+ Ok(match mode { -+ Some(mode) => Arc::new(MultiVectorScoreExec { -+ original: plan, -+ mode, -+ }), -+ None => plan, -+ }) -+} -+ -+/// Apply the final distance-ordered window without invalidating output batching. -+pub(crate) fn apply_result_window( -+ plan: Arc, -+ offset: usize, -+ limit: Option, -+) -> Result> { -+ use datafusion::physical_expr::{PhysicalSortExpr, expressions}; -+ use datafusion::physical_plan::{ -+ coalesce_partitions::CoalescePartitionsExec, limit::GlobalLimitExec, sorts::sort::SortExec, -+ }; -+ if plan -+ .downcast_ref::() -+ .is_some() -+ { -+ // Offset can split a previously strict batch. Keep Lance's final rechunker -+ // outside the window, preserving its resolved batch size, including defaults. -+ let input = apply_result_window(plan.children()[0].clone(), offset, limit)?; -+ return plan.with_new_children(vec![input]); -+ } -+ let sort = PhysicalSortExpr { -+ expr: expressions::col("_distance", plan.schema().as_ref())?, -+ options: arrow::compute::SortOptions { -+ descending: false, -+ nulls_first: false, -+ }, -+ }; -+ // Fragment-scoped payload takes can reorder batches. Restore distance order -+ // before the window; the nearest plan already bounds candidate rows by k. -+ let sorted = Arc::new(SortExec::new( -+ [sort].into(), -+ Arc::new(CoalescePartitionsExec::new(plan)), -+ )); -+ Ok(Arc::new(GlobalLimitExec::new(sorted, offset, limit))) -+} -+ -+#[derive(Clone, Debug)] -+enum Scoring { -+ Exact { -+ query: ArrayRef, -+ column: String, -+ metric: DistanceType, -+ }, -+ Indexed, -+} -+ -+#[derive(Debug)] -+struct MultiVectorScoreExec { -+ original: Arc, -+ mode: Scoring, -+} -+ -+impl DisplayAs for MultiVectorScoreExec { -+ fn fmt_as(&self, _: DisplayFormatType, f: &mut std::fmt::Formatter) -> std::fmt::Result { -+ write!(f, "MultiVectorScore: {}", self.original.name()) -+ } -+} -+ -+impl ExecutionPlan for MultiVectorScoreExec { -+ fn name(&self) -> &str { -+ "MultiVectorScoreExec" -+ } -+ fn properties(&self) -> &Arc { -+ self.original.properties() -+ } -+ fn children(&self) -> Vec<&Arc> { -+ self.original.children() -+ } -+ fn required_input_distribution(&self) -> Vec { -+ self.original.required_input_distribution() -+ } -+ fn with_new_children( -+ self: Arc, -+ children: Vec>, -+ ) -> Result> { -+ Ok(Arc::new(Self { -+ original: self.original.clone().with_new_children(children)?, -+ mode: self.mode.clone(), -+ })) -+ } -+ fn execute( -+ &self, -+ partition: usize, -+ context: Arc, -+ ) -> Result { -+ let schema = self.schema(); -+ match &self.mode { -+ Scoring::Exact { -+ query, -+ column, -+ metric, -+ } => { -+ let input = self.children()[0].execute(partition, context)?; -+ let query = query.clone(); -+ let column = column.clone(); -+ let metric = *metric; -+ let output_schema = schema.clone(); -+ let output = input -+ .map(move |batch| { -+ let query = query.clone(); -+ let column = column.clone(); -+ let schema = output_schema.clone(); -+ async move { -+ let batch = batch?; -+ tokio::task::spawn_blocking(move || { -+ exact_batch(batch, query, &column, metric, schema) -+ }) -+ .await -+ .map_err(|e| DataFusionError::External(Box::new(e)))? -+ } -+ }) -+ .buffered(lance_core::utils::tokio::get_num_compute_intensive_cpus()); -+ Ok(Box::pin(RecordBatchStreamAdapter::new(schema, output))) -+ } -+ Scoring::Indexed => { -+ let inputs = self -+ .children() -+ .into_iter() -+ .map(|child| child.execute(partition, context.clone())) -+ .collect::>>()?; -+ let output_schema = schema.clone(); -+ let output = -+ stream::once(async move { indexed_batch(inputs, output_schema).await }); -+ Ok(Box::pin(RecordBatchStreamAdapter::new(schema, output))) -+ } -+ } -+ } -+} -+ -+fn row_distance( -+ query: &dyn Array, -+ vectors: &FixedSizeListArray, -+ metric: DistanceType, -+) -> Result> -+where -+ T::Native: L2 + Cosine + Dot + Into, -+{ -+ let q = query -+ .as_any() -+ .downcast_ref::>() -+ .ok_or_else(|| invalid("multi-vector query element type mismatch"))?; -+ let values = vectors -+ .values() -+ .as_any() -+ .downcast_ref::>() -+ .ok_or_else(|| invalid("multi-vector stored element type mismatch"))?; -+ if vectors.null_count() != 0 -+ || values.null_count() != 0 -+ || values -+ .values() -+ .iter() -+ .any(|v| !Into::::into(*v).is_finite()) -+ { -+ return Err(invalid( -+ "multi-vector stored subvectors must contain only finite, non-null elements", -+ )); -+ } -+ let dimension = vectors.value_length() as usize; -+ let distance = metric.func(); -+ // Subtracting each small distance from 1 rounds it away before TopK. Sum minima -+ // directly, using f64 only for the accumulator; the base kernels and output remain f32. -+ let mut score = 0.0f64; -+ for query_vector in q.values().chunks_exact(dimension) { -+ let best = values -+ .values() -+ .chunks_exact(dimension) -+ .map(|vector| distance(query_vector, vector)) -+ // Finite zero-norm vectors have undefined cosine distance. Ignore those -+ // pairs; a query with no defined match masks this row, not the whole scan. -+ .filter(|distance| !distance.is_nan()) -+ .min_by(f32::total_cmp); -+ let Some(best) = best else { -+ return Ok(None); -+ }; -+ score += best as f64; -+ } -+ let score = score as f32; -+ if !score.is_finite() { -+ return Err(invalid("multi-vector distance is not finite")); -+ } -+ Ok(Some(score)) -+} -+ -+fn vector_column(batch: &RecordBatch, column: &str) -> Result { -+ if let Some(array) = batch.column_by_name(column) { -+ return Ok(array.clone()); -+ } -+ // The planner resolves field paths, including quoted dotted names. Its private -+ // KNN resolver is not exported, so use the same parser and struct traversal here. -+ let parts = lance_core::datatypes::parse_field_path(column) -+ .map_err(|e| invalid(format!("invalid vector column path '{column}': {e}")))?; -+ let root = parts -+ .first() -+ .ok_or_else(|| invalid("empty vector column path"))?; -+ let mut array = batch -+ .column_by_name(root) -+ .cloned() -+ .ok_or_else(|| invalid(format!("missing vector column '{column}'")))?; -+ for part in &parts[1..] { -+ array = array -+ .as_any() -+ .downcast_ref::() -+ .and_then(|parent| parent.column_by_name(part)) -+ .cloned() -+ .ok_or_else(|| { -+ invalid(format!( -+ "missing struct field '{part}' in vector column '{column}'" -+ )) -+ })?; -+ } -+ Ok(array) -+} -+ -+fn exact_batch( -+ batch: RecordBatch, -+ query: ArrayRef, -+ column: &str, -+ metric: DistanceType, -+ schema: SchemaRef, -+) -> Result { -+ if batch.num_rows() == 0 { -+ return Ok(RecordBatch::new_empty(schema)); -+ } -+ let array = vector_column(&batch, column)?; -+ let vectors = array -+ .as_any() -+ .downcast_ref::() -+ .ok_or_else(|| invalid("multi-vector scoring requires a List column"))?; -+ let row_ids = batch.column_by_name("_rowid"); -+ let mut scores = Vec::with_capacity(batch.num_rows()); -+ for (i, vector) in vectors.iter().enumerate() { -+ if row_ids.is_some_and(|ids| ids.is_null(i)) { -+ scores.push(None); -+ continue; -+ } -+ let Some(vector) = vector else { -+ scores.push(None); -+ continue; -+ }; -+ let vector = vector -+ .as_any() -+ .downcast_ref::() -+ .ok_or_else(|| invalid("multi-vector row must be a FixedSizeList"))?; -+ if vector.is_empty() { -+ scores.push(None); -+ continue; -+ } -+ let score = match query.data_type() { -+ DataType::Float16 => row_distance::(query.as_ref(), vector, metric), -+ DataType::Float32 => row_distance::(query.as_ref(), vector, metric), -+ DataType::Float64 => row_distance::(query.as_ref(), vector, metric), -+ _ => Err(invalid("unsupported multi-vector element type")), -+ }?; -+ scores.push(score); -+ } -+ let mask = BooleanArray::from_iter(scores.iter().map(|score| Some(score.is_some()))); -+ let distances: ArrayRef = Arc::new(Float32Array::from(scores)); -+ let columns = schema -+ .fields() -+ .iter() -+ .map(|field| { -+ if field.name() == "_distance" { -+ Ok(distances.clone()) -+ } else { -+ batch -+ .column_by_name(field.name()) -+ .cloned() -+ .ok_or_else(|| invalid(format!("missing score output column {}", field.name()))) -+ } -+ }) -+ .collect::>>()?; -+ Ok(arrow::compute::filter_record_batch( -+ &RecordBatch::try_new(schema, columns)?, -+ &mask, -+ )?) -+} -+ -+async fn indexed_batch( -+ inputs: Vec, -+ schema: SchemaRef, -+) -> Result { -+ // A query child may emit several batches. Reduce its entire stream exactly once; -+ // treating each batch as a query adds spurious missing-query contributions. -+ let queries = futures::future::try_join_all(inputs.into_iter().map(|mut input| async move { -+ let mut rows = HashMap::::new(); -+ let mut maximum: Option = None; -+ while let Some(batch) = input.try_next().await? { -+ let ids = batch -+ .column_by_name("_rowid") -+ .and_then(|a| a.as_any().downcast_ref::()) -+ .ok_or_else(|| invalid("indexed multi-vector scorer requires row IDs"))?; -+ let distances = batch -+ .column_by_name("_distance") -+ .and_then(|a| a.as_any().downcast_ref::()) -+ .ok_or_else(|| invalid("indexed multi-vector scorer requires distances"))?; -+ for i in 0..batch.num_rows() { -+ let distance = distances.value(i); -+ if ids.is_null(i) || distances.is_null(i) || !distance.is_finite() { -+ return Err(invalid( -+ "indexed multi-vector candidate has invalid row ID or distance", -+ )); -+ } -+ maximum = Some(maximum.map_or(distance, |old| old.max(distance))); -+ rows.entry(ids.value(i)) -+ .and_modify(|old| *old = old.min(distance)) -+ .or_insert(distance); -+ } -+ } -+ Ok::<_, DataFusionError>((rows, maximum.unwrap_or(1.0))) -+ })) -+ .await?; -+ let mut results = HashMap::::new(); -+ let mut missed = 0.0f64; -+ for (rows, maximum) in queries { -+ for (id, score) in &mut results { -+ *score += *rows.get(id).unwrap_or(&maximum) as f64; -+ } -+ for (id, score) in rows { -+ results.entry(id).or_insert(score as f64 + missed); -+ } -+ missed += maximum as f64; -+ } -+ let (ids, scores): (Vec<_>, Vec<_>) = results.into_iter().unzip(); -+ Ok(RecordBatch::try_new( -+ schema, -+ vec![ -+ Arc::new(Float32Array::from_iter_values( -+ scores.into_iter().map(|score| score as f32), -+ )), -+ Arc::new(UInt64Array::from(ids)), -+ ], -+ )?) -+} -+ -+pub(crate) fn validate_query(values: &dyn Array) -> Result<()> { -+ fn finite(values: &dyn Array) -> bool -+ where -+ T::Native: Into, -+ { -+ values -+ .as_any() -+ .downcast_ref::>() -+ .is_some_and(|array| { -+ array.null_count() == 0 -+ && array -+ .values() -+ .iter() -+ .all(|v| Into::::into(*v).is_finite()) -+ }) -+ } -+ let valid = match values.data_type() { -+ DataType::Float16 => finite::(values), -+ DataType::Float32 => finite::(values), -+ DataType::Float64 => finite::(values), -+ _ => false, -+ }; -+ if valid { -+ Ok(()) -+ } else { -+ Err(invalid( -+ "multi-vector query must contain only finite, non-null elements", -+ )) -+ } -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use arrow_schema::{Field, Schema}; -+ use lance_datafusion::exec::{ -+ LanceExecutionOptions, OneShotExec, StrictBatchSizeExec, execute_plan, -+ }; -+ -+ #[test] -+ fn result_window_preserves_strict_batching_and_distance_order() { -+ crate::runtime::block_on(async { -+ for size in [2, 3] { -+ for (offset, limit) in [ -+ (1, Some(4)), -+ (1, Some(3)), -+ (5, Some(4)), -+ (6, Some(4)), -+ (1, None), -+ ] { -+ let schema = Arc::new(Schema::new(vec![Field::new( -+ "_distance", -+ DataType::Float32, -+ false, -+ )])); -+ let batches = [[5., 0.], [4., 1.], [3., 2.]] -+ .into_iter() -+ .map(|values| { -+ RecordBatch::try_new( -+ schema.clone(), -+ vec![Arc::new(Float32Array::from(values.to_vec()))], -+ ) -+ .map_err(DataFusionError::from) -+ }) -+ .collect::>(); -+ let input = Arc::new(OneShotExec::new(Box::pin( -+ RecordBatchStreamAdapter::new(schema, stream::iter(batches)), -+ ))); -+ let plan = Arc::new(StrictBatchSizeExec::new(input, size)); -+ let plan = apply_result_window(plan, offset, limit).unwrap(); -+ let batches: Vec<_> = execute_plan( -+ plan, -+ LanceExecutionOptions { -+ batch_size: Some(2), -+ ..Default::default() -+ }, -+ ) -+ .unwrap() -+ .try_collect() -+ .await -+ .unwrap(); -+ let actual: Vec<_> = batches -+ .iter() -+ .flat_map(|b| { -+ b.column(0) -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .to_vec() -+ }) -+ .collect(); -+ let expected: Vec<_> = (0..6) -+ .skip(offset) -+ .take(limit.unwrap_or(6)) -+ .map(|i| i as f32) -+ .collect(); -+ assert_eq!(actual, expected); -+ assert_eq!( -+ batches -+ .iter() -+ .map(RecordBatch::num_rows) -+ .collect::>(), -+ expected.chunks(size).map(<[f32]>::len).collect::>() -+ ); -+ } -+ } -+ }); -+ } -+} -diff --git a/src/scanner.rs b/src/scanner.rs -index bda554d..f4b3f36 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -363,8 +363,18 @@ impl LanceScanner { - if let Some(cols) = &self.columns { - scanner.project(cols)?; - } -+ let multi_vector = self.nearest.as_ref().is_some_and(|query| { -+ matches!( -+ query.query.data_type(), -+ arrow_schema::DataType::FixedSizeList(_, _) -+ ) -+ }); - if self.limit.is_some() || self.offset.is_some() { - scanner.limit(self.limit, self.offset)?; -+ if multi_vector { -+ // Retain Lance's window validation, but defer truncation until the final sort. -+ scanner.limit(None, None)?; -+ } - } - if let Some(bs) = self.batch_size { - scanner.batch_size(bs); -@@ -440,7 +450,27 @@ impl LanceScanner { - if let Some(query_parallelism) = self.query_parallelism { - scanner.query_parallelism(query_parallelism); - } -- if let Some(rf) = self.refine_factor { -+ if multi_vector { -+ if matches!( -+ self.metric_override, -+ Some(crate::index::LanceMetricType::Hamming) -+ ) { -+ return Err(lance_core::Error::invalid_input_source( -+ "multi-vector queries support only l2, cosine, and dot metrics".into(), -+ )); -+ } -+ let refine = self.refine_factor.unwrap_or(1); -+ if refine == 0 -+ || n.k as usize -+ > crate::multivector::MAX_QUERY_VECTOR_CANDIDATES / refine as usize -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ "multi-vector refined candidate count must be in 1..=100000".into(), -+ )); -+ } -+ // Validate actual stored values and refine candidate scores before TopK. -+ scanner.refine(refine); -+ } else if let Some(rf) = self.refine_factor { - scanner.refine(rf); - } - if let Some(ef) = self.ef { -@@ -448,6 +478,9 @@ impl LanceScanner { - } - if let Some(m) = self.metric_override { - scanner.distance_metric(m.to_distance()); -+ } else if multi_vector { -+ // Resolve the same default on indexed and uncovered fragments. -+ scanner.distance_metric(lance_linalg::distance::DistanceType::L2); - } - if let Some(ui) = self.use_index { - scanner.use_index(ui); -@@ -506,6 +539,12 @@ impl LanceScanner { - scanner, - distributed_fts, - scalar_segment, -+ multi_vector_window: multi_vector.then_some(( -+ self.offset.unwrap_or(0) as usize, -+ self.limit.map(|n| n as usize), -+ )), -+ batch_size: self.batch_size, -+ scan_statistics_callback: self.scan_statistics_callback.clone(), - }) - } - } -@@ -521,6 +560,9 @@ struct PreparedScanner { - scanner: lance::dataset::scanner::Scanner, - distributed_fts: Option, - scalar_segment: Option, -+ multi_vector_window: Option<(usize, Option)>, -+ batch_size: Option, -+ scan_statistics_callback: Option, - } - - impl PreparedScanner { -@@ -532,6 +574,19 @@ impl PreparedScanner { - .try_into_stream() - .await; - } -+ if let Some((offset, limit)) = self.multi_vector_window { -+ let plan = crate::multivector::rewrite(self.scanner.create_plan().await?)?; -+ let plan = crate::multivector::apply_result_window(plan, offset, limit)?; -+ let stream = lance_datafusion::exec::execute_plan( -+ plan, -+ lance_datafusion::exec::LanceExecutionOptions { -+ batch_size: self.batch_size, -+ execution_stats_callback: self.scan_statistics_callback, -+ ..Default::default() -+ }, -+ )?; -+ return Ok(DatasetRecordBatchStream::new(stream)); -+ } - let Some(distributed_fts) = self.distributed_fts else { - return self.scanner.try_into_stream().await; - }; -@@ -2744,6 +2799,21 @@ unsafe fn scanner_nearest_inner( - } - let column_str = unsafe { helpers::parse_c_string(column)? }.unwrap(); - -+ let query = unsafe { decode_query_values(query_data, query_len, element_type)? }; -+ -+ s.nearest = Some(NearestQuery { -+ column: column_str.to_string(), -+ query, -+ k, -+ }); -+ Ok(0) -+} -+ -+unsafe fn decode_query_values( -+ query_data: *const c_void, -+ query_len: usize, -+ element_type: i32, -+) -> Result { - let dtype = match element_type { - 0 => LanceDataType::Float32, - 1 => LanceDataType::Float16, -@@ -2782,9 +2852,112 @@ unsafe fn scanner_nearest_inner( - } - }; - -+ Ok(query) -+} -+ -+/// Set one multi-vector query, supplied as a row-major matrix of floating-point values. -+/// The caller must supply dimension * num_vectors aligned elements matching the column type. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_nearest_multivector( -+ scanner: *mut LanceScanner, -+ column: *const c_char, -+ query_data: *const c_void, -+ dimension: usize, -+ num_vectors: usize, -+ element_type: i32, -+ k: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ nearest_multivector_inner( -+ scanner, -+ column, -+ query_data, -+ dimension, -+ num_vectors, -+ element_type, -+ k, -+ ) -+ },) -+} -+ -+unsafe fn nearest_multivector_inner( -+ scanner: *mut LanceScanner, -+ column: *const c_char, -+ query_data: *const c_void, -+ dimension: usize, -+ num_vectors: usize, -+ element_type: i32, -+ k: u32, -+) -> Result { -+ use arrow_schema::{DataType, Field}; -+ let invalid = |message: &str| lance_core::Error::invalid_input_source(message.into()); -+ if scanner.is_null() || column.is_null() || query_data.is_null() { -+ return Err(invalid("scanner, column, and query_data must not be NULL")); -+ } -+ if dimension == 0 || dimension > i32::MAX as usize || num_vectors == 0 || k == 0 { -+ return Err(invalid( -+ "dimension, num_vectors, and k must be positive; dimension must fit int32", -+ )); -+ } -+ if num_vectors > crate::multivector::MAX_QUERY_VECTORS -+ || num_vectors > crate::multivector::MAX_QUERY_VECTOR_CANDIDATES / k as usize -+ { -+ return Err(invalid( -+ "multi-vector query exceeds 128 subvectors or 100000 subvector-candidates", -+ )); -+ } -+ let (data_type, width) = match element_type { -+ 0 => (DataType::Float32, 4), -+ 1 => (DataType::Float16, 2), -+ 2 => (DataType::Float64, 8), -+ _ => { -+ return Err(invalid( -+ "multi-vector queries require float16, float32, or float64", -+ )); -+ } -+ }; -+ let count = dimension -+ .checked_mul(num_vectors) -+ .filter(|count| *count <= isize::MAX as usize / width) -+ .ok_or_else(|| invalid("query matrix byte size overflows"))?; -+ let s = unsafe { &mut *scanner }; -+ if s.fts_query.is_some() || s.fts_context.is_some() { -+ return Err(invalid( -+ "nearest and full-text search are mutually exclusive", -+ )); -+ } -+ let column = unsafe { helpers::parse_c_string(column)? }.unwrap(); -+ let field = s -+ .dataset -+ .schema() -+ .field(column) -+ .ok_or_else(|| invalid("multi-vector column does not exist"))?; -+ match field.data_type() { -+ DataType::List(child) if !child.is_nullable() => match child.data_type() { -+ DataType::FixedSizeList(element, dim) -+ if *dim == dimension as i32 && *element.data_type() == data_type => {} -+ _ => return Err(invalid("multi-vector dimension/type mismatch")), -+ }, -+ _ => { -+ return Err(invalid( -+ "multi-vector column must be List of non-nullable FixedSizeList", -+ )); -+ } -+ } -+ // A primitive array is interpreted as one vector by Lance. Preserve matrix shape even -+ // for a single subvector. Lance does not preserve element nullability in its schema. -+ let values = unsafe { decode_query_values(query_data, count, element_type)? }; -+ crate::multivector::validate_query(values.as_ref())?; -+ let query = arrow_array::FixedSizeListArray::try_new( -+ Arc::new(Field::new("item", data_type, false)), -+ dimension as i32, -+ values, -+ None, -+ )?; - s.nearest = Some(NearestQuery { -- column: column_str.to_string(), -- query, -+ column: column.to_string(), -+ query: Arc::new(query), - k, - }); - Ok(0) -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 04699da..24bef9f 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -13388,3 +13388,28 @@ fn test_scalar_segment_requires_explicit_domain_and_checks_uuid() { - lance_dataset_close(ds); - } - } -+ -+#[test] -+fn test_multivector_nearest_rejects_null_handle() { -+ let column = c_str("vectors"); -+ let query = [1.0f32, 0.0]; -+ let status = unsafe { -+ lance_scanner_nearest_multivector( -+ ptr::null_mut(), -+ column.as_ptr(), -+ query.as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1, -+ ) -+ }; -+ assert_eq!(status, -1); -+ let error = lance_last_error_message(); -+ assert!(!error.is_null()); -+ let message = unsafe { std::ffi::CStr::from_ptr(error) } -+ .to_string_lossy() -+ .into_owned(); -+ unsafe { lance_free_string(error) }; -+ assert!(message.contains("NULL")); -+} -diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index fc6fc82..11badc6 100644 ---- a/tests/cpp/test_cpp_api.cpp -+++ b/tests/cpp/test_cpp_api.cpp -@@ -427,6 +427,20 @@ static void test_nearest_smoke(const std::string& uri) { - PASS(); - } - -+static void test_multivector_rejects_flat_column(const std::string& uri) { -+ TEST(test_multivector_rejects_flat_column); -+ auto scanner = lance::Dataset::open(uri).scan(); -+ const float query[8] = {1.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; -+ bool caught = false; -+ try { -+ scanner.nearest_multivector("embedding", query, 8, 1, LANCE_DTYPE_FLOAT32, 1); -+ } catch (const lance::Error&) { -+ caught = true; -+ } -+ assert(caught); -+ PASS(); -+} -+ - static void test_index_segments_smoke(const std::string& /*uri*/) { - TEST(test_index_segments_smoke); - -@@ -971,6 +985,7 @@ int main(int argc, char** argv) { - test_error_exception(uri); - test_index_lifecycle(uri); - test_nearest_smoke(uri); -+ test_multivector_rejects_flat_column(uri); - test_index_segments_smoke(uri); - test_index_segment_builder(uri); - test_vector_models_and_reusable_segments(uri); -diff --git a/tests/multivector_test.rs b/tests/multivector_test.rs -new file mode 100644 -index 0000000..f02e850 ---- /dev/null -+++ b/tests/multivector_test.rs -@@ -0,0 +1,965 @@ -+// SPDX-License-Identifier: Apache-2.0 -+// SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+use std::ffi::{CString, c_void}; -+use std::ptr; -+use std::sync::Arc; -+ -+use arrow::buffer::{NullBuffer, OffsetBuffer}; -+use arrow::ffi_stream::{ArrowArrayStreamReader, FFI_ArrowArrayStream}; -+use arrow::record_batch::RecordBatchIterator; -+use arrow_array::{Array, FixedSizeListArray, Float32Array, Int32Array, ListArray, RecordBatch}; -+use arrow_schema::{DataType, Field, Schema}; -+use lance::Dataset; -+use lance_c::*; -+ -+fn fixture() -> (tempfile::TempDir, CString) { -+ let dir = tempfile::tempdir().unwrap(); -+ let path = dir.path().join("vectors.lance"); -+ let element = Arc::new(Field::new("item", DataType::Float32, false)); -+ let vectors = FixedSizeListArray::try_new( -+ element, -+ 2, -+ Arc::new(Float32Array::from(vec![ -+ 1., 0., 0., 1., 2., 0., 3., 0., 0., 3., -+ ])), -+ None, -+ ) -+ .unwrap(); -+ let child = Arc::new(Field::new("item", vectors.data_type().clone(), false)); -+ let rows = ListArray::try_new( -+ child, -+ OffsetBuffer::new(vec![0i32, 2, 3, 5, 5, 5].into()), -+ Arc::new(vectors), -+ Some(NullBuffer::from(vec![true, true, true, true, false])), -+ ) -+ .unwrap(); -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("vectors", rows.data_type().clone(), true), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from(vec![1, 2, 3, 4, 5])), -+ Arc::new(rows), -+ ], -+ ) -+ .unwrap(); -+ lance_c::runtime::block_on(async { -+ Dataset::write( -+ RecordBatchIterator::new(vec![Ok(batch)], schema), -+ path.to_str().unwrap(), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ (dir, CString::new(path.to_str().unwrap()).unwrap()) -+} -+ -+unsafe fn collect(scanner: *mut LanceScanner) -> Vec<(i32, f32)> { -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ let status = unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }; -+ if status != 0 { -+ panic!( -+ "{}", -+ unsafe { std::ffi::CStr::from_ptr(lance_last_error_message()) }.to_string_lossy() -+ ); -+ } -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ reader -+ .flat_map(|batch| { -+ let batch = batch.unwrap(); -+ let ids = batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap(); -+ let distances = batch -+ .column_by_name("_distance") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap(); -+ (0..batch.num_rows()) -+ .map(|i| (ids.value(i), distances.value(i))) -+ .collect::>() -+ }) -+ .collect() -+} -+ -+#[test] -+fn multivector_search_scores_logical_rows_and_excludes_empty_and_null_rows() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ let q = [1.0f32, 0., 0., 1.]; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ for count in [1, 2] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ 2, -+ count, -+ 0, -+ 10 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ assert_eq!(lance_scanner_set_prefilter(scan, true), 0); -+ let rows = collect(scan); -+ let expected = if count == 2 { -+ vec![(1, 0.), (2, 6.), (3, 8.)] -+ } else { -+ vec![(1, 0.), (2, 1.), (3, 4.)] -+ }; -+ assert_eq!(rows, expected); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn multivector_rejects_invalid_shape_and_preserves_previous_query() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ let q = [1.0f32, 0., 0., 1.]; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ 2, -+ 2, -+ 0, -+ 10 -+ ), -+ 0 -+ ); -+ for (dim, count, dtype, k) in [ -+ (0, 2, 0, 10), -+ (2, 0, 0, 10), -+ (3, 1, 0, 10), -+ (2, 2, 3, 10), -+ (2, 2, 4, 10), -+ (2, 2, 1, 10), -+ (2, 2, 0, 0), -+ (usize::MAX, 2, 0, 10), -+ (2, usize::MAX, 0, 10), -+ ] { -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ dim, -+ count, -+ dtype, -+ k -+ ), -+ -1 -+ ); -+ } -+ assert_eq!( -+ lance_scanner_nearest_multivector(scan, column.as_ptr(), ptr::null(), 2, 2, 0, 10), -+ -1 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ for (limit, offset) in [(-1, 0), (10, -1)] { -+ assert_eq!(lance_scanner_set_limit(scan, limit), 0); -+ assert_eq!(lance_scanner_set_offset(scan, offset), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), -1); -+ } -+ assert_eq!(lance_scanner_set_limit(scan, 10), 0); -+ assert_eq!(lance_scanner_set_offset(scan, 0), 0); -+ assert_eq!(collect(scan), vec![(1, 0.), (2, 6.), (3, 8.)]); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn fragment_scoped_indexed_search_orders_candidates_before_offset() { -+ use lance::index::DatasetIndexExt; -+ use lance::index::vector::VectorIndexParams; -+ use lance_index::IndexType; -+ use lance_linalg::distance::MetricType; -+ -+ let (_dir, uri) = fixture(); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let batch = ds.scan().try_into_batch().await.unwrap(); -+ ds.create_index( -+ &["vectors"], -+ IndexType::Vector, -+ None, -+ &VectorIndexParams::ivf_flat(1, MetricType::Cosine), -+ false, -+ ) -+ .await -+ .unwrap(); -+ ds.append( -+ RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ let column = CString::new("vectors").unwrap(); -+ let filter = CString::new("id >= 2").unwrap(); -+ let q = [1.0f32, 0., 0., 1.]; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for _ in 0..10 { -+ let scan = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scan, [0u64, 1].as_ptr(), 2), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_prefilter(scan, true), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, 1), 0); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ 2, -+ 2, -+ 0, -+ 4 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 1), 0); -+ assert_eq!(lance_scanner_set_offset(scan, 1), 0); -+ assert_eq!(lance_scanner_set_limit(scan, 2), 0); -+ assert_eq!(collect(scan), vec![(3, 0.), (2, 1.)]); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+fn custom_fixture( -+ rows: Vec>>, -+ dim: i32, -+ indexed: bool, -+) -> (tempfile::TempDir, CString) { -+ custom_fixture_at(rows, dim, indexed, "vectors") -+} -+ -+fn custom_fixture_at( -+ rows: Vec>>, -+ dim: i32, -+ indexed: bool, -+ column: &str, -+) -> (tempfile::TempDir, CString) { -+ use lance::index::DatasetIndexExt; -+ use lance::index::vector::VectorIndexParams; -+ use lance_index::IndexType; -+ use lance_linalg::distance::MetricType; -+ let dir = tempfile::tempdir().unwrap(); -+ let path = dir.path().join("vectors.lance"); -+ let mut offsets = vec![0i32]; -+ let mut values = Vec::new(); -+ for row in &rows { -+ values.extend_from_slice(row); -+ offsets.push(values.len() as i32 / dim); -+ } -+ let vectors = FixedSizeListArray::try_new( -+ Arc::new(Field::new("item", DataType::Float32, true)), -+ dim, -+ Arc::new(Float32Array::from(values)), -+ None, -+ ) -+ .unwrap(); -+ let rows_array = ListArray::try_new( -+ Arc::new(Field::new("item", vectors.data_type().clone(), false)), -+ OffsetBuffer::new(offsets.into()), -+ Arc::new(vectors), -+ None, -+ ) -+ .unwrap(); -+ let parts = lance_core::datatypes::parse_field_path(column).unwrap(); -+ let mut array: arrow_array::ArrayRef = Arc::new(rows_array); -+ for name in parts[1..].iter().rev() { -+ let field = Arc::new(Field::new(name, array.data_type().clone(), true)); -+ let labels: arrow_array::ArrayRef = Arc::new(Int32Array::from(vec![7; rows.len()])); -+ array = Arc::new(arrow_array::StructArray::from(vec![ -+ (field, array), -+ ( -+ Arc::new(Field::new("label", DataType::Int32, false)), -+ labels, -+ ), -+ ])); -+ } -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new(&parts[0], array.data_type().clone(), true), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from_iter_values(1..=rows.len() as i32)), -+ array, -+ ], -+ ) -+ .unwrap(); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::write( -+ RecordBatchIterator::new(vec![Ok(batch)], schema), -+ path.to_str().unwrap(), -+ None, -+ ) -+ .await -+ .unwrap(); -+ if indexed { -+ ds.create_index( -+ &[column], -+ IndexType::Vector, -+ None, -+ &VectorIndexParams::ivf_flat(1, MetricType::Cosine), -+ false, -+ ) -+ .await -+ .unwrap(); -+ } -+ }); -+ (dir, CString::new(path.to_str().unwrap()).unwrap()) -+} -+ -+#[test] -+fn exact_top_one_preserves_small_distances_before_truncation() { -+ let (_dir, uri) = custom_fixture(vec![vec![Some(0.00015)], vec![Some(0.0001)]], 1, false); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for count in [1, 2] { -+ for batch_size in [1, 1024] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [0.0f32, 0.0].as_ptr().cast(), -+ 1, -+ count, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, batch_size), 0); -+ let rows = collect(scan); -+ assert_eq!(rows.len(), 1); -+ assert_eq!(rows[0].0, 2); -+ assert!((rows[0].1 - count as f32 * 1e-8).abs() < 1e-14, "{rows:?}"); -+ lance_scanner_close(scan); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn indexed_top_one_is_independent_of_child_batch_boundaries() { -+ let rows = vec![ -+ vec![Some(1.), Some(0.)], -+ vec![Some(0.), Some(1.)], -+ vec![Some(1.), Some(1.)], -+ vec![Some(2.), Some(1.)], -+ vec![Some(1.), Some(2.)], -+ ]; -+ let (_dir, uri) = custom_fixture(rows, 2, true); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for batch_size in [1, 2, 1024] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0., 0., 1.].as_ptr().cast(), -+ 2, -+ 2, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 1), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, batch_size), 0); -+ let result = collect(scan); -+ assert_eq!(result[0].0, 3, "batch_size={batch_size}"); -+ assert!((result[0].1 - (2.0 - 2.0f32.sqrt())).abs() < 1e-6); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn actual_null_and_nonfinite_stored_elements_fail_search() { -+ let column = CString::new("vectors").unwrap(); -+ for value in [ -+ None, -+ Some(f32::NAN), -+ Some(f32::INFINITY), -+ Some(f32::NEG_INFINITY), -+ ] { -+ let (_dir, uri) = custom_fixture(vec![vec![value, Some(0.)]], 2, false); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [0.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), 0); -+ let mut reader = ArrowArrayStreamReader::from_raw(&mut stream).unwrap(); -+ let result = reader.next(); -+ assert!( -+ matches!(result, Some(Err(_))), -+ "value={value:?}: {result:?}" -+ ); -+ drop(reader); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+ } -+} -+ -+#[test] -+fn rejects_excessive_multivector_plan_width_and_candidate_work() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ let q = [0.0f32; 258]; -+ for (count, k) in [(129, 1), (128, 782), (1, 100001)] { -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr().cast(), -+ 2, -+ count, -+ 0, -+ k -+ ), -+ -1 -+ ); -+ } -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn omitted_metric_is_l2_on_indexed_and_unindexed_fragments() { -+ let (_dir, uri) = custom_fixture(vec![vec![Some(2.), Some(0.)]], 2, true); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let old = ds.scan().try_into_batch().await.unwrap(); -+ let vectors = FixedSizeListArray::try_new( -+ Arc::new(Field::new("item", DataType::Float32, true)), -+ 2, -+ Arc::new(Float32Array::from(vec![1.0, 0.1])), -+ None, -+ ) -+ .unwrap(); -+ let lists = ListArray::try_new( -+ Arc::new(Field::new("item", vectors.data_type().clone(), false)), -+ OffsetBuffer::new(vec![0i32, 1].into()), -+ Arc::new(vectors), -+ None, -+ ) -+ .unwrap(); -+ let batch = RecordBatch::try_new( -+ old.schema(), -+ vec![Arc::new(Int32Array::from(vec![2])), Arc::new(lists)], -+ ) -+ .unwrap(); -+ ds.append( -+ RecordBatchIterator::new(vec![Ok(batch)], old.schema()), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for (fragment, id, score) in [(0u64, 1, 1.0f32), (1u64, 2, 0.01f32)] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_prefilter(scan, true), 0); -+ assert_eq!(lance_scanner_set_fragment_ids(scan, &fragment, 1), 0); -+ let rows = collect(scan); -+ assert_eq!(rows[0].0, id); -+ assert!( -+ (rows[0].1 - score).abs() < 1e-6, -+ "fragment={fragment}: {rows:?}" -+ ); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn query_values_and_refinement_are_validated_before_execution() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ for value in [f32::NAN, f32::INFINITY, f32::NEG_INFINITY] { -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [value, 0.0f32].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1 -+ ), -+ -1 -+ ); -+ } -+ let query = [1.0f32; 256]; -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ query.as_ptr().cast(), -+ 2, -+ 128, -+ 0, -+ 781 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 3), 0); -+ let mut invalid_metric_stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ lance_scanner_to_arrow_stream(scan, &mut invalid_metric_stream), -+ -1 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 0), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, u32::MAX), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), -1); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn float16_and_float64_queries_support_all_distance_metrics() { -+ let (dir, source) = fixture(); -+ for (code, dtype) in [(1, DataType::Float16), (2, DataType::Float64)] { -+ let path = dir.path().join(format!("typed_{code}.lance")); -+ lance_c::runtime::block_on(async { -+ let source = Dataset::open(source.to_str().unwrap()).await.unwrap(); -+ let batch = source.scan().try_into_batch().await.unwrap(); -+ let vector_type = DataType::List(Arc::new(Field::new( -+ "item", -+ DataType::FixedSizeList(Arc::new(Field::new("item", dtype, true)), 2), -+ false, -+ ))); -+ let vectors = -+ arrow::compute::cast(batch.column_by_name("vectors").unwrap(), &vector_type) -+ .unwrap(); -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("vectors", vector_type, true), -+ ])); -+ let batch = -+ RecordBatch::try_new(schema.clone(), vec![batch.column(0).clone(), vectors]) -+ .unwrap(); -+ Dataset::write( -+ RecordBatchIterator::new(vec![Ok(batch)], schema), -+ path.to_str().unwrap(), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ let uri = CString::new(path.to_str().unwrap()).unwrap(); -+ let column = CString::new("vectors").unwrap(); -+ let f16_query = [1.0f32, 0., 0., 1.].map(half::f16::from_f32); -+ let f64_query = [1.0f64, 0., 0., 1.]; -+ let query = if code == 1 { -+ f16_query.as_ptr().cast() -+ } else { -+ f64_query.as_ptr().cast() -+ }; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for (metric, expected) in [(0, [0., 6., 8.]), (1, [0., 1., 0.]), (2, [0., 0., -4.])] { -+ let scanner = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scanner, -+ column.as_ptr(), -+ query, -+ 2, -+ 2, -+ code, -+ 10 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scanner, metric), 0); -+ assert_eq!(lance_scanner_set_use_index(scanner, false), 0); -+ let mut rows = collect(scanner); -+ rows.sort_by_key(|row| row.0); -+ assert_eq!( -+ rows, -+ vec![(1, expected[0]), (2, expected[1]), (3, expected[2])] -+ ); -+ lance_scanner_close(scanner); -+ } -+ lance_dataset_close(ds); -+ } -+ } -+} -+ -+#[test] -+fn cosine_zero_norm_rows_do_not_abort_exact_or_refined_search() { -+ let rows = vec![ -+ vec![Some(0.), Some(0.)], -+ vec![Some(1.), Some(0.)], -+ vec![Some(0.), Some(0.), Some(0.), Some(1.)], -+ ]; -+ for indexed in [false, true] { -+ let (_dir, uri) = custom_fixture(rows.clone(), 2, indexed); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for query in [[1.0f32, 0., 0., 1.], [0., 0., 0., 1.]] { -+ for count in [1, 2] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ c"vectors".as_ptr(), -+ query.as_ptr().cast(), -+ 2, -+ count, -+ 0, -+ 3 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, indexed), 0); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 2), 0); -+ let mut actual = collect(scan); -+ actual.sort_by_key(|row| row.0); -+ let expected = if query[0] == 0. { -+ vec![] -+ } else if count == 1 { -+ vec![(2, 0.), (3, 1.)] -+ } else { -+ vec![(2, 1.), (3, 1.)] -+ }; -+ assert_eq!( -+ actual, expected, -+ "indexed={indexed}, query={query:?}, count={count}" -+ ); -+ lance_scanner_close(scan); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+ } -+} -+ -+#[test] -+fn nested_and_quoted_columns_support_exact_and_refined_search() { -+ for column in [ -+ "payload.vectors", -+ "payload.`vectors.with.dot`", -+ "payload.inner.`vectors.with.dot`", -+ ] { -+ let rows = vec![ -+ vec![Some(1.), Some(0.), Some(0.), Some(1.)], -+ vec![Some(2.), Some(0.)], -+ vec![Some(1.), Some(1.)], -+ ]; -+ for indexed in [false, true] { -+ let (_dir, uri) = custom_fixture_at(rows.clone(), 2, indexed, column); -+ let column = CString::new(column).unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0., 0., 1.].as_ptr().cast(), -+ 2, -+ 2, -+ 0, -+ 3 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, indexed), 0); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 2), 0); -+ let actual = collect(scan); -+ assert_eq!(actual.len(), 3); -+ assert_eq!( -+ actual.iter().map(|row| row.0).collect::>(), -+ vec![1, 3, 2] -+ ); -+ assert_eq!(actual[0].1, 0.); -+ assert!((actual[1].1 - (2. - 2.0f32.sqrt())).abs() < 1e-6); -+ assert_eq!(actual[2].1, 1.); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+ } -+ } -+} -+ -+#[test] -+fn nested_projections_preserve_schema_and_values_in_exact_indexed_and_hybrid_search() { -+ for column in ["payload.vectors", "payload.`vectors.with.dot`"] { -+ for indexed in [false, true] { -+ for appended in [false, true] { -+ let (_dir, uri) = custom_fixture_at( -+ vec![vec![Some(1.), Some(0.)], vec![Some(0.), Some(1.)]], -+ 2, -+ indexed, -+ column, -+ ); -+ if appended { -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let batch = ds.scan().try_into_batch().await.unwrap(); -+ ds.append( -+ RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ } -+ for projection in [ -+ vec![], -+ vec!["id"], -+ vec!["payload.label"], -+ vec!["id", column], -+ ] { -+ let expected = lance_c::runtime::block_on(async { -+ let ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let mut scanner = ds.scan(); -+ scanner.scan_in_order(true); -+ if !projection.is_empty() { -+ scanner.project(&projection).unwrap(); -+ } -+ scanner.try_into_batch().await.unwrap() -+ }); -+ // The appended fragment repeats the same two values; distance order groups -+ // both exact matches before the orthogonal rows, regardless of tie order. -+ let indices = arrow_array::UInt32Array::from(if appended { -+ vec![0, 2, 1, 3] -+ } else { -+ vec![0, 1] -+ }); -+ let expected = RecordBatch::try_new( -+ expected.schema(), -+ expected -+ .columns() -+ .iter() -+ .map(|array| { -+ arrow::compute::take(array.as_ref(), &indices, None).unwrap() -+ }) -+ .collect(), -+ ) -+ .unwrap(); -+ unsafe { -+ let names: Vec<_> = projection -+ .iter() -+ .map(|name| CString::new(*name).unwrap()) -+ .collect(); -+ let mut columns: Vec<_> = names.iter().map(|name| name.as_ptr()).collect(); -+ columns.push(ptr::null()); -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new( -+ ds, -+ if projection.is_empty() { -+ ptr::null() -+ } else { -+ columns.as_ptr() -+ }, -+ ptr::null(), -+ ); -+ let column = CString::new(column).unwrap(); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 10 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, indexed), 0); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, 1), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), 0); -+ let batches = ArrowArrayStreamReader::from_raw(&mut stream) -+ .unwrap() -+ .collect::, _>>() -+ .unwrap(); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ let actual = -+ arrow::compute::concat_batches(&batches[0].schema(), &batches).unwrap(); -+ let distances = actual -+ .column_by_name("_distance") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap(); -+ assert_eq!( -+ distances.values().as_ref(), -+ if appended { -+ &[0., 0., 1., 1.][..] -+ } else { -+ &[0., 1.][..] -+ } -+ ); -+ let positions: Vec<_> = actual -+ .schema() -+ .fields() -+ .iter() -+ .enumerate() -+ .filter_map(|(i, field)| (field.name() != "_distance").then_some(i)) -+ .collect(); -+ assert_eq!( -+ actual.project(&positions).unwrap(), -+ expected, -+ "column={column:?} indexed={indexed} appended={appended} projection={projection:?}" -+ ); -+ } -+ } -+ } -+ } -+ } -+} -+ -+#[test] -+fn strict_batches_apply_after_multivector_offset_and_limit() { -+ let (_dir, uri) = custom_fixture( -+ (1..=6).map(|i| vec![Some(i as f32), Some(0.)]).collect(), -+ 2, -+ false, -+ ); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for batch_size in [Some(2), None] { -+ for (offset, limit) in [(1, 4), (1, 3), (5, 4), (6, 4), (0, 6)] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ c"vectors".as_ptr(), -+ [1.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 6 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ if let Some(size) = batch_size { -+ assert_eq!(lance_scanner_set_batch_size(scan, size), 0); -+ } -+ assert_eq!(lance_scanner_set_strict_batch_size(scan, true), 0); -+ assert_eq!(lance_scanner_set_offset(scan, offset), 0); -+ assert_eq!(lance_scanner_set_limit(scan, limit), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), 0); -+ let batches = ArrowArrayStreamReader::from_raw(&mut stream) -+ .unwrap() -+ .collect::, _>>() -+ .unwrap(); -+ lance_scanner_close(scan); -+ let sizes: Vec<_> = batches.iter().map(RecordBatch::num_rows).collect(); -+ let ids: Vec<_> = batches -+ .iter() -+ .flat_map(|b| { -+ b.column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .to_vec() -+ }) -+ .collect(); -+ let expected_ids: Vec<_> = -+ (1..=6).skip(offset as usize).take(limit as usize).collect(); -+ let expected_sizes: Vec<_> = expected_ids -+ .chunks(batch_size.unwrap_or(8192) as usize) -+ .map(<[i32]>::len) -+ .collect(); -+ assert_eq!(ids, expected_ids); -+ assert_eq!( -+ sizes, expected_sizes, -+ "batch_size={batch_size:?} offset={offset} limit={limit}" -+ ); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+} diff --git a/thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch b/thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch deleted file mode 100644 index 890093886fbcbb..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch +++ /dev/null @@ -1,148 +0,0 @@ -Subject: [PATCH] Include FTS prefilter metrics and full-snapshot test coverage - -Upstream: https://github.com/lance-format/lance/pull/9460 -Commit: f202fe41ac18323ca0cd7bc6efd8fd30b8722116 - -Backport the applicable review updates from Lance PR #9471 to v11. -Keep the C API and other dependencies unchanged, and upgrade previously -patched source caches with the same revision as fresh builds. - -diff --git a/Cargo.toml b/Cargo.toml ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -20,10 +20,10 @@ - [dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } --lance-core = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-index = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-io = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-table = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116", features = ["substrait"] } -+lance-core = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-index = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-io = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-table = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116", features = ["substrait"] } - datafusion = { version = "54.0.0", default-features = false } -@@ -53,5 +53,5 @@ - [dev-dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } --lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116", features = ["substrait"] } -+lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } - tokio = { version = "1", features = ["rt-multi-thread", "macros"] } -diff --git a/Cargo.lock b/Cargo.lock ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -2608,3 +2608,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3820,3 +3820,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3892,3 +3892,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3914,3 +3914,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3928,3 +3928,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3938,3 +3938,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3984,3 +3984,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4022,3 +4022,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4054,3 +4054,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4072,3 +4072,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4082,3 +4082,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4116,3 +4116,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4148,3 +4148,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4163,3 +4163,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4231,3 +4231,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4254,3 +4254,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4294,3 +4294,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4309,3 +4309,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4336,3 +4336,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4351,3 +4351,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4390,3 +4390,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ diff --git a/thirdparty/patches/lance-c-0.1.9-prefilter.patch b/thirdparty/patches/lance-c-0.1.9-prefilter.patch deleted file mode 100644 index c3df2f5d26244d..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-prefilter.patch +++ /dev/null @@ -1,147 +0,0 @@ -Subject: [PATCH] Use Lance full-snapshot prefilter optimization and loader metrics - -Upstream: https://github.com/lance-format/lance/pull/9460 -Commit: f75f3343b5e125c42da1bd7acc8d8217cd5660a6 - -Pin the tested Lance v11 change without upgrading the release or changing -the C API. Keep all Lance crates on the same revision. - -diff --git a/Cargo.toml b/Cargo.toml ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -20,10 +20,10 @@ - [dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } --lance-core = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-index = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-io = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-table = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } -+lance-core = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-index = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-io = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-table = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } - datafusion = { version = "54.0.0", default-features = false } -@@ -53,5 +53,5 @@ - [dev-dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } --lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } -+lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } - tokio = { version = "1", features = ["rt-multi-thread", "macros"] } -diff --git a/Cargo.lock b/Cargo.lock ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -2608,3 +2608,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3820,3 +3820,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3892,3 +3892,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3914,3 +3914,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3928,3 +3928,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3938,3 +3938,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3984,3 +3984,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4022,3 +4022,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4054,3 +4054,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4072,3 +4072,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4082,3 +4082,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4116,3 +4116,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4148,3 +4148,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4163,3 +4163,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4231,3 +4231,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4254,3 +4254,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4294,3 +4294,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4309,3 +4309,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4336,3 +4336,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4351,3 +4351,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4390,3 +4390,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ diff --git a/thirdparty/patches/lance-c-0.1.9-pr-73.patch b/thirdparty/patches/lance-c-foyer.patch similarity index 76% rename from thirdparty/patches/lance-c-0.1.9-pr-73.patch rename to thirdparty/patches/lance-c-foyer.patch index 33a8985b25aecc..73f244c49fd57d 100644 --- a/thirdparty/patches/lance-c-0.1.9-pr-73.patch +++ b/thirdparty/patches/lance-c-foyer.patch @@ -1,49 +1,16 @@ -From cc373477a6485cb24ed7ea88ef506e93caacedd0 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 3 Sep 2026 13:00:43 +0800 -Subject: [PATCH] foyer - ---- - Cargo.lock | 207 +++++- - Cargo.toml | 4 + - README.md | 22 + - include/lance/lance.h | 63 ++ - include/lance/lance.hpp | 29 + - src/data_cache.rs | 68 ++ - src/dataset.rs | 11 + - src/foyer_data_cache.rs | 1215 ++++++++++++++++++++++++++++++++++++ - src/lib.rs | 4 + - src/restore.rs | 8 + - src/session.rs | 13 +- - src/writer.rs | 1 + - tests/c_api_test.rs | 221 +++++++ - tests/cpp/test_c_api.c | 32 + - tests/cpp/test_cpp_api.cpp | 23 + - 15 files changed, 1917 insertions(+), 4 deletions(-) - create mode 100644 src/data_cache.rs - create mode 100644 src/foyer_data_cache.rs - +# Foyer data-cache integration for lance-format/lance-c#73. +# Base: 9bd730add2ac70316c1d642b8459011e2dd92022 +# Source: https://github.com/Gabriel39/lance-c/commit/24c7ca4bcb9422c113b0d3e07e4efe1173b0bc9f +# Regenerate the payload in lance-c; do not maintain separate downstream edits: +# git diff --full-index --binary 9bd730add2ac70316c1d642b8459011e2dd92022 24c7ca4bcb9422c113b0d3e07e4efe1173b0bc9f -- . | sed 's/^ $//' diff --git a/Cargo.lock b/Cargo.lock -index bc37cb9..199f997 100644 +index 78812f1330e146db295a14276f90f9e28654f399..daf9f85daf3e0ea59bb906e8e3c32470519b084b 100644 --- a/Cargo.lock +++ b/Cargo.lock -@@ -444,6 +444,12 @@ dependencies = [ - "slab", - ] - -+[[package]] -+name = "asyncband" -+version = "0.7.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "2e52766975a4f080528a898235c51e82e65df9db713419067b5368040eeb5659" -+ - [[package]] - name = "atoi" - version = "2.0.0" -@@ -1293,6 +1299,17 @@ version = "0.8.7" +@@ -1193,6 +1193,17 @@ version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - + +[[package]] +name = "core_affinity" +version = "0.8.3" @@ -58,23 +25,10 @@ index bc37cb9..199f997 100644 [[package]] name = "countio" version = "0.3.0" -@@ -2148,6 +2165,12 @@ dependencies = [ - "url", - ] - -+[[package]] -+name = "datasketches" -+version = "0.3.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "46c4cf71a36b46dcfc00e5014c0c20ccad2b1b6a008304d7d57d2749b2d41b3d" -+ - [[package]] - name = "defmt" - version = "1.1.1" -@@ -2366,6 +2389,16 @@ version = "0.2.3" +@@ -2241,6 +2252,16 @@ version = "0.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8eb564c5c7423d25c886fb561d1e4ee69f72354d16918afa32c08811f6b6a55" - + +[[package]] +name = "fastant" +version = "0.1.11" @@ -88,10 +42,10 @@ index bc37cb9..199f997 100644 [[package]] name = "fastrand" version = "2.3.0" -@@ -2437,12 +2470,133 @@ dependencies = [ +@@ -2312,6 +2333,26 @@ dependencies = [ "percent-encoding", ] - + +[[package]] +name = "foyer" +version = "0.22.5" @@ -99,7 +53,7 @@ index bc37cb9..199f997 100644 +checksum = "f911e6f0b4909f23d65a95c5d27bcf2f92855b8a0f629b47b743d53d14f2828b" +dependencies = [ + "anyhow", -+ "asyncband 0.7.1", ++ "asyncband", + "equivalent", + "foyer-common", + "foyer-memory", @@ -112,65 +66,22 @@ index bc37cb9..199f997 100644 + "tracing", +] + -+[[package]] -+name = "foyer-common" -+version = "0.22.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "05cdcae6cedec72c28e97ada0b453e33b1df57fbc209708091107be2a283d577" -+dependencies = [ -+ "anyhow", -+ "bytes", -+ "cfg-if 1.0.4", -+ "foyer-tokio", -+ "mixtrics", -+ "parking_lot", -+ "pin-project", -+ "twox-hash", -+] -+ -+[[package]] -+name = "foyer-intrusive-collections" -+version = "0.10.0-dev" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "6e4fee46bea69e0596130e3210e65d3424e0ac1e6df3bde6636304bdf1ca4a3b" -+dependencies = [ -+ "memoffset", -+] -+ -+[[package]] -+name = "foyer-memory" -+version = "0.22.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "8a51c8ce8e1e323a1e087ac45bf629d953676fc5e0037d14a799fadd272b42db" -+dependencies = [ -+ "anyhow", -+ "asyncband 0.7.1", -+ "bitflags 2.11.0", -+ "datasketches", -+ "equivalent", -+ "foyer-common", -+ "foyer-intrusive-collections", -+ "foyer-tokio", -+ "futures-util", -+ "hashbrown 0.17.1", -+ "itertools 0.15.0", -+ "mixtrics", -+ "parking_lot", -+ "paste", -+ "pin-project", -+ "serde", -+ "tracing", -+] -+ + [[package]] + name = "foyer-common" + version = "0.22.6" +@@ -2362,6 +2403,38 @@ dependencies = [ + "tracing", + ] + +[[package]] +name = "foyer-storage" -+version = "0.22.5" ++version = "0.22.6" +source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "127a64057f63e361123cf62b3fd50a36147783687d4cd36e7087c27f9909265b" ++checksum = "118192f532ba013f0cdc607efc22324b9ed855baed185608af0daa986e835c6b" +dependencies = [ + "allocator-api2", + "anyhow", -+ "asyncband 0.7.1", ++ "asyncband", + "bytes", + "core_affinity", + "equivalent", @@ -193,22 +104,14 @@ index bc37cb9..199f997 100644 + "twox-hash", + "zstd", +] -+ -+[[package]] -+name = "foyer-tokio" -+version = "0.22.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "cbdbb9f39443cb348a069baa1a0ec73bcea848a4a383eb2da1a5ea7a0ca05941" -+dependencies = [ -+ "tokio", -+] + [[package]] - name = "frostem" - version = "1.20260821.5" + name = "foyer-tokio" + version = "0.22.6" +@@ -2377,6 +2450,16 @@ version = "1.20260821.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "36a80a7406da302e04bfd2ca987907590d3a1f3c69958947c43890abd7426b2f" - + +[[package]] +name = "fs4" +version = "0.13.1" @@ -222,27 +125,10 @@ index bc37cb9..199f997 100644 [[package]] name = "fs_extra" version = "1.3.0" -@@ -3485,6 +3639,15 @@ dependencies = [ - "either", - ] - -+[[package]] -+name = "itertools" -+version = "0.15.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc" -+dependencies = [ -+ "either", -+] -+ - [[package]] - name = "itoa" - version = "1.0.18" -@@ -3788,8 +3951,11 @@ dependencies = [ - "arrow", +@@ -3720,8 +3803,10 @@ dependencies = [ "arrow-array", "arrow-schema", -+ "async-trait", + "async-trait", + "bytes", "chrono", "datafusion", @@ -250,78 +136,18 @@ index bc37cb9..199f997 100644 "futures", "half", "lance", -@@ -3803,6 +3969,7 @@ dependencies = [ +@@ -3735,6 +3820,7 @@ dependencies = [ "lance-table", "libc", "log", -+ "object_store", ++ "object_store 0.14.2", + "opendal", "pin-project", "prost", - "snafu", -@@ -4492,6 +4659,15 @@ dependencies = [ - "libc", - ] - -+[[package]] -+name = "memoffset" -+version = "0.9.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "488016bfae457b036d996092f6cb448677611ce4449e970ceaf42695203f218a" -+dependencies = [ -+ "autocfg", -+] -+ - [[package]] - name = "mime" - version = "0.3.17" -@@ -4529,6 +4705,16 @@ dependencies = [ - "windows-sys 0.61.2", - ] - -+[[package]] -+name = "mixtrics" -+version = "0.2.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "2c46b5adfb7a3ae4996d327a5bdc90e78fec025806dd312bdbe6f07a755e0ec9" -+dependencies = [ -+ "itertools 0.15.0", -+ "parking_lot", -+] -+ - [[package]] - name = "moka" - version = "0.12.15" -@@ -4840,7 +5026,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "48dbcef97d3eb7591db2c18d5cae95c836bcce07359b98d98dd6f4e861eb77b7" - dependencies = [ - "anyhow", -- "asyncband", -+ "asyncband 0.6.7", - "base64 0.23.1", - "bytes", - "futures", -@@ -4879,7 +5065,7 @@ version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "03f9e144b5228d741c3763ade8711d9b72e5fb6d998e779f2d7a09da0b5a3eba" - dependencies = [ -- "asyncband", -+ "asyncband 0.6.7", - "futures", - "http 1.4.0", - "opendal-core", -@@ -4943,7 +5129,7 @@ version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "2d564484a8f7d091827e825cfc91ed45bd48e64d262451ee041fe843db81bd8a" - dependencies = [ -- "asyncband", -+ "asyncband 0.6.7", - "base64 0.23.1", - "bytes", - "http 1.4.0", -@@ -6668,6 +6854,12 @@ version = "0.4.12" +@@ -6628,6 +6714,12 @@ version = "0.4.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - + +[[package]] +name = "small_ctor" +version = "0.1.2" @@ -331,10 +157,10 @@ index bc37cb9..199f997 100644 [[package]] name = "smallvec" version = "1.15.1" -@@ -7930,6 +8122,15 @@ dependencies = [ +@@ -7897,6 +7989,15 @@ dependencies = [ "windows-targets 0.52.6", ] - + +[[package]] +name = "windows-sys" +version = "0.59.0" @@ -348,31 +174,32 @@ index bc37cb9..199f997 100644 name = "windows-sys" version = "0.60.2" diff --git a/Cargo.toml b/Cargo.toml -index 3920a65..356c9dd 100644 +index 18654b862440a90aa6bf0dd846e8f1ad6036fd7c..fd134a7151e1df61da201d867a642e9bcbf645d5 100644 --- a/Cargo.toml +++ b/Cargo.toml -@@ -30,6 +30,8 @@ datafusion = { version = "54.0.0", default-features = false } +@@ -35,6 +35,7 @@ datafusion = { version = "54.0.0", default-features = false } arrow = { version = "58.0.0", features = ["prettyprint", "ffi"] } arrow-array = "58.0.0" arrow-schema = "58.0.0" -+async-trait = "0.1" +bytes = "1" # Direct to name `chrono::TimeDelta` (the field type of lance's public # `AutoCleanupParams`) and `chrono::DateTime`/`Utc` (index metadata # timestamps); already in the graph transitively via lance. -@@ -37,8 +39,10 @@ chrono = { version = "0.4", default-features = false } +@@ -42,10 +43,12 @@ chrono = { version = "0.4", default-features = false } half = "2" tokio = { version = "1", features = ["rt-multi-thread", "sync"] } futures = "0.3" +foyer = "=0.22.5" log = "0.4" libc = "0.2" -+object_store = "0.13.2" + # Explicitly install the HTTP transport when embedded in a static C/C++ executable. + opendal = { version = "=0.59.2", default-features = false, features = ["http-transport-reqwest"] } ++object_store = "0.14.1" pin-project = "1.0" prost = "0.14" snafu = "0.9" diff --git a/README.md b/README.md -index 2056671..d9a5f2b 100644 +index 9ceccdc4f3f6b5d13dc271be3d2659c7be400306..94de4d1c26d8ab869d1726e2b416d21f21661d6d 100644 --- a/README.md +++ b/README.md @@ -68,6 +68,7 @@ Based on the [liblance RFC](https://github.com/lance-format/lance/discussions/60 @@ -380,19 +207,27 @@ index 2056671..d9a5f2b 100644 | [x] | Dataset metadata | `lance_dataset_version()`, `lance_dataset_count_rows()`, `lance_dataset_latest_version()` | | [x] | Filter pushdown | `lance_scanner_set_substrait_filter()` accepts a serialized Substrait `ExtendedExpression`; `lance_scanner_additional_sql_filter()` adds SQL predicates with AND before scanning starts | +| [x] | Data-file cache | Optional Foyer memory/disk cache for immutable `data/*.lance` reads | - - ## Building - -@@ -197,6 +198,27 @@ auto ds = lance::Dataset::open_with_session(session, "data.lance"); + + ## Multi-vector search + +@@ -231,6 +232,35 @@ auto ds = lance::Dataset::open_with_session(session, "data.lance"); auto stats = session.cache_stats(); ``` - + +To add a process-local memory/disk cache for remote Lance data-file reads, +create the session with Foyer configuration. The cache is deliberately narrow: +whole-object, single-range, and batched range reads of direct `data/*.lance` +children are cached. Conditional and versioned reads, plus manifests, deletion +files, and index files, keep using Lance's normal paths. Use one shared session -+for datasets that share the cache directory. ++for datasets that share the cache directory. Immutable data-file response metadata ++uses a separate in-memory cache budget of one eighth of the configured data ++memory capacity, avoiding a remote HEAD request on repeated range reads. ++Cache entries are isolated by the underlying object-store instance because the ++wrapper interface does not expose a complete backend identity. Datasets sharing ++the same live store can reuse entries; a new store instance or process restart ++starts a new cache namespace. Reopening the disk tier while that same store is ++still alive can recover its entries. Identical bucket/path names on different ++endpoints never share metadata, sizes, or data blocks. + +```cpp +lance::DataCacheOptions data_cache{ @@ -409,16 +244,16 @@ index 2056671..d9a5f2b 100644 +``` + ### Open at a specific version - + `lance_dataset_open` takes a `version` argument — `0` means the latest, any diff --git a/include/lance/lance.h b/include/lance/lance.h -index 31213da..152351b 100644 +index 7630913fd7849d0bff2cb0f31e1ab1199cb9f0fb..1c1d0752efdda5023b29eafe5aa9a87e03dc5303 100644 --- a/include/lance/lance.h +++ b/include/lance/lance.h -@@ -206,6 +206,36 @@ typedef struct LanceSessionCacheStats { +@@ -214,6 +214,36 @@ typedef struct LanceSessionCacheStats { uint64_t metadata_cache_size_bytes; } LanceSessionCacheStats; - + +/** + * Configuration for the optional Foyer cache of immutable Lance data files. + * @@ -452,10 +287,10 @@ index 31213da..152351b 100644 /** * Create a session that can share metadata and index caches across datasets. * -@@ -217,6 +247,25 @@ LanceSession* lance_session_new( +@@ -225,6 +255,25 @@ LanceSession* lance_session_new( uint64_t metadata_cache_size_bytes ); - + +/** + * Create a shared Lance session with a Foyer data-file cache. + * @@ -478,10 +313,10 @@ index 31213da..152351b 100644 /** * Close a session handle. Safe to call with NULL. Datasets previously opened * with the session remain valid and retain the shared cache state. -@@ -273,6 +322,20 @@ LanceDataset* lance_dataset_open_with_session( +@@ -281,6 +330,20 @@ LanceDataset* lance_dataset_open_with_session( const LanceSession* session ); - + +/** + * Copy this dataset handle's cumulative data-cache statistics. + * @@ -498,15 +333,15 @@ index 31213da..152351b 100644 + /** Close and free a dataset handle. Safe to call with NULL. */ void lance_dataset_close(LanceDataset* dataset); - + diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 404d2df..3fe9dbc 100644 +index 286724e96736e354785e046f2fcfe5ae7c65147b..5070d423453a5d60d31d03ce758ad7dbe83a6ea2 100644 --- a/include/lance/lance.hpp +++ b/include/lance/lance.hpp @@ -176,6 +176,13 @@ struct SqlColumn { - + // ─── Shared Session ────────────────────────────────────────────────────────── - + +struct DataCacheOptions { + std::string directory; + uint64_t memory_capacity_bytes; @@ -516,11 +351,11 @@ index 404d2df..3fe9dbc 100644 + class Session { Handle handle_; - -@@ -185,6 +192,21 @@ class Session { + +@@ -185,6 +192,21 @@ public: if (!handle_) check_error(); } - + + Session(uint64_t index_cache_size_bytes, + uint64_t metadata_cache_size_bytes, + const DataCacheOptions& data_cache_options) { @@ -539,10 +374,10 @@ index 404d2df..3fe9dbc 100644 LanceSessionCacheStats cache_stats() const { LanceSessionCacheStats stats{}; if (lance_session_get_cache_stats(handle_.get(), &stats) != 0) -@@ -265,6 +287,13 @@ class Dataset { +@@ -350,6 +372,13 @@ public: return Dataset(ds); } - + + LanceDataCacheStatistics data_cache_statistics() const { + LanceDataCacheStatistics statistics{}; + if (lance_dataset_get_data_cache_statistics(handle_.get(), &statistics) != 0) @@ -555,7 +390,7 @@ index 404d2df..3fe9dbc 100644 /// diff --git a/src/data_cache.rs b/src/data_cache.rs new file mode 100644 -index 0000000..430b4c8 +index 0000000000000000000000000000000000000000..430b4c891f91d31f202e2318e18db64c7a12ae9d --- /dev/null +++ b/src/data_cache.rs @@ -0,0 +1,68 @@ @@ -628,13 +463,13 @@ index 0000000..430b4c8 + Ok(0) +} diff --git a/src/dataset.rs b/src/dataset.rs -index cc1f87c..76fd39e 100644 +index cc1f87ce3f7dfe8dd33aeb6ab7f54d02e31a220a..76fd39ef9e456d1fab6978b6b2cb3c27fab1c2f3 100644 --- a/src/dataset.rs +++ b/src/dataset.rs @@ -14,6 +14,7 @@ use lance::Dataset; use lance::dataset::builder::DatasetBuilder; use lance_core::Result; - + +use crate::data_cache::DatasetDataCache; use crate::error::{ffi_try, swallow_unwind}; use crate::helpers; @@ -645,11 +480,11 @@ index cc1f87c..76fd39e 100644 pub(crate) inner: RwLock>, + pub(crate) data_cache: Option>, } - + impl LanceDataset { @@ -182,8 +184,16 @@ unsafe fn open_dataset_inner( } - + let dataset = block_on(builder.load())?; + let (dataset, data_cache) = + if let Some(factory) = session.and_then(|session| session.data_cache_factory.clone()) { @@ -674,10 +509,10 @@ index cc1f87c..76fd39e 100644 } diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs new file mode 100644 -index 0000000..9a10e43 +index 0000000000000000000000000000000000000000..cfc47f1528d11b6be068c09dd4703ff4308a9d16 --- /dev/null +++ b/src/foyer_data_cache.rs -@@ -0,0 +1,1215 @@ +@@ -0,0 +1,1620 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + @@ -689,21 +524,22 @@ index 0000000..9a10e43 +use std::ops::Range; +use std::path::Path as FsPath; +use std::sync::atomic::{AtomicU64, Ordering}; -+use std::sync::{Arc, Mutex, Weak}; ++use std::sync::{Arc, LazyLock, Mutex, Weak}; + +use async_trait::async_trait; +use bytes::{Bytes, BytesMut}; +use foyer::{ -+ BlockEngineConfig, DeviceBuilder, FsDeviceBuilder, HybridCache, HybridCacheBuilder, -+ HybridCachePolicy, PsyncIoEngineConfig, ++ BlockEngineConfig, Cache, CacheBuilder, DeviceBuilder, FsDeviceBuilder, HybridCache, ++ HybridCacheBuilder, HybridCachePolicy, PsyncIoEngineConfig, +}; +use futures::stream::BoxStream; +use lance_io::object_store::WrappingObjectStore; ++use object_store::list::PaginatedListStore; +use object_store::path::Path; +use object_store::{ -+ CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, ObjectMeta, -+ ObjectStore, ObjectStoreExt, PutMultipartOptions, PutOptions, PutPayload, PutResult, -+ RenameOptions, Result, ++ Attribute, Attributes, CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, ++ MultipartUpload, ObjectMeta, ObjectStore, ObjectStoreExt, PutMultipartOptions, PutOptions, ++ PutPayload, PutResult, RenameOptions, Result, +}; + +use crate::data_cache::{DataCacheFactory, DatasetDataCache, LanceDataCacheStatistics}; @@ -712,9 +548,32 @@ index 0000000..9a10e43 +use crate::runtime::block_on; +use crate::session::{LanceSession, session_new_with_data_cache_factory}; + -+const CACHE_KEY_VERSION: &str = "lance-data-v1"; ++const CACHE_KEY_VERSION: &str = "lance-data-v2"; +const FOYER_PAGE_SIZE: usize = 4096; + ++struct OriginNamespace { ++ origin: Weak, ++ namespace: uuid::Uuid, ++} ++ ++static ORIGIN_NAMESPACES: LazyLock>> = ++ LazyLock::new(|| Mutex::new(HashMap::new())); ++ ++fn origin_namespace(origin: &Arc) -> uuid::Uuid { ++ // store_prefix omits endpoint/credentials, and this interface exposes no stable ++ // backend identity. Only the same live origin may reuse metadata or data. ++ // Persist a random namespace, never an address that another process can reuse. ++ let mut namespaces = ORIGIN_NAMESPACES.lock().unwrap(); ++ namespaces.retain(|_, entry| entry.origin.strong_count() != 0); ++ namespaces ++ .entry(Arc::as_ptr(origin) as *const () as usize) ++ .or_insert_with(|| OriginNamespace { ++ origin: Arc::downgrade(origin), ++ namespace: uuid::Uuid::new_v4(), ++ }) ++ .namespace ++} ++ +/// Configuration for the optional Foyer data-file cache. +#[repr(C)] +#[derive(Clone, Copy, Debug)] @@ -846,6 +705,7 @@ index 0000000..9a10e43 +#[derive(Clone)] +pub(crate) struct FoyerDataCache { + cache: HybridCache, ++ metadata: Cache, + read_block_size: usize, + wrapped_stores: Arc>>, +} @@ -879,6 +739,33 @@ index 0000000..9a10e43 + .with_capacity(disk_capacity) + .build()?; + let engine = BlockEngineConfig::new(device).with_block_size(engine_block_size); ++ // Single-range reads use get_opts(), which must return object metadata. ++ // Share a bounded metadata cache across dataset scopes so a data-cache hit ++ // does not require another HEAD request for an immutable data file. ++ let metadata_capacity = memory_capacity / 8; ++ let metadata = CacheBuilder::new(metadata_capacity) ++ .with_shards(1) ++ .with_weighter( ++ |key: &String, (meta, attributes): &(ObjectMeta, Attributes)| { ++ key.len() ++ + std::mem::size_of::<(ObjectMeta, Attributes)>() ++ + meta.location.as_ref().len() ++ + meta.e_tag.as_ref().map_or(0, String::len) ++ + meta.version.as_ref().map_or(0, String::len) ++ + attributes ++ .iter() ++ .map(|(attribute, value)| { ++ std::mem::size_of::<(Attribute, object_store::AttributeValue)>() ++ + value.as_ref().len() ++ + match attribute { ++ Attribute::Metadata(name) => name.len(), ++ _ => 0, ++ } ++ }) ++ .sum::() ++ }, ++ ) ++ .build(); + let cache = HybridCacheBuilder::new() + .with_name("lance_data") + .with_policy(HybridCachePolicy::WriteOnInsertion) @@ -895,6 +782,7 @@ index 0000000..9a10e43 + .await?; + Ok(Self { + cache, ++ metadata, + read_block_size, + wrapped_stores: Arc::new(Mutex::new(HashMap::new())), + }) @@ -1034,7 +922,7 @@ index 0000000..9a10e43 + let original = self.cache.unwrap_store(original); + let reader = DataCacheReader { + cache: self.cache.clone(), -+ store_prefix: store_prefix.to_owned(), ++ store_prefix: format!("{}\0{store_prefix}", origin_namespace(&original)), + original: original.clone(), + statistics: self.statistics.clone(), + }; @@ -1047,6 +935,15 @@ index 0000000..9a10e43 + self.cache.remember_wrapper(&wrapped, &original); + wrapped + } ++ ++ fn wrap_paginated( ++ &self, ++ _store_prefix: &str, ++ original: Arc, ++ ) -> Option> { ++ // Data caching does not hide or rewrite paths, so keep listing pushdown. ++ Some(original) ++ } +} + +#[derive(Debug)] @@ -1087,31 +984,52 @@ index 0000000..9a10e43 + } + + async fn cached_get(&self, location: &Path, options: GetOptions) -> Result { -+ // Fetch metadata separately so the returned GetResult retains the origin's identity while -+ // its payload uses the same block cache as get_ranges(). This also provides the object size -+ // needed to resolve bounded, offset, and suffix ranges. -+ let GetResult { -+ meta: metadata, -+ attributes, -+ .. -+ } = self ++ let metadata_key = self + .reader -+ .original -+ .get_opts( -+ location, -+ GetOptions { -+ head: true, -+ ..Default::default() -+ }, -+ ) -+ .await?; ++ .cache ++ .size_key(&self.reader.store_prefix, location); ++ let cached_metadata = self.reader.cache.metadata.get(&metadata_key); ++ let (metadata, attributes, extensions) = if let Some(entry) = cached_metadata { ++ let (metadata, attributes) = entry.value(); ++ (metadata.clone(), attributes.clone(), Default::default()) ++ } else { ++ let result = self ++ .reader ++ .original ++ .get_opts( ++ location, ++ GetOptions { ++ head: true, ++ ..Default::default() ++ }, ++ ) ++ .await?; ++ // HTTP responses carry transport extensions even for immutable files. ++ // Cache metadata independently; request-specific extensions are never replayed. ++ self.reader.cache.metadata.insert( ++ metadata_key.clone(), ++ (result.meta.clone(), result.attributes.clone()), ++ ); ++ (result.meta, result.attributes, result.extensions) ++ }; + let object_size = metadata.size; -+ self.reader.cache.cache.insert( ++ let size_bytes = object_size.to_le_bytes(); ++ // WriteOnInsertion enqueues disk I/O even for an identical value. Look in ++ // both tiers so warm reads, including recovered entries, remain read-only. ++ let size_is_cached = match self.reader.cache.cache.get(&metadata_key).await { ++ Ok(Some(entry)) => entry.value().as_ref() == size_bytes, ++ Ok(None) => false, ++ Err(error) => { ++ log::warn!("Foyer data-cache size lookup failed for {location}: {error}"); ++ false ++ } ++ }; ++ if !size_is_cached { + self.reader + .cache -+ .size_key(&self.reader.store_prefix, location), -+ Bytes::copy_from_slice(&object_size.to_le_bytes()), -+ ); ++ .cache ++ .insert(metadata_key, Bytes::copy_from_slice(&size_bytes)); ++ } + + let range = match options.range.clone() { + Some(requested) => match requested.as_range(object_size) { @@ -1157,6 +1075,7 @@ index 0000000..9a10e43 + meta: metadata, + range, + attributes, ++ extensions, + }) + } +} @@ -1493,6 +1412,282 @@ index 0000000..9a10e43 + (scope.wrap("memory://test", original), scope) + } + ++ async fn http_store( ++ data: Bytes, ++ ) -> ( ++ Arc, ++ Arc, ++ Arc, ++ tokio::task::JoinHandle<()>, ++ ) { ++ use object_store::aws::AmazonS3Builder; ++ use tokio::io::{AsyncReadExt, AsyncWriteExt}; ++ ++ let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); ++ let address = listener.local_addr().unwrap(); ++ let heads = Arc::new(AtomicU64::new(0)); ++ let gets = Arc::new(AtomicU64::new(0)); ++ let server_heads = heads.clone(); ++ let server_gets = gets.clone(); ++ let server_data = data.clone(); ++ let server = tokio::spawn(async move { ++ loop { ++ let (mut socket, _) = listener.accept().await.unwrap(); ++ let mut request = Vec::new(); ++ while !request.ends_with(b"\r\n\r\n") { ++ request.push(socket.read_u8().await.unwrap()); ++ assert!(request.len() < 16 * 1024); ++ } ++ let request = String::from_utf8(request).unwrap(); ++ let is_head = request.starts_with("HEAD "); ++ let range = request.lines().find_map(|line| { ++ line.to_ascii_lowercase() ++ .strip_prefix("range: bytes=") ++ .map(|range| { ++ let (start, end) = range.split_once('-').unwrap(); ++ start.parse::().unwrap()..end.parse::().unwrap() + 1 ++ }) ++ }); ++ let (status, content_range, range) = match range { ++ Some(range) => ( ++ "206 Partial Content", ++ format!( ++ "Content-Range: bytes {}-{}/{}\r\n", ++ range.start, ++ range.end - 1, ++ server_data.len() ++ ), ++ range, ++ ), ++ None => ("200 OK", String::new(), 0..server_data.len()), ++ }; ++ if is_head { ++ server_heads.fetch_add(1, Ordering::SeqCst); ++ } else { ++ server_gets.fetch_add(1, Ordering::SeqCst); ++ } ++ let headers = format!( ++ "HTTP/1.1 {status}\r\nContent-Length: {}\r\n{content_range}Last-Modified: Tue, 01 Sep 2026 00:00:00 GMT\r\nETag: \"sample\"\r\nContent-Type: application/octet-stream\r\nConnection: close\r\n\r\n", ++ range.len() ++ ); ++ socket.write_all(headers.as_bytes()).await.unwrap(); ++ if !is_head { ++ socket.write_all(&server_data[range]).await.unwrap(); ++ } ++ } ++ }); ++ ++ // Use a real HTTP client: in-memory stores do not exercise transport extensions. ++ let original = Arc::new( ++ AmazonS3Builder::new() ++ .with_bucket_name("example-bucket") ++ .with_region("us-east-1") ++ .with_endpoint(format!("http://{address}")) ++ .with_allow_http(true) ++ // An explicit excluded proxy also disables environment proxy discovery. ++ .with_proxy_url("http://127.0.0.1:1") ++ .with_proxy_excludes("127.0.0.1") ++ .with_skip_signature(true) ++ .build() ++ .unwrap(), ++ ); ++ (original, heads, gets, server) ++ } ++ ++ #[tokio::test] ++ async fn cached_http_ranges_do_not_repeat_head_requests() { ++ let data = Bytes::from((0..8192).map(|value| value as u8).collect::>()); ++ let (original, heads, gets, server) = http_store(data.clone()).await; ++ let directory = tempfile::tempdir().unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let path = Path::from("table.lance/data/sample.lance"); ++ let (wrapped, _) = wrap_for_test(&cache, original.clone()); ++ let first = wrapped ++ .get_opts( ++ &path, ++ GetOptions { ++ range: Some(GetRange::Bounded(100..200)), ++ ..Default::default() ++ }, ++ ) ++ .await ++ .unwrap(); ++ assert!( ++ !first.extensions.is_empty(), ++ "HTTP response must exercise transport extensions" ++ ); ++ let meta = first.meta.clone(); ++ let attributes = first.attributes.clone(); ++ assert_eq!(first.range, 100..200); ++ assert_eq!(first.bytes().await.unwrap(), data.slice(100..200)); ++ assert_eq!(heads.load(Ordering::SeqCst), 1); ++ assert_eq!(gets.load(Ordering::SeqCst), 1); ++ ++ // Metadata must be shared with fresh dataset scopes, just like cached blocks. ++ let (fresh, _) = wrap_for_test(&cache, original); ++ let mut replayed_extensions = false; ++ for store in [&wrapped, &fresh] { ++ let result = store ++ .get_opts( ++ &path, ++ GetOptions { ++ range: Some(GetRange::Bounded(120..180)), ++ ..Default::default() ++ }, ++ ) ++ .await ++ .unwrap(); ++ replayed_extensions |= !result.extensions.is_empty(); ++ assert_eq!(result.meta, meta); ++ assert_eq!(result.attributes, attributes); ++ assert_eq!(result.range, 120..180); ++ assert_eq!(result.bytes().await.unwrap(), data.slice(120..180)); ++ } ++ assert_eq!( ++ heads.load(Ordering::SeqCst), ++ 1, ++ "cache hits must not issue HEAD" ++ ); ++ assert_eq!( ++ gets.load(Ordering::SeqCst), ++ 1, ++ "cache hits must not issue GET" ++ ); ++ assert!( ++ !replayed_extensions, ++ "cached metadata must not replay response extensions" ++ ); ++ wrapped.head(&path).await.unwrap(); ++ assert_eq!( ++ heads.load(Ordering::SeqCst), ++ 2, ++ "explicit HEAD must bypass cache" ++ ); ++ server.abort(); ++ } ++ ++ #[tokio::test] ++ async fn same_bucket_and_path_on_distinct_endpoints_remain_isolated() { ++ let (first, _, _, first_server) = http_store(Bytes::from(vec![11; 8192])).await; ++ let (second, second_heads, second_gets, second_server) = ++ http_store(Bytes::from(vec![29; 4096])).await; ++ let directory = tempfile::tempdir().unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let path = Path::from("table.lance/data/shared.lance"); ++ let first = cache.create_scope().wrap("s3$example-bucket", first); ++ let second = cache.create_scope().wrap("s3$example-bucket", second); ++ assert_eq!( ++ first.get(&path).await.unwrap().bytes().await.unwrap(), ++ Bytes::from(vec![11; 8192]) ++ ); ++ let result = second.get(&path).await.unwrap(); ++ assert_eq!( ++ result.meta.size, 4096, ++ "metadata must belong to the second endpoint" ++ ); ++ assert_eq!(result.bytes().await.unwrap(), Bytes::from(vec![29; 4096])); ++ assert_eq!(second_heads.load(Ordering::SeqCst), 1); ++ assert_eq!(second_gets.load(Ordering::SeqCst), 1); ++ first_server.abort(); ++ second_server.abort(); ++ } ++ ++ #[tokio::test] ++ async fn batched_ranges_do_not_reuse_another_store_or_mask_not_found() { ++ let directory = tempfile::tempdir().unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let first = Arc::new(InMemory::new()); ++ let second = Arc::new(InMemory::new()); ++ let missing = Arc::new(InMemory::new()); ++ let path = Path::from("table.lance/data/shared.lance"); ++ first ++ .put(&path, Bytes::from(vec![11; 8192]).into()) ++ .await ++ .unwrap(); ++ second ++ .put(&path, Bytes::from(vec![29; 8192]).into()) ++ .await ++ .unwrap(); ++ let (first, _) = wrap_for_test(&cache, first); ++ let (second, _) = wrap_for_test(&cache, second); ++ let (missing, _) = wrap_for_test(&cache, missing); ++ assert_eq!( ++ first ++ .get_ranges(&path, &[0..4096, 4096..8192]) ++ .await ++ .unwrap()[0], ++ Bytes::from(vec![11; 4096]) ++ ); ++ assert_eq!( ++ second ++ .get_ranges(&path, &[0..4096, 4096..8192]) ++ .await ++ .unwrap()[0], ++ Bytes::from(vec![29; 4096]) ++ ); ++ assert!(matches!( ++ missing.get(&path).await, ++ Err(object_store::Error::NotFound { .. }) ++ )); ++ assert!(matches!( ++ missing.get_ranges(&path, &[0..16, 16..32]).await, ++ Err(object_store::Error::NotFound { .. }) ++ )); ++ } ++ ++ #[tokio::test] ++ async fn warm_range_reads_do_not_rewrite_size_entries_to_disk() { ++ let directory = tempfile::tempdir().unwrap(); ++ let original = Arc::new(InMemory::new()); ++ let path = Path::from("table.lance/data/sample.lance"); ++ original ++ .put(&path, Bytes::from(vec![7; 8192]).into()) ++ .await ++ .unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let (wrapped, _) = wrap_for_test(&cache, original.clone()); ++ assert_eq!( ++ wrapped ++ .get(&path) ++ .await ++ .unwrap() ++ .bytes() ++ .await ++ .unwrap() ++ .len(), ++ 8192 ++ ); ++ drop(wrapped); ++ cache.cache.close().await.unwrap(); ++ drop(cache); ++ ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let (wrapped, _) = wrap_for_test(&cache, original); ++ for _ in 0..5 { ++ assert_eq!( ++ wrapped.get_range(&path, 100..200).await.unwrap(), ++ Bytes::from(vec![7; 100]) ++ ); ++ } ++ // Drain the real disk writer so asynchronous enqueues cannot hide behind the assertion. ++ cache.cache.close().await.unwrap(); ++ assert_eq!( ++ cache.cache.statistics().disk_write_bytes(), ++ 0, ++ "warm reads must not enqueue size records for disk storage" ++ ); ++ } ++ + #[tokio::test] + async fn caches_only_immutable_data_file_ranges() { + let directory = tempfile::tempdir().unwrap(); @@ -1674,7 +1869,7 @@ index 0000000..9a10e43 + original.put(path, data.clone().into()).await.unwrap(); + } + -+ let (wrapped, statistics) = wrap_for_test(&cache, original); ++ let (wrapped, statistics) = wrap_for_test(&cache, original.clone()); + for (path, requested, expected_range) in cases { + let expected = data.slice(expected_range.start as usize..expected_range.end as usize); + let before = statistics.snapshot(); @@ -1691,14 +1886,16 @@ index 0000000..9a10e43 + before.bytes_read_from_remote + expected.len() as u64 + ); + ++ // Cached data reads must not depend on an extra origin HEAD request. ++ let expected_meta = original.head(&path).await.unwrap(); ++ original.delete(&path).await.unwrap(); + let second = wrapped + .get_opts(&path, GetOptions::new().with_range(Some(requested))) + .await -+ .unwrap() -+ .bytes() -+ .await + .unwrap(); -+ assert_eq!(second, expected); ++ assert_eq!(second.meta, expected_meta); ++ assert_eq!(second.range, expected_range); ++ assert_eq!(second.bytes().await.unwrap(), expected); + assert_eq!( + statistics.snapshot().bytes_read_from_cache, + before.bytes_read_from_cache + expected.len() as u64 @@ -1769,6 +1966,49 @@ index 0000000..9a10e43 + } + + #[tokio::test] ++ async fn recovered_disk_entries_do_not_outlive_their_origin_identity() { ++ let directory = tempfile::tempdir().unwrap(); ++ let path = Path::from("table.lance/data/shared.lance"); ++ let original = Arc::new(InMemory::new()); ++ let weak = Arc::downgrade(&original); ++ original ++ .put(&path, Bytes::from(vec![11; 8192]).into()) ++ .await ++ .unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let (wrapped, _) = wrap_for_test(&cache, original); ++ assert_eq!(wrapped.get_range(&path, 0..8192).await.unwrap().len(), 8192); ++ drop(wrapped); ++ assert!( ++ weak.upgrade().is_none(), ++ "the namespace registry must not retain stores" ++ ); ++ cache.cache.close().await.unwrap(); ++ drop(cache); ++ ++ let recovered = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let replacement = Arc::new(InMemory::new()); ++ replacement ++ .put(&path, Bytes::from(vec![29; 8192]).into()) ++ .await ++ .unwrap(); ++ let (wrapped, statistics) = wrap_for_test(&recovered, replacement); ++ assert_eq!( ++ wrapped ++ .get_ranges(&path, &[0..4096, 4096..8192]) ++ .await ++ .unwrap(), ++ vec![Bytes::from(vec![29; 4096]); 2] ++ ); ++ assert_eq!(statistics.snapshot().bytes_read_from_cache, 0); ++ assert_eq!(statistics.snapshot().bytes_read_from_remote, 8192); ++ } ++ ++ #[tokio::test] + async fn dataset_scopes_share_cache_without_sharing_statistics() { + let directory = tempfile::tempdir().unwrap(); + let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 64 * 1024) @@ -1894,12 +2134,12 @@ index 0000000..9a10e43 + } +} diff --git a/src/lib.rs b/src/lib.rs -index 8b212f5..923528a 100644 +index 34817600548d8d68c0081597d2f0ea9a41efc8d8..4a96bb58652b467d3ff278596b4f4fd95eea97fb 100644 --- a/src/lib.rs +++ b/src/lib.rs -@@ -25,11 +25,13 @@ mod alter_columns; - mod async_dispatcher; +@@ -26,11 +26,13 @@ mod async_dispatcher; mod batch; + mod blob; mod compact; +mod data_cache; mod data_statistics; @@ -1911,15 +2151,15 @@ index 8b212f5..923528a 100644 mod fragment_writer; mod fts_query; mod helpers; -@@ -51,6 +53,7 @@ pub use add_columns::*; - pub use alter_columns::*; +@@ -55,6 +57,7 @@ pub use alter_columns::*; pub use batch::*; + pub use blob::*; pub use compact::*; +pub use data_cache::{LanceDataCacheStatistics, lance_dataset_get_data_cache_statistics}; pub use data_statistics::*; pub use dataset::*; pub use delete::*; -@@ -58,6 +61,7 @@ pub use drop_columns::*; +@@ -62,6 +65,7 @@ pub use drop_columns::*; pub use error::{ LanceErrorCode, lance_free_string, lance_last_error_code, lance_last_error_message, }; @@ -1928,13 +2168,13 @@ index 8b212f5..923528a 100644 pub use fts_query::*; pub use index::*; diff --git a/src/restore.rs b/src/restore.rs -index 7804b55..fa2d26d 100644 +index 7804b55818c3b0ed2f14de5cc9f632098e0eb25e..fa2d26da5670c78224a3e33f98f747b24190b6d6 100644 --- a/src/restore.rs +++ b/src/restore.rs @@ -65,8 +65,16 @@ unsafe fn restore_inner(dataset: *const LanceDataset, version: u64) -> Result<*m Ok::<_, lance_core::Error>(checked_out) })?; - + + let (restored, data_cache) = if let Some(data_cache) = &ds.data_cache { + let (restored, data_cache) = data_cache.attach_fresh(restored); + (restored, Some(data_cache)) @@ -1949,24 +2189,24 @@ index 7804b55..fa2d26d 100644 Ok(Box::into_raw(Box::new(handle))) } diff --git a/src/session.rs b/src/session.rs -index 60a1623..9ed8cfe 100644 +index 60a16234557eaeb93e8eaad971a9eba73a562a83..9ed8cfe4b5207afc78c65163e8cc4c7dc1faab09 100644 --- a/src/session.rs +++ b/src/session.rs @@ -8,12 +8,14 @@ use std::sync::Arc; use lance::session::Session; use lance_core::Result; - + +use crate::data_cache::DataCacheFactory; use crate::error::{ffi_try, swallow_unwind}; use crate::runtime::block_on; - + -/// Opaque handle for sharing Lance metadata and index caches across datasets. +/// Opaque handle for shared Lance caches across datasets. pub struct LanceSession { pub(crate) inner: Arc, + pub(crate) data_cache_factory: Option>, } - + /// Snapshot of a session's metadata and index cache statistics. @@ -47,6 +49,14 @@ pub extern "C" fn lance_session_new( fn session_new_inner( @@ -1990,9 +2230,9 @@ index 60a1623..9ed8cfe 100644 + data_cache_factory, }))) } - + diff --git a/src/writer.rs b/src/writer.rs -index 1971510..ba51c87 100644 +index 1971510de4a13ee6f04fc39c159b088e78e21d55..ba51c87a8cebb6897c48fb55509f71dabb58f3b4 100644 --- a/src/writer.rs +++ b/src/writer.rs @@ -282,6 +282,7 @@ unsafe fn write_dataset_inner( @@ -2004,13 +2244,13 @@ index 1971510..ba51c87 100644 // SAFETY: `out_dataset` is non-NULL (checked above) and the caller // guarantees it points to caller-owned, writable storage of size diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index bde742d..a4ea8f8 100644 +index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..c341859e9434d4578b2833eca6ec313a28f31f32 100644 --- a/tests/c_api_test.rs +++ b/tests/c_api_test.rs -@@ -96,10 +96,83 @@ fn create_large_dataset(num_rows: i32) -> (tempfile::TempDir, String) { +@@ -100,10 +100,83 @@ fn create_large_dataset(num_rows: i32) -> (tempfile::TempDir, String) { (tmp, uri) } - + +/// Helper: create two fragments large enough for Lance's batched range-read +/// path, which is the path wrapped by the Foyer data cache. +fn create_large_multi_fragment_dataset(num_rows_per_fragment: i32) -> (tempfile::TempDir, String) { @@ -2051,7 +2291,7 @@ index bde742d..a4ea8f8 100644 fn c_str(s: &str) -> CString { CString::new(s).unwrap() } - + +fn file_object_store_uri(path: &str) -> CString { + let path = path.replace('\\', "/"); + let leading_slash = if path.starts_with('/') { "" } else { "/" }; @@ -2091,10 +2331,10 @@ index bde742d..a4ea8f8 100644 #[derive(Default)] struct CapturedScanStatistics { calls: usize, -@@ -313,6 +386,120 @@ fn test_shared_session_rejects_null_inputs() { +@@ -445,6 +518,129 @@ fn test_shared_session_rejects_null_inputs() { } } - + +#[test] +fn test_session_with_data_cache_serves_repeated_scan() { + let (tmp, uri) = create_large_multi_fragment_dataset(10_000); @@ -2116,6 +2356,12 @@ index bde742d..a4ea8f8 100644 + !cached_dataset.is_null(), + "second dataset open should succeed" + ); ++ // A fresh open may create a different underlying store. Warm its isolated ++ // namespace before removing origin files; only the same live store can reuse ++ // entries because bucket/path alone does not identify a storage backend. ++ assert_eq!(scanned_row_count(cached_dataset), 20_000); ++ let reopened_statistics = data_cache_statistics(cached_dataset); ++ assert!(reopened_statistics.bytes_read_from_remote > 0); + unsafe { lance_session_close(session) }; + + for entry in std::fs::read_dir(tmp.path().join("large_ds/data")).unwrap() { @@ -2123,8 +2369,11 @@ index bde742d..a4ea8f8 100644 + } + assert_eq!(scanned_row_count(cached_dataset), 20_000); + let cached_statistics = data_cache_statistics(cached_dataset); -+ assert!(cached_statistics.bytes_read_from_cache > 0); -+ assert_eq!(cached_statistics.bytes_read_from_remote, 0); ++ assert!(cached_statistics.bytes_read_from_cache > reopened_statistics.bytes_read_from_cache); ++ assert_eq!( ++ cached_statistics.bytes_read_from_remote, ++ reopened_statistics.bytes_read_from_remote ++ ); + + assert_eq!(data_cache_statistics(dataset), first_statistics); + @@ -2212,10 +2461,10 @@ index bde742d..a4ea8f8 100644 #[test] fn test_open_nonexistent() { let c_uri = c_str("memory://nonexistent_dataset_xyz"); -@@ -2977,6 +3164,40 @@ fn test_dataset_restore_to_prior_version() { +@@ -3318,6 +3514,40 @@ fn test_dataset_restore_to_prior_version() { unsafe { lance_dataset_close(ds) }; } - + +#[test] +fn test_restored_handle_has_independent_data_cache_statistics() { + let (_tmp, uri) = create_large_multi_fragment_dataset(10_000); @@ -2254,13 +2503,13 @@ index bde742d..a4ea8f8 100644 fn test_dataset_restore_to_current_latest_writes_new_manifest() { // Restoring to the current latest still writes a new manifest. The diff --git a/tests/cpp/test_c_api.c b/tests/cpp/test_c_api.c -index c49ecfa..dd674eb 100644 +index 5df9cde4d8fa971823b9e6c974a8a0db6c9ec9ca..1444e55161631ac8ad976f427d867ad91d358195 100644 --- a/tests/cpp/test_c_api.c +++ b/tests/cpp/test_c_api.c -@@ -126,6 +126,37 @@ static void test_shared_session(const char *uri) { +@@ -171,6 +171,37 @@ static void test_shared_session(const char *uri) { (unsigned long long)stats.metadata_cache_entries); } - + +static void test_data_cache_session(const char *uri, const char *write_uri) { + printf(" test_data_cache_session... "); + @@ -2294,23 +2543,23 @@ index c49ecfa..dd674eb 100644 + static void test_scan(const char *uri) { printf(" test_scan... "); - -@@ -973,6 +1004,7 @@ int main(int argc, char **argv) { - + +@@ -1323,6 +1354,7 @@ int main(int argc, char **argv) { + test_open_and_metadata(uri); test_shared_session(uri); + test_data_cache_session(uri, write_uri); test_scan(uri); test_scan_with_limit(uri); - test_versions(uri); + test_scanner_blob_handling(blob_uri); diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index 28762b8..5f34bde 100644 +index 0332d1a2687354984d0551e10716df08344a4013..ef64ab0d02bed2f1750c3d388792e4aa44265e24 100644 --- a/tests/cpp/test_cpp_api.cpp +++ b/tests/cpp/test_cpp_api.cpp -@@ -99,6 +99,28 @@ static void test_shared_session(const std::string& uri) { +@@ -131,6 +131,28 @@ static void test_shared_session(const std::string& uri) { PASS(); } - + +static void test_data_cache_session(const std::string& uri, + const std::string& write_uri) { + TEST(test_data_cache_session); @@ -2335,9 +2584,9 @@ index 28762b8..5f34bde 100644 + static void test_dataset_schema(const std::string& uri) { TEST(test_dataset_schema); - -@@ -929,6 +951,7 @@ int main(int argc, char** argv) { - + +@@ -1211,6 +1233,7 @@ int main(int argc, char** argv) { + test_dataset_open(uri); test_shared_session(uri); + test_data_cache_session(uri, write_uri); diff --git a/thirdparty/test/lance-prefilter-patch-test.sh b/thirdparty/test/lance-prefilter-patch-test.sh new file mode 100755 index 00000000000000..e3dabff6c176d6 --- /dev/null +++ b/thirdparty/test/lance-prefilter-patch-test.sh @@ -0,0 +1,124 @@ +#!/usr/bin/env bash +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." &>/dev/null && pwd)" +ARCHIVE_DIR="${1:?Usage: $0 directory-containing-the-pinned-lance-c-archive}" +ARCHIVE_DIR="$(cd "${ARCHIVE_DIR}" && pwd)" +# Platform definitions must remain safe when this harness enables nounset. +for test_arch in x86_64 arm64; do + bash -eu -c ' + uname() { if [[ "$1" == -s ]]; then echo Darwin; else echo "$TEST_ARCH"; fi; } + unset ARROW_ADBC_FLIGHTSQL_SOURCE + TP_DIR="$1" + TEST_ARCH="$2" + source "$TP_DIR/vars.sh" + [[ " ${TP_ARCHIVES[*]} " != *" ARROW_ADBC_FLIGHTSQL "* ]] + ' _ "${ROOT}" "${test_arch}" +done +echo "PASS: macOS platform definitions under nounset" + +TP_DIR="${ROOT}" +# Load only repository-owned definitions, never extracted dependency code. +source "${ROOT}/vars.sh" +mkdir -p "${ROOT}/src" +tmpdir="$(mktemp -d "${ROOT}/src/lance-prefilter-test.XXXXXX")" +trap 'rm -rf "${tmpdir}"' EXIT + +fail() { + echo "FAIL: $*" >&2 + exit 1 +} + +prepare() { + local dest="$1" + mkdir -p "${dest}/src" + cp "${ROOT}/vars.sh" "${dest}/vars.sh" + ln -s "${ROOT}/patches" "${dest}/patches" + cp "${ARCHIVE_DIR}/${LANCE_C_NAME}" "${dest}/src/" +} + +run_download() { + TP_DIR="$1" DORIS_HOME="${tmpdir}" bash "${ROOT}/download-thirdparty.sh" lance_c +} + +check_sources() { + local source="$1/src/${LANCE_C_SOURCE}" + grep -q 'fn test_scanner_nearest_segment_prefilter_statistics' "${source}/tests/c_api_test.rs" \ + || fail "missing upstream segment-prefilter regression" + grep -q 'lance_session_new_with_data_cache' "${source}/src/foyer_data_cache.rs" \ + || fail "missing retained Foyer API" + grep -q 'lance_dataset_get_data_cache_statistics' "${source}/src/data_cache.rs" \ + || fail "missing retained cache statistics API" + grep -q 'source = "git+https://github.com/lance-format/lance.git' "${source}/Cargo.lock" \ + || fail "Lance must come from the upstream git dependency" + [[ -f "${source}/patched_mark_foyer" ]] || fail "missing Foyer patch marker" +} + +prepare "${tmpdir}/fresh" +run_download "${tmpdir}/fresh" +check_sources "${tmpdir}/fresh" +cp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.toml" "${tmpdir}/manifest" +cp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" "${tmpdir}/lock" +touch "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/cache_reuse_sentinel" +run_download "${tmpdir}/fresh" +[[ -f "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/cache_reuse_sentinel" ]] \ + || fail "unchanged patch unnecessarily replaced cached sources" +cmp "${tmpdir}/manifest" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.toml" +cmp "${tmpdir}/lock" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" +echo "PASS: fresh archive and idempotent Foyer patch" + +# Re-extraction must apply Foyer again; no separate Lance source is required. +rm -rf "${tmpdir}/fresh/src/${LANCE_C_SOURCE}" +run_download "${tmpdir}/fresh" +check_sources "${tmpdir}/fresh" +cmp "${tmpdir}/lock" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" +echo "PASS: re-extracted archive" + +prepare "${tmpdir}/cached" +tar xzf "${ARCHIVE_DIR}/${LANCE_C_NAME}" -C "${tmpdir}/cached/src" +# A generic marker from an earlier build must not suppress the Foyer patch. +touch "${tmpdir}/cached/src/${LANCE_C_SOURCE}/patched_mark" +run_download "${tmpdir}/cached" +check_sources "${tmpdir}/cached" +cmp "${tmpdir}/lock" "${tmpdir}/cached/src/${LANCE_C_SOURCE}/Cargo.lock" +echo "PASS: cached sources with an existing generic marker" + +# A cached source tree may already contain the previous Foyer patch. The patch +# fingerprint must invalidate it even when the upstream archive is unchanged. +cached_source="${tmpdir}/cached/src/${LANCE_C_SOURCE}" +for stale_marker in '' '0 0'; do + printf '%s\n' 'stale patch contents' > "${cached_source}/src/foyer_data_cache.rs" + printf '%s\n' "${stale_marker}" > "${cached_source}/patched_mark_foyer" + run_download "${tmpdir}/cached" + check_sources "${tmpdir}/cached" + cmp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/src/foyer_data_cache.rs" \ + "${cached_source}/src/foyer_data_cache.rs" +done +echo "PASS: old and mismatched Foyer markers refresh cached sources" + +prepare "${tmpdir}/invalid" +tar xzf "${ARCHIVE_DIR}/${LANCE_C_NAME}" -C "${tmpdir}/invalid/src" +printf '%s\n' 'incompatible manifest' > "${tmpdir}/invalid/src/${LANCE_C_SOURCE}/Cargo.toml" +if run_download "${tmpdir}/invalid" > "${tmpdir}/invalid.log" 2>&1; then + fail "expected an incompatible source to reject the patch" +fi +[[ ! -f "${tmpdir}/invalid/src/${LANCE_C_SOURCE}/patched_mark_foyer" ]] \ + || fail "failed patch was marked complete" +echo "PASS: patch failure does not mark sources ready" diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index f86c204d579350..506017449e7541 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -580,10 +580,11 @@ PUGIXML_SOURCE=pugixml-1.15 PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c -LANCE_C_DOWNLOAD="https://github.com/lance-format/lance-c/archive/refs/tags/v0.1.9.tar.gz" -LANCE_C_NAME="lance-c-v0.1.9.tar.gz" -LANCE_C_SOURCE="lance-c-0.1.9" -LANCE_C_MD5SUM="7138ed44e92d4bc91d5b522a6b92ed64" +# Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/9bd730add2ac70316c1d642b8459011e2dd92022" +LANCE_C_NAME="lance-c-9bd730add2ac70316c1d642b8459011e2dd92022.tar.gz" +LANCE_C_SOURCE="lance-c-9bd730add2ac70316c1d642b8459011e2dd92022" +LANCE_C_MD5SUM="63851b09bf1689032579f1a094ff2f37" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=(